From 5ea1d55e566bc3dbd9782cdcac655f4f8d198c41 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 26 Apr 2026 08:19:02 +0200 Subject: [PATCH] feat: removed chunk-mode modules and read/edit entrypoints from pi-natives - Removed `pi-natives` chunk language classifier modules and all core chunk subsystems (kind, state, render, edit, resolve). - Removed chunk-mode CLI/read/edit entrypoints, including `read` command and chunk mode registration/prompt tooling. - Removed chunk selectors from `read` and `grep` tools, switching behavior to raw/L-range handling. - Fixed poll wait parsing to keep defaulting to `30s` when the provided value is empty. --- crates/pi-natives/build.rs | 829 --- crates/pi-natives/src/chunk/ast_astro.rs | 231 - .../src/chunk/ast_bash_make_diff.rs | 275 - crates/pi-natives/src/chunk/ast_c_cpp_objc.rs | 431 -- crates/pi-natives/src/chunk/ast_clojure.rs | 78 - crates/pi-natives/src/chunk/ast_cmake.rs | 174 - .../pi-natives/src/chunk/ast_csharp_java.rs | 372 - crates/pi-natives/src/chunk/ast_css.rs | 124 - .../pi-natives/src/chunk/ast_data_formats.rs | 278 - crates/pi-natives/src/chunk/ast_dockerfile.rs | 184 - crates/pi-natives/src/chunk/ast_elixir.rs | 159 - crates/pi-natives/src/chunk/ast_erlang.rs | 242 - crates/pi-natives/src/chunk/ast_go.rs | 320 - crates/pi-natives/src/chunk/ast_graphql.rs | 294 - .../pi-natives/src/chunk/ast_haskell_scala.rs | 195 - crates/pi-natives/src/chunk/ast_html_xml.rs | 163 - crates/pi-natives/src/chunk/ast_ini.rs | 88 - crates/pi-natives/src/chunk/ast_ipynb.rs | 918 --- crates/pi-natives/src/chunk/ast_js_ts.rs | 564 -- crates/pi-natives/src/chunk/ast_just.rs | 129 - crates/pi-natives/src/chunk/ast_markup.rs | 341 - crates/pi-natives/src/chunk/ast_misc.rs | 1163 --- crates/pi-natives/src/chunk/ast_nix_hcl.rs | 228 - crates/pi-natives/src/chunk/ast_ocaml.rs | 236 - crates/pi-natives/src/chunk/ast_perl.rs | 137 - crates/pi-natives/src/chunk/ast_powershell.rs | 290 - crates/pi-natives/src/chunk/ast_proto.rs | 193 - crates/pi-natives/src/chunk/ast_python.rs | 244 - crates/pi-natives/src/chunk/ast_r.rs | 173 - crates/pi-natives/src/chunk/ast_ruby_lua.rs | 253 - crates/pi-natives/src/chunk/ast_rust.rs | 404 -- crates/pi-natives/src/chunk/ast_sql.rs | 282 - crates/pi-natives/src/chunk/ast_svelte.rs | 299 - crates/pi-natives/src/chunk/ast_tlaplus.rs | 387 - crates/pi-natives/src/chunk/ast_vue.rs | 318 - crates/pi-natives/src/chunk/atom_list.rs | 145 - crates/pi-natives/src/chunk/classify.rs | 509 -- crates/pi-natives/src/chunk/common.rs | 903 --- crates/pi-natives/src/chunk/conflict.rs | 690 -- crates/pi-natives/src/chunk/defaults.rs | 85 - crates/pi-natives/src/chunk/edit.rs | 6223 ----------------- crates/pi-natives/src/chunk/indent.rs | 618 -- crates/pi-natives/src/chunk/kind.rs | 699 -- crates/pi-natives/src/chunk/mod.rs | 3443 --------- crates/pi-natives/src/chunk/render.rs | 1299 ---- crates/pi-natives/src/chunk/resolve.rs | 1257 ---- crates/pi-natives/src/chunk/schema.rs | 130 - crates/pi-natives/src/chunk/shape.rs | 354 - crates/pi-natives/src/chunk/state.rs | 689 -- crates/pi-natives/src/chunk/types.rs | 403 -- crates/pi-natives/src/lib.rs | 1 - packages/coding-agent/CHANGELOG.md | 9 + packages/coding-agent/src/cli.ts | 1 - packages/coding-agent/src/cli/read-cli.ts | 67 - packages/coding-agent/src/commands/read.ts | 33 - .../src/config/prompt-templates.ts | 30 - .../src/config/settings-schema.ts | 36 +- packages/coding-agent/src/config/settings.ts | 2 +- packages/coding-agent/src/edit/index.ts | 46 - packages/coding-agent/src/edit/line-hash.ts | 53 - packages/coding-agent/src/edit/modes/chunk.ts | 832 --- packages/coding-agent/src/edit/renderer.ts | 40 +- packages/coding-agent/src/edit/streaming.ts | 66 - .../src/modes/components/settings-defs.ts | 5 - .../src/modes/components/tool-execution.ts | 10 +- .../src/prompts/tools/chunk-edit.md | 158 - .../coding-agent/src/prompts/tools/grep.md | 5 +- .../coding-agent/src/prompts/tools/poll.md | 2 +- .../src/prompts/tools/read-chunk.md | 73 - .../coding-agent/src/prompts/tools/read.md | 2 +- .../src/prompts/tools/todo-write.md | 4 +- .../src/tools/fs-cache-invalidation.ts | 5 - packages/coding-agent/src/tools/grep.ts | 123 +- packages/coding-agent/src/tools/poll-tool.ts | 2 +- packages/coding-agent/src/tools/read.ts | 125 +- packages/coding-agent/src/utils/edit-mode.ts | 3 +- .../src/utils/file-display-mode.ts | 3 - .../coding-agent/test/cli/read-cli.test.ts | 33 - packages/coding-agent/test/edit-diff.test.ts | 125 - .../test/edit-streaming-preview.test.ts | 39 - packages/natives/CHANGELOG.md | 4 + packages/natives/native/index.d.ts | 311 - packages/natives/native/index.js | 31 - .../typescript-edit-benchmark/src/index.ts | 2 +- .../typescript-edit-benchmark/src/report.ts | 20 +- .../typescript-edit-benchmark/src/runner.ts | 47 - scripts/edit-benchmark.py | 9 +- scripts/rate-edit-tool.py | 2 +- 88 files changed, 49 insertions(+), 30753 deletions(-) delete mode 100644 crates/pi-natives/src/chunk/ast_astro.rs delete mode 100644 crates/pi-natives/src/chunk/ast_bash_make_diff.rs delete mode 100644 crates/pi-natives/src/chunk/ast_c_cpp_objc.rs delete mode 100644 crates/pi-natives/src/chunk/ast_clojure.rs delete mode 100644 crates/pi-natives/src/chunk/ast_cmake.rs delete mode 100644 crates/pi-natives/src/chunk/ast_csharp_java.rs delete mode 100644 crates/pi-natives/src/chunk/ast_css.rs delete mode 100644 crates/pi-natives/src/chunk/ast_data_formats.rs delete mode 100644 crates/pi-natives/src/chunk/ast_dockerfile.rs delete mode 100644 crates/pi-natives/src/chunk/ast_elixir.rs delete mode 100644 crates/pi-natives/src/chunk/ast_erlang.rs delete mode 100644 crates/pi-natives/src/chunk/ast_go.rs delete mode 100644 crates/pi-natives/src/chunk/ast_graphql.rs delete mode 100644 crates/pi-natives/src/chunk/ast_haskell_scala.rs delete mode 100644 crates/pi-natives/src/chunk/ast_html_xml.rs delete mode 100644 crates/pi-natives/src/chunk/ast_ini.rs delete mode 100644 crates/pi-natives/src/chunk/ast_ipynb.rs delete mode 100644 crates/pi-natives/src/chunk/ast_js_ts.rs delete mode 100644 crates/pi-natives/src/chunk/ast_just.rs delete mode 100644 crates/pi-natives/src/chunk/ast_markup.rs delete mode 100644 crates/pi-natives/src/chunk/ast_misc.rs delete mode 100644 crates/pi-natives/src/chunk/ast_nix_hcl.rs delete mode 100644 crates/pi-natives/src/chunk/ast_ocaml.rs delete mode 100644 crates/pi-natives/src/chunk/ast_perl.rs delete mode 100644 crates/pi-natives/src/chunk/ast_powershell.rs delete mode 100644 crates/pi-natives/src/chunk/ast_proto.rs delete mode 100644 crates/pi-natives/src/chunk/ast_python.rs delete mode 100644 crates/pi-natives/src/chunk/ast_r.rs delete mode 100644 crates/pi-natives/src/chunk/ast_ruby_lua.rs delete mode 100644 crates/pi-natives/src/chunk/ast_rust.rs delete mode 100644 crates/pi-natives/src/chunk/ast_sql.rs delete mode 100644 crates/pi-natives/src/chunk/ast_svelte.rs delete mode 100644 crates/pi-natives/src/chunk/ast_tlaplus.rs delete mode 100644 crates/pi-natives/src/chunk/ast_vue.rs delete mode 100644 crates/pi-natives/src/chunk/atom_list.rs delete mode 100644 crates/pi-natives/src/chunk/classify.rs delete mode 100644 crates/pi-natives/src/chunk/common.rs delete mode 100644 crates/pi-natives/src/chunk/conflict.rs delete mode 100644 crates/pi-natives/src/chunk/defaults.rs delete mode 100644 crates/pi-natives/src/chunk/edit.rs delete mode 100644 crates/pi-natives/src/chunk/indent.rs delete mode 100644 crates/pi-natives/src/chunk/kind.rs delete mode 100644 crates/pi-natives/src/chunk/mod.rs delete mode 100644 crates/pi-natives/src/chunk/render.rs delete mode 100644 crates/pi-natives/src/chunk/resolve.rs delete mode 100644 crates/pi-natives/src/chunk/schema.rs delete mode 100644 crates/pi-natives/src/chunk/shape.rs delete mode 100644 crates/pi-natives/src/chunk/state.rs delete mode 100644 crates/pi-natives/src/chunk/types.rs delete mode 100644 packages/coding-agent/src/cli/read-cli.ts delete mode 100644 packages/coding-agent/src/commands/read.ts delete mode 100644 packages/coding-agent/src/edit/modes/chunk.ts delete mode 100644 packages/coding-agent/src/prompts/tools/chunk-edit.md delete mode 100644 packages/coding-agent/src/prompts/tools/read-chunk.md delete mode 100644 packages/coding-agent/test/cli/read-cli.test.ts diff --git a/crates/pi-natives/build.rs b/crates/pi-natives/build.rs index ae31592ac..9078f4319 100644 --- a/crates/pi-natives/build.rs +++ b/crates/pi-natives/build.rs @@ -1,373 +1,12 @@ use std::{ - collections::{BTreeMap, BTreeSet, HashMap}, env, fmt::Write as _, fs, path::{Path, PathBuf}, }; -use serde::Deserialize; - -const IDENTIFIER_FIELD_PRIORITY: &[&str] = &[ - "name", - "identifier", - "attrpath", - "key", - "label", - "alias", - "field", - "member", - "property", - "tag", - "target", - "variable", -]; - -const BODY_FIELD_PRIORITY: &[&str] = &["body", "value", "declaration_list", "block", "members"]; -const PROMOTION_FIELD_PRIORITY: &[&str] = &["definition", "declaration", "item", "member"]; - -#[derive(Clone, Copy)] -struct GrammarSpec { - language: &'static str, - package: &'static str, - node_types_rel: &'static str, -} - -struct LockedPackage { - version: String, - source: Option, -} - -#[derive(Deserialize)] -struct RawTypeRef { - #[serde(rename = "type")] - kind: Option, - named: bool, -} - -#[derive(Deserialize)] -struct RawFieldSpec { - #[serde(default)] - multiple: bool, - #[serde(default)] - types: Vec, -} - -#[derive(Deserialize)] -struct RawNodeType { - #[serde(rename = "type")] - kind: Option, - fields: Option>, - children: Option, - subtypes: Option>, -} - -#[derive(serde::Serialize)] -struct GeneratedSchema { - languages: BTreeMap>, -} - -#[derive(serde::Serialize)] -struct GeneratedNodeTypeSchema { - identifier_fields: Vec, - body_fields: Vec, - promotion_fields: Vec, - container_child_kinds: Vec, - is_supertype: bool, - has_structural_children: bool, -} - -const GRAMMARS: &[GrammarSpec] = &[ - GrammarSpec { - language: "astro", - package: "tree-sitter-astro-next", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "bash", - package: "tree-sitter-bash", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "c", - package: "tree-sitter-c", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "clojure", - package: "tree-sitter-clojure", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "cmake", - package: "tree-sitter-cmake", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "cpp", - package: "tree-sitter-cpp", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "csharp", - package: "tree-sitter-c-sharp", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "dart", - package: "tree-sitter-dart", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "css", - package: "tree-sitter-css", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "diff", - package: "tree-sitter-diff", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "dockerfile", - package: "tree-sitter-dockerfile-updated", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "elixir", - package: "tree-sitter-elixir", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "erlang", - package: "tree-sitter-erlang", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "go", - package: "tree-sitter-go", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "graphql", - package: "tree-sitter-graphql", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "handlebars", - package: "tree-sitter-glimmer", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "haskell", - package: "tree-sitter-haskell", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "hcl", - package: "tree-sitter-hcl", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "html", - package: "tree-sitter-html", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "ini", - package: "tree-sitter-ini", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "java", - package: "tree-sitter-java", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "javascript", - package: "tree-sitter-javascript", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "json", - package: "tree-sitter-json", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "toml", - package: "tree-sitter-toml-ng", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "just", - package: "tree-sitter-just", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "julia", - package: "tree-sitter-julia", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "kotlin", - package: "tree-sitter-kotlin-sg", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "lua", - package: "tree-sitter-lua", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "make", - package: "tree-sitter-make", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "markdown", - package: "tree-sitter-md", - node_types_rel: "tree-sitter-markdown/src/node-types.json", - }, - GrammarSpec { - language: "nix", - package: "tree-sitter-nix", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "objc", - package: "tree-sitter-objc", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "odin", - package: "tree-sitter-odin", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "perl", - package: "tree-sitter-perl-next", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "php", - package: "tree-sitter-php", - node_types_rel: "php/src/node-types.json", - }, - GrammarSpec { - language: "powershell", - package: "tree-sitter-powershell", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "protobuf", - package: "tree-sitter-proto", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "python", - package: "tree-sitter-python", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "r", - package: "tree-sitter-r", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "regex", - package: "tree-sitter-regex", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "ruby", - package: "tree-sitter-ruby", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "rust", - package: "tree-sitter-rust", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "scala", - package: "tree-sitter-scala", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "solidity", - package: "tree-sitter-solidity", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "sql", - package: "tree-sitter-sequel", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "starlark", - package: "tree-sitter-starlark", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "svelte", - package: "tree-sitter-svelte-next", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "swift", - package: "tree-sitter-swift", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "toml", - package: "tree-sitter-toml-ng", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "tlaplus", - package: "tree-sitter-tlaplus", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "tsx", - package: "tree-sitter-typescript", - node_types_rel: "tsx/src/node-types.json", - }, - GrammarSpec { - language: "typescript", - package: "tree-sitter-typescript", - node_types_rel: "typescript/src/node-types.json", - }, - GrammarSpec { - language: "verilog", - package: "tree-sitter-verilog", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "vue", - package: "tree-sitter-vue-next", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "xml", - package: "tree-sitter-xml", - node_types_rel: "xml/src/node-types.json", - }, - GrammarSpec { - language: "yaml", - package: "tree-sitter-yaml", - node_types_rel: "src/node-types.json", - }, - GrammarSpec { - language: "zig", - package: "tree-sitter-zig", - node_types_rel: "src/node-types.json", - }, -]; - fn main() { napi_build::setup(); - generate_chunk_schema(); generate_minimizer_builtin_filters(); } @@ -423,471 +62,3 @@ fn generate_minimizer_builtin_filters() { fs::write(&output_path, concatenated) .unwrap_or_else(|e| panic!("failed to write {}: {e}", output_path.display())); } - -fn generate_chunk_schema() { - let manifest_dir = env::var("CARGO_MANIFEST_DIR").expect("CARGO_MANIFEST_DIR should be set"); - let workspace_root = Path::new(&manifest_dir) - .parent() - .and_then(Path::parent) - .expect("pi-natives should live under the workspace root"); - let out_dir = PathBuf::from(env::var("OUT_DIR").expect("OUT_DIR should be set")); - let output_path = out_dir.join("chunk_schema.json"); - let locked_packages = locked_packages(&workspace_root.join("Cargo.lock")); - let registry_roots = cargo_registry_roots(); - let git_roots = cargo_git_checkout_roots(); - let mut languages = BTreeMap::new(); - - for grammar in GRAMMARS { - let Some(locked) = locked_packages.get(grammar.package) else { - continue; - }; - let Some(package_dir) = - find_locked_package_dir(®istry_roots, &git_roots, grammar.package, locked) - else { - continue; - }; - let node_types_path = package_dir.join(grammar.node_types_rel); - if !node_types_path.exists() { - continue; - } - - println!("cargo:rerun-if-changed={}", node_types_path.display()); - let source = - fs::read_to_string(&node_types_path).expect("node-types.json should be readable"); - let raw_nodes: Vec = - serde_json::from_str(&source).expect("node-types.json should parse"); - let schemas = build_language_schema(raw_nodes); - if !schemas.is_empty() { - languages.insert(grammar.language.to_string(), schemas); - } - } - - let generated = GeneratedSchema { languages }; - let json = serde_json::to_string(&generated).expect("schema JSON should serialize"); - fs::write(output_path, json).expect("schema JSON should write"); -} - -fn build_language_schema(raw_nodes: Vec) -> BTreeMap { - let mut raw_by_kind = HashMap::new(); - for raw in raw_nodes { - let Some(kind) = raw.kind.clone() else { - continue; - }; - raw_by_kind.insert(kind, raw); - } - - let structural_state = compute_structural_state(&raw_by_kind); - let mut out = BTreeMap::new(); - for (kind, raw) in &raw_by_kind { - let identifier_fields = pick_priority_fields(raw.fields.as_ref(), IDENTIFIER_FIELD_PRIORITY); - let body_fields = pick_priority_fields(raw.fields.as_ref(), BODY_FIELD_PRIORITY); - let promotion_fields = - collect_promotion_fields(raw, &structural_state, &identifier_fields, &body_fields); - let container_child_kinds = collect_child_container_kinds(raw, &structural_state); - let is_supertype = is_supertype(raw); - let has_structural_children = structural_state - .get(kind) - .is_some_and(|state| state.has_structural_children); - - if identifier_fields.is_empty() - && body_fields.is_empty() - && promotion_fields.is_empty() - && container_child_kinds.is_empty() - && !is_supertype - && !has_structural_children - { - continue; - } - - out.insert(kind.clone(), GeneratedNodeTypeSchema { - identifier_fields, - body_fields, - promotion_fields, - container_child_kinds, - is_supertype, - has_structural_children, - }); - } - - out -} - -fn pick_priority_fields( - fields: Option<&BTreeMap>, - priority: &[&str], -) -> Vec { - let Some(fields) = fields else { - return Vec::new(); - }; - - priority - .iter() - .filter(|field| fields.contains_key(**field)) - .map(|field| (*field).to_string()) - .collect() -} - -fn collect_child_container_kinds( - raw: &RawNodeType, - structural_state: &HashMap, -) -> Vec { - let mut kinds = BTreeSet::new(); - let child_types = raw - .children - .as_ref() - .map(|children| children.types.as_slice()) - .unwrap_or_default(); - - for child in child_types { - if !child.named { - continue; - } - let Some(kind) = child.kind.as_deref() else { - continue; - }; - if structural_state - .get(kind) - .copied() - .is_some_and(StructuralState::is_structural) - { - kinds.insert(kind.to_string()); - } - } - - kinds.into_iter().collect() -} - -fn collect_promotion_fields( - raw: &RawNodeType, - structural_state: &HashMap, - identifier_fields: &[String], - body_fields: &[String], -) -> Vec { - let Some(fields) = raw.fields.as_ref() else { - return Vec::new(); - }; - - PROMOTION_FIELD_PRIORITY - .iter() - .filter_map(|field_name| { - let spec = fields.get(*field_name)?; - if spec.multiple - || identifier_fields.iter().any(|field| field == field_name) - || body_fields.iter().any(|field| field == field_name) - { - return None; - } - - let has_structural_type = spec.types.iter().any(|field_type| { - field_type.named - && field_type - .kind - .as_deref() - .and_then(|kind| structural_state.get(kind)) - .copied() - .is_some_and(StructuralState::is_structural) - }); - has_structural_type.then(|| (*field_name).to_string()) - }) - .collect() -} - -#[derive(Clone, Copy, Default)] -struct StructuralState { - is_structural: bool, - has_structural_children: bool, -} - -impl StructuralState { - const fn is_structural(self) -> bool { - self.is_structural - } -} - -fn compute_structural_state( - raw_by_kind: &HashMap, -) -> HashMap { - let mut state = raw_by_kind - .iter() - .map(|(kind, raw)| { - let base_structural = is_supertype(raw) - || raw.fields.as_ref().is_some_and(|fields| !fields.is_empty()) - || !named_child_type_kinds(raw).is_empty(); - (kind.clone(), StructuralState { - is_structural: base_structural, - has_structural_children: false, - }) - }) - .collect::>(); - - loop { - let mut changed = false; - for (kind, raw) in raw_by_kind { - let next_has_structural_children = - named_child_type_kinds(raw).into_iter().any(|child_kind| { - state - .get(child_kind.as_str()) - .copied() - .is_some_and(StructuralState::is_structural) - }); - - let entry = state - .get_mut(kind.as_str()) - .expect("every raw node should have structural state"); - let next_is_structural = entry.is_structural || next_has_structural_children; - if next_is_structural != entry.is_structural - || next_has_structural_children != entry.has_structural_children - { - entry.is_structural = next_is_structural; - entry.has_structural_children = next_has_structural_children; - changed = true; - } - } - if !changed { - break; - } - } - - state -} - -fn is_supertype(raw: &RawNodeType) -> bool { - raw.subtypes - .as_ref() - .is_some_and(|subtypes| !subtypes.is_empty()) -} - -fn named_child_type_kinds(raw: &RawNodeType) -> BTreeSet { - let mut kinds = BTreeSet::new(); - if let Some(fields) = raw.fields.as_ref() { - for field in fields.values() { - for field_type in &field.types { - if field_type.named - && let Some(kind) = &field_type.kind - { - kinds.insert(kind.clone()); - } - } - } - } - - if let Some(children) = raw.children.as_ref() { - for child in &children.types { - if child.named - && let Some(kind) = &child.kind - { - kinds.insert(kind.clone()); - } - } - } - - kinds -} - -fn cargo_registry_roots() -> Vec { - let mut roots = Vec::new(); - if let Some(cargo_home) = env::var_os("CARGO_HOME") { - roots.push(PathBuf::from(cargo_home).join("registry").join("src")); - } - if let Some(home) = env::var_os("HOME") { - roots.push( - PathBuf::from(home) - .join(".cargo") - .join("registry") - .join("src"), - ); - } - roots -} - -fn cargo_git_checkout_roots() -> Vec { - let mut roots = Vec::new(); - if let Some(cargo_home) = env::var_os("CARGO_HOME") { - roots.push(PathBuf::from(cargo_home).join("git").join("checkouts")); - } - if let Some(home) = env::var_os("HOME") { - roots.push( - PathBuf::from(home) - .join(".cargo") - .join("git") - .join("checkouts"), - ); - } - roots -} - -fn find_locked_package_dir( - registry_roots: &[PathBuf], - git_roots: &[PathBuf], - package: &str, - locked: &LockedPackage, -) -> Option { - match locked.source.as_deref() { - Some(source) if source.starts_with("git+") => { - find_git_package_dir(git_roots, package, &locked.version, git_revision(source)) - }, - _ => find_registry_package_dir(registry_roots, package, &locked.version), - } -} - -fn find_registry_package_dir( - registry_roots: &[PathBuf], - package: &str, - version: &str, -) -> Option { - for registry_root in registry_roots { - let Ok(registry_dirs) = fs::read_dir(registry_root) else { - continue; - }; - for registry_dir in registry_dirs.flatten() { - let candidate = registry_dir.path().join(format!("{package}-{version}")); - if candidate.exists() { - return Some(candidate); - } - } - } - None -} - -fn find_git_package_dir( - git_roots: &[PathBuf], - package: &str, - version: &str, - revision: Option<&str>, -) -> Option { - for git_root in git_roots { - let Ok(checkout_dirs) = fs::read_dir(git_root) else { - continue; - }; - for checkout_dir in checkout_dirs.flatten() { - let Ok(revision_dirs) = fs::read_dir(checkout_dir.path()) else { - continue; - }; - for revision_dir in revision_dirs.flatten() { - let revision_path = revision_dir.path(); - let Some(revision_name) = revision_path.file_name().and_then(|name| name.to_str()) - else { - continue; - }; - if !revision_matches(revision_name, revision) { - continue; - } - if let Some(package_dir) = find_manifest_package_dir(&revision_path, package, version) { - return Some(package_dir); - } - } - } - } - None -} - -fn revision_matches(revision_name: &str, revision: Option<&str>) -> bool { - revision.is_none_or(|revision| { - revision.starts_with(revision_name) || revision_name.starts_with(revision) - }) -} - -fn find_manifest_package_dir(root: &Path, package: &str, version: &str) -> Option { - if manifest_matches_package(&root.join("Cargo.toml"), package, version) { - return Some(root.to_path_buf()); - } - - let Ok(entries) = fs::read_dir(root) else { - return None; - }; - for entry in entries.flatten() { - let candidate = entry.path(); - if candidate.is_dir() - && manifest_matches_package(&candidate.join("Cargo.toml"), package, version) - { - return Some(candidate); - } - } - None -} - -fn manifest_matches_package(manifest_path: &Path, package: &str, version: &str) -> bool { - let Ok(source) = fs::read_to_string(manifest_path) else { - return false; - }; - let mut in_package = false; - let mut name_matches = false; - let mut version_matches = false; - - for line in source.lines() { - let trimmed = line.trim(); - if trimmed.starts_with('[') { - in_package = trimmed == "[package]"; - continue; - } - if !in_package { - continue; - } - if let Some(value) = toml_string_value(trimmed, "name") { - name_matches = value == package; - continue; - } - if let Some(value) = toml_string_value(trimmed, "version") { - version_matches = value == version; - } - } - - name_matches && version_matches -} - -fn git_revision(source: &str) -> Option<&str> { - source.rsplit_once('#').and_then(|(_, revision)| { - if revision.is_empty() { - None - } else { - Some(revision) - } - }) -} - -fn locked_packages(lock_path: &Path) -> HashMap { - let source = fs::read_to_string(lock_path).expect("Cargo.lock should be readable"); - let mut packages = HashMap::new(); - let mut current_name = None; - let mut current_version = None; - let mut current_source = None; - - for line in source.lines() { - let trimmed = line.trim(); - if trimmed == "[[package]]" { - if let (Some(name), Some(version)) = (current_name.take(), current_version.take()) { - packages.insert(name, LockedPackage { version, source: current_source.take() }); - } - current_source = None; - continue; - } - if let Some(value) = toml_string_value(trimmed, "name") { - current_name = Some(value.to_string()); - continue; - } - if let Some(value) = toml_string_value(trimmed, "version") { - current_version = Some(value.to_string()); - continue; - } - if let Some(value) = toml_string_value(trimmed, "source") { - current_source = Some(value.to_string()); - } - } - - if let (Some(name), Some(version)) = (current_name, current_version) { - packages.insert(name, LockedPackage { version, source: current_source }); - } - - packages -} - -fn toml_string_value<'a>(line: &'a str, key: &str) -> Option<&'a str> { - let value = line - .strip_prefix(key)? - .trim_start() - .strip_prefix('=')? - .trim_start() - .strip_prefix('"')?; - let end = value.find('"')?; - Some(&value[..end]) -} diff --git a/crates/pi-natives/src/chunk/ast_astro.rs b/crates/pi-natives/src/chunk/ast_astro.rs deleted file mode 100644 index 40db2619e..000000000 --- a/crates/pi-natives/src/chunk/ast_astro.rs +++ /dev/null @@ -1,231 +0,0 @@ -//! Language-specific chunk classifiers for Astro. - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier, StructuralOverrides}, - common::*, - kind::ChunkKind, -}; -use crate::language::SupportLang; - -pub struct AstroClassifier; - -impl LangClassifier for AstroClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: StructuralOverrides { - extra_trivia: &[], - preserved_trivia: &[], - extra_root_wrappers: &["document"], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - _context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - classify_astro_node(node, source) - } -} - -fn classify_astro_node<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "frontmatter" => Some(classify_frontmatter(node, source)), - "frontmatter_js_block" => Some(group_candidate(node, ChunkKind::Code, source)), - "element" => classify_element(node, source), - "script_element" => Some(classify_script_element(node, source)), - "style_element" => Some(classify_style_element(node, source)), - "html_interpolation" => Some(classify_html_interpolation(node, source)), - "attribute_interpolation" => Some(classify_attribute_interpolation(node, source)), - "attribute_js_expr" => Some(group_candidate(node, ChunkKind::Expression, source)), - "text" => Some(group_candidate(node, ChunkKind::Text, source)), - _ => None, - } -} - -fn classify_frontmatter<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let Some(content_node) = child_by_kind(node, &["frontmatter_js_block"]) else { - return make_kind_chunk(node, ChunkKind::Frontmatter, None, source, None); - }; - let candidate = with_region_node( - make_kind_chunk(node, ChunkKind::Frontmatter, None, source, None), - Some(content_node), - ); - with_injected_subtree(candidate, SupportLang::TypeScript, content_node) -} - -fn classify_element<'t>(node: Node<'t>, source: &str) -> Option> { - let tag_name = extract_tag_name(node, source)?; - let recurse = Some(recurse_self(node, ChunkContext::ClassBody)); - if is_component_name(tag_name.as_str()) { - Some(force_container(make_explicit_candidate( - node, - ChunkKind::Tag, - format!("component_{tag_name}"), - source, - recurse, - ))) - } else { - Some(force_container(make_container_chunk( - node, - ChunkKind::Tag, - Some(tag_name), - source, - recurse, - ))) - } -} - -fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let identifier = has_attribute(node, "is:inline", source).then_some("inline".to_string()); - classify_raw_text_block(node, ChunkKind::Script, identifier, source, SupportLang::TypeScript) -} - -fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let identifier = if has_attribute(node, "define:vars", source) { - Some("vars".to_string()) - } else if has_attribute(node, "is:global", source) { - Some("global".to_string()) - } else { - None - }; - classify_raw_text_block(node, ChunkKind::Style, identifier, source, SupportLang::Css) -} - -fn classify_html_interpolation<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let identifier = child_by_kind(node, &["permissible_text"]) - .and_then(|expr| sanitize_identifier(node_text(source, expr.start_byte(), expr.end_byte()))); - - if let Some(nested_element) = - child_by_kind(node, &["element", "script_element", "style_element"]) - { - force_container(make_container_chunk( - node, - ChunkKind::Expression, - identifier, - source, - Some(recurse_self(nested_element, ChunkContext::ClassBody)), - )) - } else { - make_kind_chunk(node, ChunkKind::Expression, identifier, source, None) - } -} - -fn classify_attribute_interpolation<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let identifier = child_by_kind(node, &["attribute_js_expr"]) - .and_then(|expr| sanitize_identifier(node_text(source, expr.start_byte(), expr.end_byte()))) - .map_or_else(|| "expr".to_string(), |expr| format!("expr_{expr}")); - make_kind_chunk(node, ChunkKind::Attr, Some(identifier), source, None) -} - -fn make_explicit_candidate<'t>( - node: Node<'t>, - kind: ChunkKind, - identifier: impl Into>, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'t> { - make_candidate( - node, - kind, - identifier, - NameStyle::Named, - signature_for_node(node, source), - recurse, - source, - ) -} - -const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> { - candidate.force_recurse = true; - candidate -} - -fn classify_raw_text_block<'t>( - node: Node<'t>, - kind: ChunkKind, - identifier: Option, - source: &str, - default_language: SupportLang, -) -> RawChunkCandidate<'t> { - let Some(content_node) = child_by_kind(node, &["raw_text"]) else { - return make_kind_chunk(node, kind, identifier, source, None); - }; - let candidate = - with_region_node(make_kind_chunk(node, kind, identifier, source, None), Some(content_node)); - match resolve_embedded_language(node, source, default_language) { - Some(language) => with_injected_subtree(candidate, language, content_node), - None => candidate, - } -} - -fn extract_tag_name(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["start_tag", "self_closing_tag"]) - .and_then(|tag| child_by_kind(tag, &["tag_name"])) - .and_then(|tag_name| { - sanitize_identifier(node_text(source, tag_name.start_byte(), tag_name.end_byte())) - }) -} - -fn has_attribute(node: Node<'_>, name: &str, source: &str) -> bool { - child_by_kind(node, &["start_tag", "self_closing_tag"]) - .into_iter() - .flat_map(named_children) - .filter(|child| child.kind() == "attribute") - .filter_map(|attr| extract_attribute_name(attr, source)) - .any(|attr_name| attr_name == name) -} - -fn extract_attribute_name(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["attribute_name"]).map(|name| { - node_text(source, name.start_byte(), name.end_byte()) - .trim() - .to_string() - }) -} - -fn resolve_embedded_language( - node: Node<'_>, - source: &str, - default_language: SupportLang, -) -> Option { - if let Some(language) = attribute_value(node, "lang", source) { - return SupportLang::from_alias(language.as_str()); - } - Some(default_language) -} - -fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option { - let start = child_by_kind(node, &["start_tag", "self_closing_tag"])?; - for child in named_children(start) { - if child.kind() != "attribute" { - continue; - } - if extract_attribute_name(child, source).as_deref() != Some(name) { - continue; - } - if let Some(value) = child_by_kind(child, &["attribute_value", "quoted_attribute_value"]) { - return sanitize_identifier(&unquote_text(node_text( - source, - value.start_byte(), - value.end_byte(), - ))); - } - return Some(name.to_string()); - } - None -} - -fn is_component_name(tag_name: &str) -> bool { - tag_name.chars().next().is_some_and(char::is_uppercase) -} diff --git a/crates/pi-natives/src/chunk/ast_bash_make_diff.rs b/crates/pi-natives/src/chunk/ast_bash_make_diff.rs deleted file mode 100644 index 21a017365..000000000 --- a/crates/pi-natives/src/chunk/ast_bash_make_diff.rs +++ /dev/null @@ -1,275 +0,0 @@ -//! Language-specific chunk classifiers for Bash, Make, and Diff. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct ShellBuildClassifier; - -impl ShellBuildClassifier { - /// Extract a Make rule target name (child node of kind `targets`). - fn extract_rule_target(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["targets"]) - .and_then(|t| sanitize_identifier(node_text(source, t.start_byte(), t.end_byte()))) - } - - /// Extract a Make variable/define name (field `name`). - fn extract_var_name(node: Node<'_>, source: &str) -> Option { - node - .child_by_field_name("name") - .and_then(|n| sanitize_identifier(node_text(source, n.start_byte(), n.end_byte()))) - } - - /// Strip the conventional `a/` or `b/` prefix from git diff paths. - fn strip_ab_prefix(path: &str) -> &str { - path - .strip_prefix("a/") - .or_else(|| path.strip_prefix("b/")) - .unwrap_or(path) - } - - /// Extract the file path from a diff `block` node. - /// - /// Extraction priority: - /// 1. `new_file` child -> `filename` child text (skip if `/dev/null`) - /// 2. `old_file` child -> `filename` child text (skip if `/dev/null`) - /// 3. `command` child -> parse `a/path b/path` from filename children - fn extract_diff_filename(node: Node<'_>, source: &str) -> Option { - // Try new_file first (most diffs have it) - if let Some(new_file) = child_by_kind(node, &["new_file"]) - && let Some(filename) = child_by_kind(new_file, &["filename"]) - { - let text = node_text(source, filename.start_byte(), filename.end_byte()).trim(); - if text != "/dev/null" { - return sanitize_identifier(Self::strip_ab_prefix(text)); - } - } - - // Fall back to old_file (deleted files) - if let Some(old_file) = child_by_kind(node, &["old_file"]) - && let Some(filename) = child_by_kind(old_file, &["filename"]) - { - let text = node_text(source, filename.start_byte(), filename.end_byte()).trim(); - if text != "/dev/null" { - return sanitize_identifier(Self::strip_ab_prefix(text)); - } - } - - // Last resort: extract from the `command` line ("diff --git a/path b/path"). - // The grammar's `filename` rule is `repeat1(/\S+/)`, so it captures both - // paths as a single node like "a/foo.ts b/foo.ts". Take the last - // space-delimited segment (the b-side path). - if let Some(command) = child_by_kind(node, &["command"]) - && let Some(filename) = child_by_kind(command, &["filename"]) - { - let text = node_text(source, filename.start_byte(), filename.end_byte()).trim(); - let b_side = text.rsplit_once(' ').map_or(text, |(_, b)| b); - return sanitize_identifier(Self::strip_ab_prefix(b_side)); - } - - None - } -} - -impl LangClassifier for ShellBuildClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[ - semantic_rule( - "conditional", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "command", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "pipeline", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "if_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "case_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "while_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "for_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "function_definition", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "hunks", - ChunkKind::Hunks, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - ], - class: &[semantic_rule( - "hunk", - ChunkKind::Hunk, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - )], - function: &[ - semantic_rule( - "if_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "case_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "while_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "for_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "command", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "pipeline", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "subshell", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - ], - structural_overrides: StructuralOverrides { - extra_trivia: &[], - preserved_trivia: &[], - extra_root_wrappers: &["makefile"], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_root_custom(node, source), - _ => None, - } - } - - fn preserve_children( - &self, - _parent: &RawChunkCandidate<'_>, - children: &[RawChunkCandidate<'_>], - ) -> bool { - // Diff file blocks should always preserve hunk children - children.iter().any(|c| c.kind == ChunkKind::Hunk) - } -} - -fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "rule" => { - let name = ShellBuildClassifier::extract_rule_target(node, source) - .unwrap_or_else(|| "anonymous".to_string()); - Some(make_container_chunk( - node, - ChunkKind::Rule, - Some(name), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["recipe"]), - )) - }, - "variable_assignment" | "shell_assignment" => { - let name = ShellBuildClassifier::extract_var_name(node, source) - .unwrap_or_else(|| "anonymous".to_string()); - Some(make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)) - }, - "define_directive" => { - let name = ShellBuildClassifier::extract_var_name(node, source) - .unwrap_or_else(|| "anonymous".to_string()); - Some(make_kind_chunk(node, ChunkKind::Define, Some(name), source, None)) - }, - "block" => { - let identifier = ShellBuildClassifier::extract_diff_filename(node, source); - let recurse = recurse_into(node, ChunkContext::ClassBody, &[], &["hunks"]); - let mut candidate = - make_container_chunk(node, ChunkKind::File, identifier, source, recurse); - // Always expand hunks so individual @@ sections are addressable, - // even for small diffs below the leaf threshold. - candidate.force_recurse = recurse.is_some(); - Some(candidate) - }, - _ => None, - } -} diff --git a/crates/pi-natives/src/chunk/ast_c_cpp_objc.rs b/crates/pi-natives/src/chunk/ast_c_cpp_objc.rs deleted file mode 100644 index 324957b5a..000000000 --- a/crates/pi-natives/src/chunk/ast_c_cpp_objc.rs +++ /dev/null @@ -1,431 +0,0 @@ -//! Language-specific chunk classifiers for C, C++, and Objective-C. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - defaults::classify_var_decl, - kind::ChunkKind, -}; - -pub struct CCppClassifier; - -/// Extract the function name from a C/C++ `function_definition` or -/// `function_declaration` node by traversing into the `declarator` chain. -fn extract_c_function_name(node: Node<'_>, source: &str) -> Option { - let decl = node.child_by_field_name("declarator")?; - extract_c_declarator_name(decl, source) -} - -/// Recursively resolve a C/C++ declarator to its leaf identifier. -/// Handles `function_declarator`, `pointer_declarator`, `reference_declarator`, -/// `qualified_identifier`, `destructor_name`, `template_function`, etc. -fn extract_c_declarator_name(node: Node<'_>, source: &str) -> Option { - match node.kind() { - "identifier" | "field_identifier" | "type_identifier" => { - sanitize_identifier(node_text(source, node.start_byte(), node.end_byte())) - }, - "destructor_name" => { - // ~ClassName - sanitize_identifier(node_text(source, node.start_byte(), node.end_byte())) - }, - "qualified_identifier" | "scoped_identifier" => { - // e.g. Entity::update — extract the "name" field or last identifier - node - .child_by_field_name("name") - .and_then(|n| extract_c_declarator_name(n, source)) - .or_else(|| { - named_children(node) - .into_iter() - .rev() - .find(|c| { - matches!( - c.kind(), - "identifier" | "destructor_name" | "template_function" | "field_identifier" - ) - }) - .and_then(|c| extract_c_declarator_name(c, source)) - }) - }, - "template_function" => { - // template_function has a "name" field or direct identifier child - node - .child_by_field_name("name") - .and_then(|n| sanitize_identifier(node_text(source, n.start_byte(), n.end_byte()))) - .or_else(|| { - named_children(node) - .into_iter() - .find(|c| c.kind() == "identifier") - .and_then(|c| { - sanitize_identifier(node_text(source, c.start_byte(), c.end_byte())) - }) - }) - }, - _ => { - // function_declarator, pointer_declarator, reference_declarator, etc. - // recurse into the "declarator" field - node - .child_by_field_name("declarator") - .and_then(|inner| extract_c_declarator_name(inner, source)) - .or_else(|| { - // fallback: look for direct identifier-like child - named_children(node) - .into_iter() - .find(|c| { - matches!( - c.kind(), - "identifier" - | "field_identifier" - | "qualified_identifier" - | "scoped_identifier" - | "destructor_name" - | "template_function" - ) - }) - .and_then(|c| extract_c_declarator_name(c, source)) - }) - }, - } -} - -/// Extract the field name from a C/C++ `field_declaration` node. -/// The name sits in the `declarator` field which may be a plain -/// `field_identifier`, or a `function_declarator` / `pointer_declarator` etc. -fn extract_c_field_name(node: Node<'_>, source: &str) -> Option { - let decl = node.child_by_field_name("declarator")?; - extract_c_declarator_name(decl, source) -} - -const C_CPP_ROOT_RULES: &[super::classify::SemanticRule] = &[ - // ── Imports ── - semantic_rule( - "include_directive", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "preproc_include", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "using_directive", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "using_statement", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "import_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "module_import", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // ── Statements ── - semantic_rule( - "expression_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const C_CPP_TABLES: ClassifierTables = ClassifierTables { - root: C_CPP_ROOT_RULES, - class: &[], - function: &[], - structural_overrides: StructuralOverrides::EMPTY, -}; - -impl LangClassifier for CCppClassifier { - fn tables(&self) -> &'static ClassifierTables { - &C_CPP_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_root_custom(self, node, source), - ChunkContext::ClassBody => classify_class_custom(self, node, source), - ChunkContext::FunctionBody => Some(classify_function_c(node, source)), - } - } -} - -fn classify_root_custom<'t>( - classifier: &CCppClassifier, - node: Node<'t>, - source: &str, -) -> Option> { - match node.kind() { - // ── Functions ── - "function_definition" | "function_declaration" => Some(make_kind_chunk( - node, - ChunkKind::Function, - extract_c_function_name(node, source), - source, - recurse_body(node, ChunkContext::FunctionBody), - )), - "constructor_definition" => Some(make_kind_chunk( - node, - ChunkKind::Constructor, - None, - source, - recurse_body(node, ChunkContext::FunctionBody), - )), - - // ── Templates (unwrap to find the inner declaration) ── - "template_declaration" => { - // Find the inner function_definition / class_specifier / etc. - let inner = named_children(node).into_iter().find(|c| { - matches!( - c.kind(), - "function_definition" - | "function_declaration" - | "class_specifier" - | "struct_specifier" - | "type_alias_declaration" - ) - }); - match inner { - Some(inner) => { - let mut candidate = - classifier.classify_override(ChunkContext::Root, inner, source)?; - // Expand range to include the template<...> prefix - candidate.range_start_byte = node.start_byte(); - candidate.range_start_line = node.start_position().row + 1; - candidate.checksum_start_byte = node.start_byte(); - Some(candidate) - }, - None => Some(make_candidate( - node, - ChunkKind::Template, - None, - NameStyle::Named, - signature_for_node(node, source), - recurse_body(node, ChunkContext::FunctionBody), - source, - )), - } - }, - - // ── Containers ── - "class_specifier" | "class_declaration" | "class_interface" | "class_implementation" => { - Some(container_candidate(node, ChunkKind::Class, source, recurse_class(node))) - }, - "struct_specifier" | "struct_declaration" => { - Some(container_candidate(node, ChunkKind::Struct, source, recurse_class(node))) - }, - "enum_specifier" | "enum_declaration" => { - Some(container_candidate(node, ChunkKind::Enum, source, recurse_enum(node))) - }, - "namespace_definition" => { - Some(container_candidate(node, ChunkKind::Module, source, recurse_class(node))) - }, - "union_declaration" => { - Some(container_candidate(node, ChunkKind::Union, source, recurse_class(node))) - }, - - // ── Types ── - "type_alias_declaration" | "user_defined_type_definition" => { - Some(named_candidate(node, ChunkKind::Type, source, recurse_class(node))) - }, - - // ── Variables / assignments ── - "variable_declaration" => Some(classify_var_decl(node, source)), - "assignment_statement" | "property_declaration" => { - Some(group_candidate(node, ChunkKind::Declarations, source)) - }, - - // ── Macros ── - "macro_definition" => Some(named_candidate( - node, - ChunkKind::Macro, - source, - recurse_body(node, ChunkContext::FunctionBody), - )), - - // ── Control flow (top-level scripts) ── - "if_statement" | "switch_statement" | "for_statement" | "while_statement" - | "do_statement" | "try_block" => Some(classify_function_c(node, source)), - - _ => None, - } -} - -fn classify_class_custom<'t>( - classifier: &CCppClassifier, - node: Node<'t>, - source: &str, -) -> Option> { - match node.kind() { - // ── Methods ── - "function_definition" | "function_declaration" | "method_declaration" => { - let name = extract_c_function_name(node, source) - .or_else(|| extract_identifier(node, source)) - .unwrap_or_else(|| "anonymous".to_string()); - if name == "constructor" { - Some(make_kind_chunk( - node, - ChunkKind::Constructor, - None, - source, - recurse_body(node, ChunkContext::FunctionBody), - )) - } else { - Some(make_kind_chunk( - node, - ChunkKind::Function, - Some(name), - source, - recurse_body(node, ChunkContext::FunctionBody), - )) - } - }, - - // ── Constructors ── - "constructor_definition" | "constructor_declaration" => Some(make_kind_chunk( - node, - ChunkKind::Constructor, - None, - source, - recurse_body(node, ChunkContext::FunctionBody), - )), - - // ── Fields ── - "field_declaration" => Some(match extract_c_field_name(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None), - None => group_candidate(node, ChunkKind::Fields, source), - }), - - // ── Enum variants ── - "enum_constant" => Some(match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None), - None => group_candidate(node, ChunkKind::Variants, source), - }), - - // ── Nested containers ── - "class_specifier" | "class_declaration" | "class_interface" | "class_implementation" => { - Some(container_candidate(node, ChunkKind::Class, source, recurse_class(node))) - }, - "struct_specifier" | "struct_declaration" => { - Some(container_candidate(node, ChunkKind::Struct, source, recurse_class(node))) - }, - "enum_specifier" | "enum_declaration" => { - Some(container_candidate(node, ChunkKind::Enum, source, recurse_enum(node))) - }, - "union_declaration" => { - Some(container_candidate(node, ChunkKind::Union, source, recurse_class(node))) - }, - "namespace_definition" => { - Some(container_candidate(node, ChunkKind::Module, source, recurse_class(node))) - }, - - // ── Templates (class body) ── - "template_declaration" => { - let inner = named_children(node).into_iter().find(|c| { - matches!( - c.kind(), - "function_definition" - | "function_declaration" - | "class_specifier" - | "struct_specifier" - | "type_alias_declaration" - ) - }); - match inner { - Some(inner) => { - let mut candidate = - classifier.classify_override(ChunkContext::ClassBody, inner, source)?; - candidate.range_start_byte = node.start_byte(); - candidate.range_start_line = node.start_position().row + 1; - candidate.checksum_start_byte = node.start_byte(); - Some(candidate) - }, - None => Some(make_candidate( - node, - ChunkKind::Template, - None, - NameStyle::Named, - signature_for_node(node, source), - recurse_body(node, ChunkContext::FunctionBody), - source, - )), - } - }, - - // ── Types ── - "type_alias_declaration" => Some(named_candidate(node, ChunkKind::Type, source, None)), - - _ => None, - } -} - -fn classify_function_c<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> { - let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody); - match node.kind() { - "if_statement" => { - make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source) - }, - "switch_statement" => { - make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source) - }, - "try_block" | "catch_clause" | "finally_clause" => { - make_candidate(node, ChunkKind::Try, None, NameStyle::Named, None, fn_recurse(), source) - }, - "for_statement" => { - make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source) - }, - "while_statement" => { - make_candidate(node, ChunkKind::While, None, NameStyle::Named, None, fn_recurse(), source) - }, - "do_statement" => { - make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source) - }, - "variable_declaration" => { - let span = line_span(node.start_position().row + 1, node.end_position().row + 1); - if span > 1 { - if let Some(name) = extract_single_declarator_name(node, source) { - make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None) - } else { - group_candidate(node, ChunkKind::Variable, source) - } - } else { - group_candidate(node, ChunkKind::Variable, source) - } - }, - _ => { - let kind_name = sanitize_node_kind(node.kind()); - let kind = ChunkKind::from_sanitized_kind(kind_name); - group_candidate(node, kind, source) - }, - } -} diff --git a/crates/pi-natives/src/chunk/ast_clojure.rs b/crates/pi-natives/src/chunk/ast_clojure.rs deleted file mode 100644 index 598d7b58d..000000000 --- a/crates/pi-natives/src/chunk/ast_clojure.rs +++ /dev/null @@ -1,78 +0,0 @@ -//! Language-specific chunk classifier for Clojure. - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier}, - common::*, - kind::ChunkKind, -}; - -pub struct ClojureClassifier; - -/// Extract the head symbol of a Clojure list form (first -/// `sym_lit`/`kwd_lit`/`symbol`/`word`). -fn form_head(node: Node<'_>, source: &str) -> Option { - named_children(node).into_iter().find_map(|child| { - matches!(child.kind(), "sym_lit" | "kwd_lit" | "symbol" | "word") - .then(|| node_text(source, child.start_byte(), child.end_byte()).to_string()) - }) -} - -/// Extract the name from a Clojure form: the second named child after the head -/// symbol (e.g. `greet` in `(defn greet [x] x)`). -fn form_name(node: Node<'_>, source: &str) -> Option { - let mut children = named_children(node).into_iter(); - let _head = children.next()?; - children.find_map(|child| { - sanitize_identifier(node_text(source, child.start_byte(), child.end_byte())) - }) -} - -/// Classify a `list_lit` Clojure form based on its head symbol. -fn classify_form<'t>(node: Node<'t>, source: &str, at_root: bool) -> RawChunkCandidate<'t> { - let Some(head) = form_head(node, source) else { - return positional_candidate(node, ChunkKind::Form, source); - }; - match head.as_str() { - "ns" | "require" | "use" | "import" | "refer-clojure" => { - group_candidate(node, ChunkKind::Imports, source) - }, - "defn" | "defn-" | "defmacro" | "defmulti" | "defmethod" => { - make_kind_chunk(node, ChunkKind::Function, form_name(node, source), source, None) - }, - "def" | "defonce" => { - make_kind_chunk(node, ChunkKind::Decl, form_name(node, source), source, None) - }, - "defprotocol" => { - make_container_chunk(node, ChunkKind::Proto, form_name(node, source), source, None) - }, - "deftype" | "defrecord" | "extend-type" | "extend-protocol" => { - make_container_chunk(node, ChunkKind::Type, form_name(node, source), source, None) - }, - _ if at_root => positional_candidate(node, ChunkKind::Form, source), - _ => group_candidate(node, ChunkKind::Block, source), - } -} - -impl LangClassifier for ClojureClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: super::classify::StructuralOverrides::EMPTY, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - (node.kind() == "list_lit") - .then(|| classify_form(node, source, matches!(context, ChunkContext::Root))) - } -} diff --git a/crates/pi-natives/src/chunk/ast_cmake.rs b/crates/pi-natives/src/chunk/ast_cmake.rs deleted file mode 100644 index 9b6609c04..000000000 --- a/crates/pi-natives/src/chunk/ast_cmake.rs +++ /dev/null @@ -1,174 +0,0 @@ -//! CMake-specific chunk classifier. - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier, StructuralOverrides}, - common::*, - kind::ChunkKind, -}; - -pub struct CMakeClassifier; - -fn child_text<'a>(source: &'a str, node: Node<'_>) -> &'a str { - node_text(source, node.start_byte(), node.end_byte()) -} - -fn first_named_child(node: Node<'_>) -> Option> { - named_children(node).into_iter().next() -} - -fn first_named_child_of_kind<'t>(node: Node<'t>, kind: &str) -> Option> { - named_children(node) - .into_iter() - .find(|child| child.kind() == kind) -} - -fn command_name(node: Node<'_>, source: &str) -> Option { - first_named_child(node).and_then(|child| sanitize_identifier(child_text(source, child))) -} - -fn argument_nodes(node: Node<'_>) -> Vec> { - first_named_child_of_kind(node, "argument_list") - .map(named_children) - .unwrap_or_default() - .into_iter() - .filter(|child| child.kind() == "argument") - .collect() -} - -fn nth_argument_name(node: Node<'_>, index: usize, source: &str) -> Option { - argument_nodes(node) - .into_iter() - .nth(index) - .and_then(|arg| sanitize_identifier(child_text(source, arg))) -} - -fn classify_definition<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "function_def" => { - let header = first_named_child_of_kind(node, "function_command")?; - let name = nth_argument_name(header, 0, source).unwrap_or_else(|| "anonymous".to_string()); - Some(make_container_chunk( - node, - ChunkKind::Function, - Some(name), - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["body"]), - )) - }, - "macro_def" => { - let header = first_named_child_of_kind(node, "macro_command")?; - let name = nth_argument_name(header, 0, source).unwrap_or_else(|| "anonymous".to_string()); - Some(make_container_chunk( - node, - ChunkKind::Macro, - Some(name), - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["body"]), - )) - }, - "if_condition" => Some(make_container_chunk( - node, - ChunkKind::If, - None, - source, - Some(recurse_self(node, ChunkContext::FunctionBody)), - )), - "foreach_loop" | "while_loop" => Some(make_container_chunk( - node, - ChunkKind::Loop, - None, - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["body"]), - )), - _ => None, - } -} - -fn classify_command<'t>(node: Node<'t>, source: &str) -> Option> { - if node.kind() != "normal_command" { - return None; - } - - let command = command_name(node, source)?; - Some(match command.as_str() { - "cmake_minimum_required" => make_kind_chunk(node, ChunkKind::VersionGate, None, source, None), - "project" => { - let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string()); - make_kind_chunk(node, ChunkKind::Project, Some(name), source, None) - }, - "include" | "find_package" => group_candidate(node, ChunkKind::Imports, source), - "option" => { - let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string()); - make_kind_chunk(node, ChunkKind::Option, Some(name), source, None) - }, - "set" => { - let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string()); - make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None) - }, - "add_library" | "add_executable" | "add_custom_target" => { - let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string()); - make_kind_chunk(node, ChunkKind::Target, Some(name), source, None) - }, - "install" | "export" => group_candidate(node, ChunkKind::Install, source), - other => make_kind_chunk(node, ChunkKind::Cmd, Some(other.to_string()), source, None), - }) -} - -fn classify_if_child<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "if_command" => Some(group_candidate(node, ChunkKind::Cond, source)), - "elseif_command" => Some(positional_candidate(node, ChunkKind::Elif, source)), - "else_command" => Some(positional_candidate(node, ChunkKind::Else, source)), - "body" => Some(make_container_chunk( - node, - ChunkKind::Block, - None, - source, - Some(recurse_self(node, ChunkContext::FunctionBody)), - )), - _ => None, - } -} - -impl LangClassifier for CMakeClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: StructuralOverrides { - extra_trivia: &[ - "endif_command", - "endforeach_command", - "endwhile_command", - "endfunction_command", - "endmacro_command", - ], - preserved_trivia: &[], - extra_root_wrappers: &[], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => { - classify_definition(node, source).or_else(|| classify_command(node, source)) - }, - ChunkContext::FunctionBody => classify_definition(node, source) - .or_else(|| classify_if_child(node, source)) - .or_else(|| classify_command(node, source)), - ChunkContext::ClassBody => None, - } - } -} diff --git a/crates/pi-natives/src/chunk/ast_csharp_java.rs b/crates/pi-natives/src/chunk/ast_csharp_java.rs deleted file mode 100644 index fff82af77..000000000 --- a/crates/pi-natives/src/chunk/ast_csharp_java.rs +++ /dev/null @@ -1,372 +0,0 @@ -//! Language-specific chunk classifiers for C# and Java. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - defaults::classify_var_decl, - kind::ChunkKind, -}; - -pub struct CSharpJavaClassifier; - -const CSHARP_JAVA_ROOT_RULES: &[super::classify::SemanticRule] = &[ - // ── Imports ── - semantic_rule( - "import_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "using_directive", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "package_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "namespace_statement", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // ── Functions ── - semantic_rule( - "method_declaration", - ChunkKind::Method, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "function_declaration", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "function_definition", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // ── Constructors ── - semantic_rule( - "constructor_declaration", - ChunkKind::Constructor, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // ── Containers ── - semantic_rule( - "class_declaration", - ChunkKind::Class, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "interface_declaration", - ChunkKind::Iface, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "enum_declaration", - ChunkKind::Enum, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "struct_declaration", - ChunkKind::Struct, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "record_declaration", - ChunkKind::Struct, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "namespace_declaration", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "file_scoped_namespace_declaration", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - // ── Types ── - semantic_rule( - "type_alias_declaration", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - // ── Declarations ── - semantic_rule( - "property_declaration", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "state_variable_declaration", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // ── Statements ── - semantic_rule( - "expression_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const CSHARP_JAVA_CLASS_RULES: &[super::classify::SemanticRule] = &[ - // ── Containers ── - semantic_rule( - "class_declaration", - ChunkKind::Class, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "interface_declaration", - ChunkKind::Iface, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "enum_declaration", - ChunkKind::Enum, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "struct_declaration", - ChunkKind::Struct, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "record_declaration", - ChunkKind::Struct, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "namespace_declaration", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "file_scoped_namespace_declaration", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - // ── Static blocks ── - semantic_rule( - "class_static_block", - ChunkKind::StaticInit, - RuleStyle::Named, - NamingMode::None, - RecurseMode::None, - ), -]; - -const CSHARP_JAVA_TABLES: ClassifierTables = ClassifierTables { - root: CSHARP_JAVA_ROOT_RULES, - class: CSHARP_JAVA_CLASS_RULES, - function: &[], - structural_overrides: StructuralOverrides::EMPTY, -}; - -impl LangClassifier for CSharpJavaClassifier { - fn tables(&self) -> &'static ClassifierTables { - &CSHARP_JAVA_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => match node.kind() { - // ── Variables / assignments ── - "variable_declaration" | "lexical_declaration" => Some(classify_var_decl(node, source)), - // ── Control flow (top-level scripts) ── - "if_statement" | "switch_statement" | "switch_expression" | "for_statement" - | "foreach_statement" | "while_statement" | "do_statement" | "try_statement" => { - Some(classify_function_csharp_java(node, source)) - }, - _ => None, - }, - ChunkContext::ClassBody => match node.kind() { - // ── Methods (conditional constructor detection) ── - "method_declaration" | "function_declaration" | "function_definition" => { - let name = - extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string()); - if name == "constructor" { - Some(make_kind_chunk( - node, - ChunkKind::Constructor, - None, - source, - recurse_body(node, ChunkContext::FunctionBody), - )) - } else { - Some(make_kind_chunk( - node, - ChunkKind::Function, - Some(name), - source, - recurse_body(node, ChunkContext::FunctionBody), - )) - } - }, - // ── Constructors ── - "constructor_declaration" | "secondary_constructor" => Some(make_kind_chunk( - node, - ChunkKind::Constructor, - None, - source, - recurse_body(node, ChunkContext::FunctionBody), - )), - // ── Fields ── - "field_declaration" - | "property_declaration" - | "constant_declaration" - | "event_field_declaration" => Some(match extract_field_name(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None), - None => group_candidate(node, ChunkKind::Fields, source), - }), - // ── Enum members ── - "enum_member_declaration" | "enum_constant" | "enum_entry" => { - Some(match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None), - None => group_candidate(node, ChunkKind::Variants, source), - }) - }, - _ => None, - }, - ChunkContext::FunctionBody => Some(classify_function_csharp_java(node, source)), - } - } -} - -/// Extract the variable name from a field/constant declaration. -/// -/// Java `field_declaration` has the structure: -/// `field_declaration` { modifiers, type: `type_identifier`, declarator: -/// `variable_declarator` { name: identifier } } -/// -/// `extract_identifier` would find `type_identifier` first, so we look into -/// `variable_declarator` children for the actual variable name. -fn extract_field_name(node: Node<'_>, source: &str) -> Option { - for child in named_children(node) { - if child.kind() == "variable_declarator" { - return extract_identifier(child, source); - } - } - extract_identifier(node, source) -} - -fn classify_function_csharp_java<'tree>( - node: Node<'tree>, - source: &str, -) -> RawChunkCandidate<'tree> { - let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody); - match node.kind() { - "if_statement" => { - make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source) - }, - "switch_statement" | "switch_expression" => { - make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source) - }, - "try_statement" | "catch_clause" | "finally_clause" => { - make_candidate(node, ChunkKind::Try, None, NameStyle::Named, None, fn_recurse(), source) - }, - "for_statement" => { - make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source) - }, - "foreach_statement" => { - make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source) - }, - "while_statement" => { - make_candidate(node, ChunkKind::While, None, NameStyle::Named, None, fn_recurse(), source) - }, - "do_statement" => { - make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source) - }, - "variable_declaration" | "lexical_declaration" => { - let span = line_span(node.start_position().row + 1, node.end_position().row + 1); - if span > 1 { - if let Some(name) = extract_single_declarator_name(node, source) { - make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None) - } else { - group_from_sanitized(node, source) - } - } else { - group_from_sanitized(node, source) - } - }, - _ => group_from_sanitized(node, source), - } -} - -fn group_from_sanitized<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let sanitized = sanitize_node_kind(node.kind()); - let kind = ChunkKind::from_sanitized_kind(sanitized); - let identifier = if kind == ChunkKind::Chunk { - Some(sanitized.to_string()) - } else { - None - }; - make_candidate(node, kind, identifier, NameStyle::Group, None, None, source) -} diff --git a/crates/pi-natives/src/chunk/ast_css.rs b/crates/pi-natives/src/chunk/ast_css.rs deleted file mode 100644 index 269132654..000000000 --- a/crates/pi-natives/src/chunk/ast_css.rs +++ /dev/null @@ -1,124 +0,0 @@ -//! Language-specific chunk classifiers for CSS and SCSS. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct CssClassifier; - -const CSS_SHARED_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "keyframe_block", - ChunkKind::Frame, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "declaration", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const CSS_TABLES: ClassifierTables = ClassifierTables { - root: CSS_SHARED_RULES, - class: CSS_SHARED_RULES, - function: &[], - structural_overrides: StructuralOverrides { - extra_trivia: &[], - preserved_trivia: &[], - extra_root_wrappers: &["stylesheet"], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, -}; - -/// Extract a CSS selector name from a `rule_set` or `at_rule` node. -/// -/// Tries known child kinds first (`selectors`, `selector_query`, `identifier`), -/// then falls back to parsing the normalised header text. -fn extract_css_selector(node: Node<'_>, source: &str) -> Option { - if let Some(sel) = child_by_kind(node, &["selectors", "selector_query", "identifier"]) { - return sanitize_identifier(node_text(source, sel.start_byte(), sel.end_byte())); - } - let header = normalized_header(source, node.start_byte(), node.end_byte()); - let selector = header - .trim_start_matches('@') - .split('{') - .next() - .unwrap_or(header.as_str()) - .trim(); - sanitize_identifier(selector) -} - -/// Classify a CSS `rule_set` as a named container. -fn classify_rule_set<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let name = extract_css_selector(node, source).unwrap_or_else(|| "anonymous".to_string()); - make_container_chunk( - node, - ChunkKind::Rule, - Some(name), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["block"]), - ) -} - -/// Classify a CSS at-rule (`@media`, `@keyframes`, `@supports`, etc.) as a -/// named container. -fn classify_at_rule<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let name = extract_css_selector(node, source).unwrap_or_else(|| "rule".to_string()); - make_container_chunk( - node, - ChunkKind::At, - Some(name), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["block", "keyframe_block_list"]), - ) -} - -/// Shared dispatch for CSS node kinds, used in both root and class-body -/// contexts. -fn classify_css_node<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "rule_set" => Some(classify_rule_set(node, source)), - "at_rule" | "media_statement" | "keyframes_statement" | "supports_statement" => { - Some(classify_at_rule(node, source)) - }, - "keyframe_block" => Some(named_candidate( - node, - ChunkKind::Frame, - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - )), - "declaration" => Some(group_candidate(node, ChunkKind::Fields, source)), - _ => None, - } -} - -impl LangClassifier for CssClassifier { - fn tables(&self) -> &'static ClassifierTables { - &CSS_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - if matches!(context, ChunkContext::Root | ChunkContext::ClassBody) { - return classify_css_node(node, source); - } - None - } -} diff --git a/crates/pi-natives/src/chunk/ast_data_formats.rs b/crates/pi-natives/src/chunk/ast_data_formats.rs deleted file mode 100644 index 321da8f0e..000000000 --- a/crates/pi-natives/src/chunk/ast_data_formats.rs +++ /dev/null @@ -1,278 +0,0 @@ -//! Chunk classifiers for data formats: JSON, TOML, YAML. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct DataFormatsClassifier; - -const DATA_FORMAT_STRUCTURAL_OVERRIDES: StructuralOverrides = StructuralOverrides { - extra_trivia: &["bare_key", "quoted_key", "dotted_key"], - preserved_trivia: &[], - extra_root_wrappers: &[ - "array", - "block_mapping", - "block_node", - "block_sequence", - "document", - "flow_mapping", - "flow_node", - "flow_sequence", - "object", - "stream", - ], - preserved_root_wrappers: &[], - absorbable_attrs: &[], -}; - -const DATA_FORMAT_ROOT_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "inline_table", - ChunkKind::Table, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "object", - ChunkKind::Object, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "array", - ChunkKind::Array, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "block_mapping", - ChunkKind::Map, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "flow_mapping", - ChunkKind::Map, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "block_sequence", - ChunkKind::List, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "flow_sequence", - ChunkKind::List, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "attribute", - ChunkKind::Attr, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::ValueContainer, - ), -]; - -const DATA_FORMAT_CLASS_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "inline_table", - ChunkKind::Table, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "object", - ChunkKind::Object, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "array", - ChunkKind::Array, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "block_mapping", - ChunkKind::Map, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "flow_mapping", - ChunkKind::Map, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "block_sequence", - ChunkKind::List, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "flow_sequence", - ChunkKind::List, - RuleStyle::Named, - NamingMode::None, - RecurseMode::SelfNode(ChunkContext::ClassBody), - ), - semantic_rule( - "block_sequence_item", - ChunkKind::Item, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "attribute", - ChunkKind::Attr, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::ValueContainer, - ), -]; - -const DATA_FORMAT_TABLES: ClassifierTables = ClassifierTables { - root: DATA_FORMAT_ROOT_RULES, - class: DATA_FORMAT_CLASS_RULES, - function: &[], - structural_overrides: DATA_FORMAT_STRUCTURAL_OVERRIDES, -}; - -impl LangClassifier for DataFormatsClassifier { - fn tables(&self) -> &'static ClassifierTables { - &DATA_FORMAT_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_data_node(node, source, true), - ChunkContext::ClassBody => classify_data_node(node, source, false), - ChunkContext::FunctionBody => None, - } - } - - fn preserve_children( - &self, - parent: &RawChunkCandidate<'_>, - _children: &[RawChunkCandidate<'_>], - ) -> bool { - // YAML keys with container values should always expose sub-chunks - // so that deeply nested keys are individually addressable. - parent.force_recurse && parent.kind == ChunkKind::Key - } -} - -fn classify_data_node<'t>( - node: Node<'t>, - source: &str, - is_root: bool, -) -> Option> { - match node.kind() { - // Key-value pairs (JSON pairs, YAML mappings) - "pair" => { - let name = extract_pair_key(node, source).unwrap_or_else(|| "anonymous".to_string()); - Some(make_kind_chunk( - node, - ChunkKind::Key, - Some(name), - source, - recurse_value_container(node), - )) - }, - "block_mapping_pair" | "flow_pair" => { - let name = extract_yaml_key(node, source).unwrap_or_else(|| "anonymous".to_string()); - let recurse = recurse_value_container(node); - let mut candidate = make_kind_chunk(node, ChunkKind::Key, Some(name), source, recurse); - // YAML structure is inherently hierarchical. Keys whose value is a - // container (mapping/sequence) should always produce sub-chunks so - // that deeply nested keys are individually addressable. - if candidate.recurse.is_some() { - candidate.force_recurse = true; - } - Some(candidate) - }, - // TOML tables - "table" => { - let name = extract_toml_table_name(node, source); - Some(make_container_chunk( - node, - ChunkKind::Table, - name, - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - )) - }, - // TOML array tables - "table_array_element" => Some(make_candidate( - node, - ChunkKind::Table, - extract_toml_table_name(node, source).unwrap_or_else(|| "table_array".to_string()), - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::ClassBody)), - source, - )), - // YAML sequence items (only when nested, not at root level) - "block_sequence_item" if !is_root => { - Some(positional_candidate(node, ChunkKind::Item, source)) - }, - _ => None, - } -} - -/// Extract key from a `pair` node (JSON or TOML). -/// JSON pairs have a `"key"` field; TOML pairs have no field names, so we fall -/// back to looking for the first `bare_key`, `quoted_key`, or `dotted_key` -/// child. -fn extract_pair_key(node: Node<'_>, source: &str) -> Option { - let key = node - .child_by_field_name("key") - .or_else(|| child_by_kind(node, &["bare_key", "quoted_key", "dotted_key"]))?; - sanitize_identifier(unquote_text(node_text(source, key.start_byte(), key.end_byte())).as_str()) -} - -fn extract_toml_table_name(node: Node<'_>, source: &str) -> Option { - let key = child_by_kind(node, &["dotted_key", "bare_key", "quoted_key"])?; - sanitize_identifier(node_text(source, key.start_byte(), key.end_byte())) -} - -/// Extract key from a YAML `block_mapping_pair` or `flow_pair` node. -/// Descends into the key to find the first scalar child for complex keys. -fn extract_yaml_key(node: Node<'_>, source: &str) -> Option { - let key = node.child_by_field_name("key")?; - let key_node = first_scalar_child(key).unwrap_or(key); - sanitize_identifier( - unquote_text(node_text(source, key_node.start_byte(), key_node.end_byte())).as_str(), - ) -} diff --git a/crates/pi-natives/src/chunk/ast_dockerfile.rs b/crates/pi-natives/src/chunk/ast_dockerfile.rs deleted file mode 100644 index 258e41d2f..000000000 --- a/crates/pi-natives/src/chunk/ast_dockerfile.rs +++ /dev/null @@ -1,184 +0,0 @@ -//! Chunk classifier for Dockerfile syntax. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct DockerfileClassifier; - -const DOCKERFILE_ROOT_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "run_instruction", - ChunkKind::Cmd, - RuleStyle::Named, - NamingMode::SanitizedKind, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "cmd_instruction", - ChunkKind::Cmd, - RuleStyle::Named, - NamingMode::SanitizedKind, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "entrypoint_instruction", - ChunkKind::Cmd, - RuleStyle::Named, - NamingMode::SanitizedKind, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "copy_instruction", - ChunkKind::Copy, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "add_instruction", - ChunkKind::Add, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "workdir_instruction", - ChunkKind::Workdir, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "expose_instruction", - ChunkKind::Expose, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "user_instruction", - ChunkKind::User, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const DOCKERFILE_FUNCTION_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "cmd_instruction", - ChunkKind::Cmd, - RuleStyle::Named, - NamingMode::SanitizedKind, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "shell_command", - ChunkKind::Shell, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "json_string_array", - ChunkKind::Argv, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const DOCKERFILE_TABLES: ClassifierTables = ClassifierTables { - root: DOCKERFILE_ROOT_RULES, - class: &[], - function: DOCKERFILE_FUNCTION_RULES, - structural_overrides: super::classify::StructuralOverrides::EMPTY, -}; - -fn child_text<'a>(source: &'a str, node: Node<'_>) -> &'a str { - node_text(source, node.start_byte(), node.end_byte()) -} - -fn first_named_child(node: Node<'_>) -> Option> { - named_children(node).into_iter().next() -} - -fn first_named_child_of_kind<'t>(node: Node<'t>, kind: &str) -> Option> { - named_children(node) - .into_iter() - .find(|child| child.kind() == kind) -} - -fn extract_stage_name(node: Node<'_>, source: &str) -> Option { - if let Some(alias) = child_by_kind(node, &["image_alias"]) { - return sanitize_identifier(child_text(source, alias)); - } - - child_by_kind(node, &["image_spec"]).and_then(|image| { - let image_name = child_by_kind(image, &["image_name"]).unwrap_or(image); - sanitize_identifier(child_text(source, image_name)) - }) -} - -fn extract_pair_key(node: Node<'_>, pair_kind: &str, source: &str) -> Option { - first_named_child_of_kind(node, pair_kind) - .and_then(first_named_child) - .and_then(|key| sanitize_identifier(unquote_text(child_text(source, key)).as_str())) -} - -fn extract_arg_name(node: Node<'_>, source: &str) -> Option { - first_named_child(node).and_then(|name| sanitize_identifier(child_text(source, name))) -} - -impl LangClassifier for DockerfileClassifier { - fn tables(&self) -> &'static ClassifierTables { - &DOCKERFILE_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => match node.kind() { - "from_instruction" => { - let name = - extract_stage_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - Some(make_kind_chunk(node, ChunkKind::Stage, Some(name), source, None)) - }, - "arg_instruction" => { - let name = extract_arg_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - Some(make_kind_chunk(node, ChunkKind::Arg, Some(name), source, None)) - }, - "env_instruction" => { - let name = extract_pair_key(node, "env_pair", source) - .unwrap_or_else(|| "anonymous".to_string()); - Some(make_kind_chunk(node, ChunkKind::Env, Some(name), source, None)) - }, - "label_instruction" => { - let name = extract_pair_key(node, "label_pair", source) - .unwrap_or_else(|| "anonymous".to_string()); - Some(make_kind_chunk(node, ChunkKind::Label, Some(name), source, None)) - }, - "healthcheck_instruction" => Some(make_container_chunk( - node, - ChunkKind::Healthcheck, - None, - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["cmd_instruction"]), - )), - _ => None, - }, - _ => None, - } - } -} diff --git a/crates/pi-natives/src/chunk/ast_elixir.rs b/crates/pi-natives/src/chunk/ast_elixir.rs deleted file mode 100644 index f1d472eda..000000000 --- a/crates/pi-natives/src/chunk/ast_elixir.rs +++ /dev/null @@ -1,159 +0,0 @@ -//! Language-specific chunk classifier for Elixir. - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier, StructuralOverrides}, - common::*, - kind::ChunkKind, -}; - -pub struct ElixirClassifier; - -/// Extract the call target: `target` field, or first named child. -fn call_target(node: Node<'_>, source: &str) -> Option { - node - .child_by_field_name("target") - .or_else(|| named_children(node).into_iter().next()) - .map(|n| node_text(source, n.start_byte(), n.end_byte()).to_string()) -} - -/// Classify an Elixir `call` node based on its target keyword. -fn classify_call<'t>(node: Node<'t>, source: &str, at_root: bool) -> RawChunkCandidate<'t> { - let target = call_target(node, source).unwrap_or_default(); - let name = || call_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - match target.as_str() { - "defmodule" => make_container_chunk( - node, - ChunkKind::Module, - Some(name()), - source, - recurse_body(node, ChunkContext::ClassBody), - ), - "defprotocol" => make_container_chunk( - node, - ChunkKind::Proto, - Some(name()), - source, - recurse_body(node, ChunkContext::ClassBody), - ), - "defimpl" => make_container_chunk( - node, - ChunkKind::Impl, - Some(name()), - source, - recurse_body(node, ChunkContext::ClassBody), - ), - "def" | "defp" | "defdelegate" | "defguard" | "defguardp" | "defn" | "defnp" => { - make_kind_chunk( - node, - ChunkKind::Function, - Some(name()), - source, - recurse_body(node, ChunkContext::FunctionBody), - ) - }, - "defmacro" | "defmacrop" => make_kind_chunk( - node, - ChunkKind::Macro, - Some(name()), - source, - recurse_body(node, ChunkContext::FunctionBody), - ), - "alias" | "import" | "require" | "use" => group_candidate(node, ChunkKind::Imports, source), - "defstruct" | "defexception" => group_candidate(node, ChunkKind::Declarations, source), - "if" | "unless" => positional_candidate(node, ChunkKind::If, source), - "case" | "cond" | "receive" => positional_candidate(node, ChunkKind::Switch, source), - "for" => positional_candidate(node, ChunkKind::For, source), - "try" | "with" => positional_candidate(node, ChunkKind::Block, source), - _ if at_root => group_candidate(node, ChunkKind::Statements, source), - _ => group_candidate(node, ChunkKind::Block, source), - } -} - -/// Extract the name from an Elixir `call` node. -/// -/// Skips keyword-only calls (imports, control flow) that have no meaningful -/// identifier, then returns the first non-`do_block` named child after the -/// target. -fn call_name(node: Node<'_>, source: &str) -> Option { - let target = call_target(node, source)?; - if matches!( - target.as_str(), - "alias" - | "import" - | "require" - | "use" - | "if" | "case" - | "cond" - | "for" - | "try" - | "with" - | "unless" - | "receive" - ) { - return None; - } - - // The first named child after the target is typically `arguments`. - // For `def run(x)`, arguments contains a `call` node whose target is `run`. - // For `defmodule App`, arguments contains an `alias` node with text `App`. - // For `def run(x) when is_integer(x)`, arguments contains a `binary_operator` - // with the call on the left and the guard on the right. - // Extract the meaningful name, not the full text with parameters. - named_children(node).into_iter().skip(1).find_map(|child| { - if child.kind() == "do_block" { - return None; - } - if child.kind() == "arguments" { - // Dig into arguments to find the actual name. - return named_children(child).into_iter().next().and_then(|arg| { - if arg.kind() == "call" { - // `def run(x)` → arguments has call(target=run), extract target name - call_target(arg, source).and_then(|t| sanitize_identifier(&t)) - } else if arg.kind() == "binary_operator" { - // `def run(x) when guard` → binary_operator(left=call, right=guard) - // Extract name from the left side (the actual function call). - arg.child_by_field_name("left").and_then(|left| { - if left.kind() == "call" { - call_target(left, source).and_then(|t| sanitize_identifier(&t)) - } else { - sanitize_identifier(node_text(source, left.start_byte(), left.end_byte())) - } - }) - } else { - // `defmodule App` → arguments has alias("App") - sanitize_identifier(node_text(source, arg.start_byte(), arg.end_byte())) - } - }); - } - sanitize_identifier(node_text(source, child.start_byte(), child.end_byte())) - }) -} - -impl LangClassifier for ElixirClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: StructuralOverrides { - extra_trivia: &["unary_operator"], - preserved_trivia: &[], - extra_root_wrappers: &[], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - (node.kind() == "call").then(|| classify_call(node, source, context == ChunkContext::Root)) - } -} diff --git a/crates/pi-natives/src/chunk/ast_erlang.rs b/crates/pi-natives/src/chunk/ast_erlang.rs deleted file mode 100644 index 87d47a7b8..000000000 --- a/crates/pi-natives/src/chunk/ast_erlang.rs +++ /dev/null @@ -1,242 +0,0 @@ -//! Language-specific chunk classifier for Erlang. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct ErlangClassifier; - -fn find_named_descendant_by_kind<'t>(node: Node<'t>, kinds: &[&str]) -> Option> { - if kinds.iter().any(|kind| node.kind() == *kind) { - return Some(node); - } - - for child in named_children(node) { - if let Some(found) = find_named_descendant_by_kind(child, kinds) { - return Some(found); - } - } - - None -} - -fn named_text(node: Node<'_>, source: &str) -> Option { - sanitize_identifier(node_text(source, node.start_byte(), node.end_byte())) -} - -fn erlang_name(node: Node<'_>, source: &str) -> Option { - let name_node = match node.kind() { - "module_attribute" | "record_decl" | "record_field" => { - child_by_field_or_kind(node, &["name"], &["atom"]) - }, - "type_alias" => node - .child_by_field_name("name") - .and_then(|name| find_named_descendant_by_kind(name, &["atom"])), - "spec" => child_by_field_or_kind(node, &["fun"], &["atom"]), - "pp_define" => node - .child_by_field_name("lhs") - .and_then(|lhs| find_named_descendant_by_kind(lhs, &["var"])), - "fun_decl" => node - .child_by_field_name("clause") - .and_then(|clause| child_by_field_or_kind(clause, &["name"], &["atom"])), - "function_clause" => child_by_field_or_kind(node, &["name"], &["atom"]), - _ => child_by_kind(node, &["atom", "var"]), - }?; - - named_text(name_node, source) -} - -fn recurse_clause_body(node: Node<'_>) -> Option> { - recurse_into(node, ChunkContext::FunctionBody, &["body"], &["clause_body"]) -} - -impl LangClassifier for ErlangClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[ - semantic_rule( - "export_attribute", - ChunkKind::Exports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "export_type_attribute", - ChunkKind::Exports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "import_attribute", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "pp_include", - ChunkKind::Includes, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "pp_include_lib", - ChunkKind::Includes, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - ], - class: &[], - function: &[], - structural_overrides: StructuralOverrides { - extra_trivia: &[], - preserved_trivia: &[], - extra_root_wrappers: &[], - preserved_root_wrappers: &[], - absorbable_attrs: &["spec"], - }, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_erlang_root(node, source), - ChunkContext::ClassBody => classify_erlang_class(node, source), - ChunkContext::FunctionBody => classify_erlang_function(node, source), - } - } -} - -fn classify_erlang_root<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "module_attribute" => { - make_kind_chunk(node, ChunkKind::Module, erlang_name(node, source), source, None) - }, - "pp_define" => { - make_kind_chunk(node, ChunkKind::Macro, erlang_name(node, source), source, None) - }, - "record_decl" => make_candidate( - node, - ChunkKind::Struct, - format!("record_{}", erlang_name(node, source)?), - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::ClassBody)), - source, - ), - "type_alias" => { - make_kind_chunk(node, ChunkKind::Type, erlang_name(node, source), source, None) - }, - // The Erlang grammar exposes each top-level clause as its own `fun_decl`. - // Keep that shape instead of inventing a synthetic merged function node. - "fun_decl" => make_kind_chunk( - node, - ChunkKind::Function, - erlang_name(node, source), - source, - Some(recurse_self(node, ChunkContext::FunctionBody)), - ), - "spec" => return None, - _ => return None, - }) -} - -fn classify_erlang_class<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "record_field" => { - make_kind_chunk(node, ChunkKind::Field, erlang_name(node, source), source, None) - }, - _ => return None, - }) -} - -fn classify_erlang_function<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "function_clause" => make_kind_chunk( - node, - ChunkKind::Clause, - erlang_name(node, source), - source, - recurse_clause_body(node), - ), - "fun_clause" | "cr_clause" => make_candidate( - node, - ChunkKind::Clause, - None, - NameStyle::Named, - signature_for_node(node, source), - recurse_clause_body(node), - source, - ), - "receive_after" => make_candidate( - node, - ChunkKind::After, - None, - NameStyle::Named, - signature_for_node(node, source), - recurse_clause_body(node), - source, - ), - "catch_clause" => make_candidate( - node, - ChunkKind::Catch, - None, - NameStyle::Named, - signature_for_node(node, source), - recurse_clause_body(node), - source, - ), - "receive_expr" => make_candidate( - node, - ChunkKind::Receive, - None, - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::FunctionBody)), - source, - ), - "case_expr" => make_candidate( - node, - ChunkKind::Case, - None, - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::FunctionBody)), - source, - ), - "try_expr" => make_candidate( - node, - ChunkKind::Try, - None, - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::FunctionBody)), - source, - ), - "anonymous_fun" => make_kind_chunk( - node, - ChunkKind::Function, - Some("anonymous".to_string()), - source, - Some(recurse_self(node, ChunkContext::FunctionBody)), - ), - _ => return None, - }) -} diff --git a/crates/pi-natives/src/chunk/ast_go.rs b/crates/pi-natives/src/chunk/ast_go.rs deleted file mode 100644 index 3355eaddb..000000000 --- a/crates/pi-natives/src/chunk/ast_go.rs +++ /dev/null @@ -1,320 +0,0 @@ -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct GoClassifier; - -const ROOT_RULES: &[super::classify::SemanticRule] = &[ - // ── Imports / package ── - semantic_rule( - "package_clause", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::None, - ), - semantic_rule( - "import_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // ── Functions ── - semantic_rule( - "function_declaration", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "method_declaration", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // ── Statements ── - semantic_rule( - "expression_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "go_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "defer_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "send_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const CLASS_RULES: &[super::classify::SemanticRule] = &[ - // ── Methods ── - semantic_rule( - "method_spec", - ChunkKind::Method, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::None, - ), - // ── Field / method lists ── - semantic_rule( - "field_declaration_list", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "method_spec_list", - ChunkKind::Methods, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const FUNCTION_RULES: &[super::classify::SemanticRule] = &[ - // ── Control flow ── - semantic_rule( - "if_statement", - ChunkKind::If, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "for_statement", - ChunkKind::For, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "switch_statement", - ChunkKind::Switch, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "expression_switch_statement", - ChunkKind::Switch, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "type_switch_statement", - ChunkKind::Switch, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "select_statement", - ChunkKind::Switch, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // ── Statements ── - semantic_rule( - "go_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "defer_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "send_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const GO_TABLES: ClassifierTables = ClassifierTables { - root: ROOT_RULES, - class: CLASS_RULES, - function: FUNCTION_RULES, - structural_overrides: StructuralOverrides::EMPTY, -}; - -impl LangClassifier for GoClassifier { - fn tables(&self) -> &'static ClassifierTables { - &GO_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_root_custom(node, source), - ChunkContext::ClassBody => classify_class_custom(node, source), - ChunkContext::FunctionBody => classify_function_custom(node, source), - } - } -} - -fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - // ── Variables ── - "const_declaration" | "var_declaration" | "short_var_declaration" => { - Some(match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None), - None => group_candidate(node, ChunkKind::Declarations, source), - }) - }, - - // ── Containers ── - "type_declaration" => Some(classify_type_decl(node, source)), - - // ── Control flow (top-level scripts) ── - "if_statement" - | "switch_statement" - | "expression_switch_statement" - | "type_switch_statement" - | "select_statement" - | "for_statement" => Some(classify_function_go(node, source)), - - _ => None, - } -} - -fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - // ── Fields ── - "field_declaration" | "embedded_field" => Some(match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None), - None => group_candidate(node, ChunkKind::Fields, source), - }), - _ => None, - } -} - -fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - // ── Variables ── - "short_var_declaration" | "var_declaration" | "const_declaration" => { - let span = line_span(node.start_position().row + 1, node.end_position().row + 1); - Some(if span > 1 { - if let Some(name) = extract_identifier(node, source) { - make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None) - } else { - group_from_sanitized(node, source) - } - } else { - group_from_sanitized(node, source) - }) - }, - _ => None, - } -} - -/// Classify Go function-level nodes (reused for top-level control flow -/// delegation). -fn classify_function_go<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody); - match node.kind() { - "if_statement" => { - make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source) - }, - "switch_statement" - | "expression_switch_statement" - | "type_switch_statement" - | "select_statement" => { - make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source) - }, - "for_statement" => { - make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source) - }, - _ => group_candidate(node, ChunkKind::Statements, source), - } -} - -/// Classify Go `type_declaration` nodes. -/// -/// A single `type_spec` with a struct/interface body becomes a container; -/// a single `type_spec` without one becomes a named leaf. -/// Multiple `type_spec` children (type group) become a group. -fn classify_type_decl<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let specs: Vec> = named_children(node) - .into_iter() - .filter(|c| c.kind() == "type_spec") - .collect(); - - if specs.len() == 1 { - let spec = specs[0]; - let name = extract_identifier(spec, source).unwrap_or_else(|| "anonymous".to_string()); - if let Some(recurse) = recurse_type_spec(spec) { - return make_container_chunk_from( - node, - spec, - ChunkKind::Type, - Some(name), - source, - Some(recurse), - ); - } - return make_kind_chunk_from(node, spec, ChunkKind::Type, Some(name), source, None); - } - - group_candidate(node, ChunkKind::Declarations, source) -} - -fn group_from_sanitized<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let sanitized = sanitize_node_kind(node.kind()); - let kind = ChunkKind::from_sanitized_kind(sanitized); - let identifier = if kind == ChunkKind::Chunk { - Some(sanitized.to_string()) - } else { - None - }; - make_candidate(node, kind, identifier, NameStyle::Group, None, None, source) -} - -/// For a `type_spec`, find a `struct_type` or `interface_type` child and return -/// its body (`field_declaration_list` or `method_spec_list`) as a recurse spec. -fn recurse_type_spec(node: Node<'_>) -> Option> { - let container = child_by_kind(node, &["struct_type", "interface_type"])?; - let body = child_by_kind(container, &["field_declaration_list", "method_spec_list"]) - .unwrap_or(container); - Some(RecurseSpec { node: body, context: ChunkContext::ClassBody }) -} diff --git a/crates/pi-natives/src/chunk/ast_graphql.rs b/crates/pi-natives/src/chunk/ast_graphql.rs deleted file mode 100644 index 30db6a24f..000000000 --- a/crates/pi-natives/src/chunk/ast_graphql.rs +++ /dev/null @@ -1,294 +0,0 @@ -//! GraphQL-specific chunk classifier. - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier, StructuralOverrides}, - common::*, - kind::ChunkKind, -}; - -pub struct GraphqlClassifier; - -impl LangClassifier for GraphqlClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: StructuralOverrides { - extra_trivia: &["comma"], - preserved_trivia: &[], - extra_root_wrappers: &[ - "document", - "definition", - "type_system_definition", - "type_definition", - "executable_definition", - ], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_graphql_root(node, source), - ChunkContext::ClassBody => classify_graphql_class(node, source), - ChunkContext::FunctionBody => classify_graphql_function(node, source), - } - } -} - -fn classify_graphql_root<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "schema_definition" => Some(make_container_chunk( - node, - ChunkKind::Schema, - None, - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - )), - "directive_definition" => Some(make_named_graphql_chunk( - node, - ChunkKind::Directive, - extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string()), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["arguments_definition"]), - )), - "scalar_type_definition" => Some(make_named_graphql_chunk( - node, - ChunkKind::Type, - format!( - "scalar_{}", - extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string()) - ), - source, - None, - )), - "object_type_definition" => Some(make_container_chunk( - node, - ChunkKind::Type, - extract_graphql_name(node, source), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["fields_definition"]), - )), - "interface_type_definition" => Some(make_container_chunk( - node, - ChunkKind::Interface, - extract_graphql_name(node, source), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["fields_definition"]), - )), - "union_type_definition" => Some(make_kind_chunk( - node, - ChunkKind::Union, - extract_graphql_name(node, source), - source, - None, - )), - "enum_type_definition" => Some(make_container_chunk( - node, - ChunkKind::Enum, - extract_graphql_name(node, source), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["enum_values_definition"]), - )), - "input_object_type_definition" => Some(make_named_graphql_chunk( - node, - ChunkKind::Type, - format!( - "input_{}", - extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string()) - ), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["input_fields_definition"]), - )), - "operation_definition" => Some(make_named_graphql_chunk( - node, - ChunkKind::Operation, - extract_graphql_operation_chunk_name(node, source), - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["selection_set"]), - )), - "fragment_definition" => Some(make_named_graphql_chunk( - node, - ChunkKind::Operation, - format!( - "fragment_{}", - extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string()) - ), - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["selection_set"]), - )), - _ => None, - } -} - -fn classify_graphql_class<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "root_operation_type_definition" => Some(make_named_graphql_chunk( - node, - ChunkKind::Root, - extract_graphql_operation_type(node, source).unwrap_or_else(|| "anonymous".to_string()), - source, - None, - )), - "field_definition" => { - let name = extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - let recurse = recurse_into(node, ChunkContext::ClassBody, &[], &["arguments_definition"]); - Some(match recurse { - Some(recurse) => { - make_container_chunk(node, ChunkKind::Field, Some(name), source, Some(recurse)) - }, - None => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None), - }) - }, - "input_value_definition" => Some(classify_graphql_input_value(node, source)), - "enum_value_definition" => Some(make_named_graphql_chunk( - node, - ChunkKind::Variant, - format!( - "value_{}", - extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string()) - ), - source, - None, - )), - _ => None, - } -} - -fn classify_graphql_function<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "selection" => classify_graphql_selection(node, source), - _ => None, - } -} - -fn classify_graphql_selection<'t>(node: Node<'t>, source: &str) -> Option> { - let child = first_named_child(node)?; - match child.kind() { - "field" => { - let name = extract_graphql_name(child, source).unwrap_or_else(|| "anonymous".to_string()); - let recurse = recurse_into(child, ChunkContext::FunctionBody, &[], &["selection_set"]); - Some(match recurse { - Some(recurse) => make_container_chunk_from( - node, - child, - ChunkKind::Field, - Some(name), - source, - Some(recurse), - ), - None => make_kind_chunk_from(node, child, ChunkKind::Field, Some(name), source, None), - }) - }, - "fragment_spread" => Some(make_named_graphql_chunk_from( - node, - child, - ChunkKind::Operation, - format!( - "spread_{}", - extract_graphql_name(child, source).unwrap_or_else(|| "anonymous".to_string()) - ), - source, - None, - )), - "inline_fragment" => Some(make_container_chunk_from( - node, - child, - ChunkKind::InlineFragment, - None, - source, - recurse_into(child, ChunkContext::FunctionBody, &[], &["selection_set"]), - )), - _ => None, - } -} - -fn classify_graphql_input_value<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let name = extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - match node.parent().map(|parent| parent.kind()) { - Some("input_fields_definition") => { - make_kind_chunk(node, ChunkKind::Field, Some(name), source, None) - }, - _ => make_kind_chunk(node, ChunkKind::Arg, Some(name), source, None), - } -} - -fn make_named_graphql_chunk<'t>( - node: Node<'t>, - kind: ChunkKind, - identifier: impl Into>, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'t> { - make_candidate( - node, - kind, - identifier, - NameStyle::Named, - signature_for_node(node, source), - recurse, - source, - ) -} - -fn make_named_graphql_chunk_from<'t>( - range_node: Node<'t>, - signature_node: Node<'t>, - kind: ChunkKind, - identifier: impl Into>, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'t> { - make_candidate( - range_node, - kind, - identifier, - NameStyle::Named, - signature_for_node(signature_node, source), - recurse, - source, - ) -} - -fn extract_graphql_name(node: Node<'_>, source: &str) -> Option { - find_graphql_name_node(node) - .and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))) -} - -fn find_graphql_name_node(node: Node<'_>) -> Option> { - match node.kind() { - "name" | "fragment_name" => Some(node), - _ => named_children(node) - .into_iter() - .find_map(find_graphql_name_node), - } -} - -fn extract_graphql_operation_type(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["operation_type"]) - .and_then(|kind| sanitize_identifier(node_text(source, kind.start_byte(), kind.end_byte()))) -} - -fn extract_graphql_operation_chunk_name(node: Node<'_>, source: &str) -> String { - let operation = - extract_graphql_operation_type(node, source).unwrap_or_else(|| "operation".to_string()); - match extract_graphql_name(node, source) { - Some(name) => format!("{operation}_{name}"), - None => operation, - } -} - -fn first_named_child(node: Node<'_>) -> Option> { - (0..node.named_child_count()).find_map(|index| node.named_child(index)) -} diff --git a/crates/pi-natives/src/chunk/ast_haskell_scala.rs b/crates/pi-natives/src/chunk/ast_haskell_scala.rs deleted file mode 100644 index b01584441..000000000 --- a/crates/pi-natives/src/chunk/ast_haskell_scala.rs +++ /dev/null @@ -1,195 +0,0 @@ -//! Language-specific chunk classifiers for Haskell and Scala. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct HaskellScalaClassifier; - -const HASKELL_SCALA_ROOT_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "import_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "package_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "module", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "function_declaration", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "function_definition", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "class_definition", - ChunkKind::Class, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "object_definition", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "trait_definition", - ChunkKind::Iface, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "type_alias_declaration", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "type_item", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "variable_declaration", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "assignment", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "expression_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const HASKELL_SCALA_FUNCTION_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "if_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "match_expression", - ChunkKind::Match, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "for_expression", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "while_expression", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "block_expression", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), -]; - -const HASKELL_SCALA_TABLES: ClassifierTables = ClassifierTables { - root: HASKELL_SCALA_ROOT_RULES, - class: &[], - function: HASKELL_SCALA_FUNCTION_RULES, - structural_overrides: super::classify::StructuralOverrides::EMPTY, -}; - -impl LangClassifier for HaskellScalaClassifier { - fn tables(&self) -> &'static ClassifierTables { - &HASKELL_SCALA_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - if context != ChunkContext::ClassBody { - return None; - } - - match node.kind() { - "function_declaration" | "function_definition" | "method_definition" => { - let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string()); - let kind = if name == "constructor" { - ChunkKind::Constructor - } else { - ChunkKind::Function - }; - let identifier = (kind != ChunkKind::Constructor).then_some(name); - Some(make_kind_chunk( - node, - kind, - identifier, - source, - resolve_recurse(node, ChunkContext::FunctionBody), - )) - }, - "variable_declaration" | "property_declaration" => { - Some(extract_identifier(node, source).map_or_else( - || group_candidate(node, ChunkKind::Fields, source), - |name| make_kind_chunk(node, ChunkKind::Field, Some(name), source, None), - )) - }, - _ => None, - } - } -} diff --git a/crates/pi-natives/src/chunk/ast_html_xml.rs b/crates/pi-natives/src/chunk/ast_html_xml.rs deleted file mode 100644 index 05d9d5c65..000000000 --- a/crates/pi-natives/src/chunk/ast_html_xml.rs +++ /dev/null @@ -1,163 +0,0 @@ -//! Language-specific chunk classifiers for HTML and XML. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule, - }, - common::*, - kind::ChunkKind, -}; -use crate::language::SupportLang; - -pub struct HtmlXmlClassifier; - -const HTML_XML_SHARED_RULES: &[super::classify::SemanticRule] = &[semantic_rule( - "text_node", - ChunkKind::Text, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, -)]; - -const HTML_XML_TABLES: ClassifierTables = ClassifierTables { - root: HTML_XML_SHARED_RULES, - class: HTML_XML_SHARED_RULES, - function: &[], - structural_overrides: super::classify::StructuralOverrides::EMPTY, -}; - -/// Classify an element-like node as a container with tag semantics. -/// -/// Uses `extract_markup_tag_name` directly because the shared -/// `extract_identifier` does not handle HTML/XML start-tag structures. -fn classify_element<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "script_element" => Some(classify_script_element(node, source)), - "style_element" => Some(classify_style_element(node, source)), - "element" => { - let tag_name = - extract_markup_tag_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - // HTML: child elements are direct children of `element`. - // XML: child elements are inside a `content` wrapper node. - let recurse_target = child_by_kind(node, &["content"]).unwrap_or(node); - Some(force_container(make_container_chunk( - node, - ChunkKind::Tag, - Some(tag_name), - source, - Some(recurse_self(recurse_target, ChunkContext::ClassBody)), - ))) - }, - "text_node" => Some(group_candidate(node, ChunkKind::Text, source)), - _ => None, - } -} - -fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - classify_injected_raw_text_block(node, ChunkKind::Script, source, SupportLang::JavaScript) -} - -fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - classify_injected_raw_text_block(node, ChunkKind::Style, source, SupportLang::Css) -} - -fn classify_injected_raw_text_block<'t>( - node: Node<'t>, - kind: ChunkKind, - source: &str, - default_language: SupportLang, -) -> RawChunkCandidate<'t> { - let Some(content_node) = child_by_kind(node, &["raw_text"]) else { - return positional_candidate(node, kind, source); - }; - - let candidate = with_region_node(positional_candidate(node, kind, source), Some(content_node)); - match resolve_embedded_language(node, source, default_language) { - Some(language) => with_injected_subtree(candidate, language, content_node), - None => candidate, - } -} - -/// Extract the tag name from an HTML/XML element node. -/// -/// HTML: `element` → `start_tag`/`self_closing_tag` → `tag_name` -/// XML (tree-sitter-xml): `element` → `STag`/`EmptyElemTag` → `Name` -fn extract_markup_tag_name(node: Node<'_>, source: &str) -> Option { - named_children(node).into_iter().find_map(|child| { - let tag_name_kinds: &[&str] = match child.kind() { - // HTML - "start_tag" | "self_closing_tag" => &["tag_name"], - // XML (tree-sitter-xml grammar) - "STag" | "EmptyElemTag" => &["Name"], - _ => return None, - }; - child_by_kind(child, tag_name_kinds) - .and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte()))) - }) -} - -fn resolve_embedded_language( - node: Node<'_>, - source: &str, - default_language: SupportLang, -) -> Option { - if let Some(language) = attribute_value(node, "lang", source) { - return SupportLang::from_alias(language.as_str()); - } - Some(default_language) -} - -fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option { - let start = start_like(node)?; - for child in named_children(start) { - if child.kind() != "attribute" { - continue; - } - if extract_attribute_name(child, source).as_deref() != Some(name) { - continue; - } - if let Some(value) = child_by_kind(child, &["attribute_value", "quoted_attribute_value"]) { - return sanitize_identifier(&unquote_text(node_text( - source, - value.start_byte(), - value.end_byte(), - ))); - } - return Some(name.to_string()); - } - None -} - -fn extract_attribute_name(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["attribute_name"]) - .and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))) -} - -fn start_like(node: Node<'_>) -> Option> { - child_by_kind(node, &["start_tag", "self_closing_tag"]) -} - -const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> { - candidate.force_recurse = true; - candidate -} - -impl LangClassifier for HtmlXmlClassifier { - fn tables(&self) -> &'static ClassifierTables { - &HTML_XML_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - if matches!(context, ChunkContext::Root | ChunkContext::ClassBody) { - return classify_element(node, source); - } - None - } -} diff --git a/crates/pi-natives/src/chunk/ast_ini.rs b/crates/pi-natives/src/chunk/ast_ini.rs deleted file mode 100644 index ef5dc1e43..000000000 --- a/crates/pi-natives/src/chunk/ast_ini.rs +++ /dev/null @@ -1,88 +0,0 @@ -//! Chunk classifier for INI. -//! -//! The tree-sitter INI grammar is intentionally flat: a document contains -//! root-level `setting` nodes and `section` containers, and a section contains -//! only its own `setting` children. Mirror that structure directly instead of -//! inventing deeper hierarchy. - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier}, - common::*, - kind::ChunkKind, -}; - -pub struct IniClassifier; - -impl LangClassifier for IniClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: super::classify::StructuralOverrides::EMPTY, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_ini_root(node, source), - ChunkContext::ClassBody => classify_ini_class(node, source), - ChunkContext::FunctionBody => None, - } - } -} - -fn classify_ini_root<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "section" => make_container_chunk( - node, - ChunkKind::Section, - Some(ini_name(node, source)?), - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - ), - // INI permits settings before any section header; keep them as first-class - // chunks instead of forcing them under a synthetic container. - "setting" => { - make_kind_chunk(node, ChunkKind::Key, Some(ini_name(node, source)?), source, None) - }, - _ => return None, - }) -} - -fn classify_ini_class<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "setting" => { - make_kind_chunk(node, ChunkKind::Key, Some(ini_name(node, source)?), source, None) - }, - _ => return None, - }) -} - -fn ini_name(node: Node<'_>, source: &str) -> Option { - find_named_text(node, source, &["section_name", "setting_name", "text"]).and_then(|text| { - sanitize_identifier(text.trim().trim_start_matches('[').trim_end_matches(']')) - }) -} - -fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> { - if kinds.iter().any(|kind| node.kind() == *kind) { - return Some(node_text(source, node.start_byte(), node.end_byte())); - } - - for child in named_children(node) { - if let Some(text) = find_named_text(child, source, kinds) { - return Some(text); - } - } - - None -} diff --git a/crates/pi-natives/src/chunk/ast_ipynb.rs b/crates/pi-natives/src/chunk/ast_ipynb.rs deleted file mode 100644 index 7af7e604d..000000000 --- a/crates/pi-natives/src/chunk/ast_ipynb.rs +++ /dev/null @@ -1,918 +0,0 @@ -//! Jupyter notebook (`.ipynb`) chunker. -//! -//! Notebooks are JSON documents whose `cells` array carries the code/markdown -//! that users actually edit. This module parses the JSON, extracts each cell -//! as its own source fragment, and assembles a *virtual source* — the -//! concatenation of all cell bodies with one-line marker headers — that the -//! rest of the chunk pipeline can treat like any other text file. -//! -//! The per-cell sub-chunks come from recursively running [`build_chunk_tree`] -//! on the individual cell sources with their appropriate language (code cells -//! use the notebook's kernel language, defaulting to Python; markdown cells -//! use `markdown`; raw cells fall through to the blank-line fallback). Their -//! byte and line offsets are shifted into the virtual source and then -//! rewrapped under `cell_` parent chunks so the resulting tree looks to -//! edit.rs exactly like a normal multi-symbol file. -//! -//! On write-back, [`notebook_to_json`] walks the (possibly edited) virtual -//! source, splits it at the cell markers, updates each cell's `source` field -//! in a preserved [`NotebookContext`], and serializes the whole notebook back -//! to JSON. Cell metadata (`metadata`, `outputs`, `execution_count`, `id`, -//! attachments, etc.) is preserved verbatim. - -use std::sync::Arc; - -use serde::Serialize; -use serde_json::{Map, Value}; - -use crate::chunk::{ - build_chunk_tree, chunk_checksum, - kind::ChunkKind, - line_start_offsets, - types::{ChunkNode, ChunkTree}, -}; - -/// Marker line prefix placed before every cell body in the virtual source. -/// -/// Format: `# %%% oh-my-pi cell_ []` -/// -/// The leading `#` makes the marker a valid comment in Python and most other -/// code languages, and the `oh-my-pi` tag makes accidental collision with -/// user content vanishingly unlikely. Markdown cells get the same marker — -/// `#` in markdown is a heading, but the marker line itself is stripped -/// before the markdown chunker parses the cell body (see -/// [`build_cell_sub_tree`]). -const MARKER_PREFIX: &str = "# %%% oh-my-pi cell_"; - -/// Returns true if `line` is a cell marker; parses the cell index and type. -fn parse_marker_line(line: &str) -> Option<(usize, &str)> { - let rest = line.strip_prefix(MARKER_PREFIX)?; - let (num_str, after_num) = rest.split_once(' ')?; - let index: usize = num_str.parse().ok()?; - let cell_type = after_num.strip_prefix('[')?.strip_suffix(']')?; - Some((index, cell_type)) -} - -fn format_marker(index: usize, cell_type: &str) -> String { - format!("{MARKER_PREFIX}{index} [{cell_type}]") -} - -/// Metadata for a single notebook cell. Everything except the joined `source` -/// is preserved verbatim for JSON round-tripping. -#[derive(Clone)] -pub struct NotebookCell { - pub cell_type: String, - pub source: String, - pub metadata: Value, - pub outputs: Option, - pub execution_count: Option, - /// Additional fields on the cell object (`id`, `attachments`, …) so we - /// emit the same keys we consumed. - pub other: Map, - /// Whether the original cell used `"source"` as a string (`true`) or an - /// array of lines (`false`). Preserved so round-trips stay byte-identical - /// when only metadata or unrelated cells change. - pub source_was_string: bool, -} - -/// Preserved notebook-level state used for rebuilding the JSON after edits. -#[derive(Clone)] -pub struct NotebookContext { - /// Cells in document order. - pub cells: Vec, - /// Top-level notebook fields other than `cells`, in original key order. - pub top_fields: Map, - /// Indent string detected from the original JSON (typically `" "`). - pub indent: String, - /// Whether the original file ended with a trailing `\n`. - pub trailing_newline: bool, - /// Normalized kernel language used for code cells (e.g. `python`). - pub kernel_language: String, -} - -/// Result of a notebook parse: the virtual source ready for chunking plus -/// the context needed to rebuild the JSON later. -pub struct NotebookParse { - pub virtual_source: String, - pub context: NotebookContext, -} - -// ──────────────────────────────────────────────────────────────────── -// JSON parsing -// ──────────────────────────────────────────────────────────────────── - -/// Parse the raw ipynb JSON into a [`NotebookContext`] and the derived -/// virtual source text. Returns a descriptive error if the JSON is invalid -/// or not a notebook document. -pub fn parse_notebook(source: &str) -> Result { - let normalized = strip_bom(source); - let value: Value = serde_json::from_str(normalized) - .map_err(|err| format!("Invalid Jupyter notebook JSON: {err}"))?; - let Value::Object(obj) = value else { - return Err("Invalid Jupyter notebook: top-level value must be an object.".to_string()); - }; - - let cells_val = obj - .get("cells") - .ok_or_else(|| "Invalid Jupyter notebook: missing `cells` array.".to_string())?; - let cells_arr = cells_val - .as_array() - .ok_or_else(|| "Invalid Jupyter notebook: `cells` is not an array.".to_string())?; - - let mut cells = Vec::with_capacity(cells_arr.len()); - for (i, raw) in cells_arr.iter().enumerate() { - cells.push(parse_cell(raw, i)?); - } - - // Top-level fields minus `cells`, preserving insertion order. - let mut top_fields = Map::new(); - for (k, v) in &obj { - if k != "cells" { - top_fields.insert(k.clone(), v.clone()); - } - } - - let kernel_language = - extract_kernel_language(&top_fields).unwrap_or_else(|| "python".to_string()); - let indent = detect_json_indent(normalized); - let trailing_newline = normalized.ends_with('\n'); - - let ctx = NotebookContext { cells, top_fields, indent, trailing_newline, kernel_language }; - let virtual_source = build_virtual_source(&ctx); - Ok(NotebookParse { virtual_source, context: ctx }) -} - -fn parse_cell(raw: &Value, index: usize) -> Result { - let obj = raw - .as_object() - .ok_or_else(|| format!("Invalid Jupyter notebook: cell {} is not an object.", index + 1))?; - - let cell_type = obj - .get("cell_type") - .and_then(Value::as_str) - .unwrap_or("code") - .to_string(); - - let (source, source_was_string) = match obj.get("source") { - Some(Value::String(s)) => (s.clone(), true), - Some(Value::Array(arr)) => { - let mut joined = String::new(); - for item in arr { - match item { - Value::String(s) => joined.push_str(s), - _ => { - return Err(format!( - "Invalid Jupyter notebook: cell {} source array contains non-string element.", - index + 1 - )); - }, - } - } - (joined, false) - }, - Some(Value::Null) | None => (String::new(), false), - Some(_) => { - return Err(format!( - "Invalid Jupyter notebook: cell {} has a non-string `source` field.", - index + 1 - )); - }, - }; - - let metadata = obj - .get("metadata") - .cloned() - .unwrap_or_else(|| Value::Object(Map::new())); - let outputs = obj.get("outputs").cloned(); - let execution_count = obj.get("execution_count").cloned(); - - // Additional fields (id, attachments, …) that we pass through untouched. - let mut other = Map::new(); - for (k, v) in obj { - if !matches!(k.as_str(), "cell_type" | "source" | "metadata" | "outputs" | "execution_count") - { - other.insert(k.clone(), v.clone()); - } - } - - Ok(NotebookCell { - cell_type, - source, - metadata, - outputs, - execution_count, - other, - source_was_string, - }) -} - -fn strip_bom(source: &str) -> &str { - source.strip_prefix('\u{feff}').unwrap_or(source) -} - -fn extract_kernel_language(top_fields: &Map) -> Option { - if let Some(meta) = top_fields.get("metadata").and_then(Value::as_object) { - if let Some(lang) = meta - .get("kernelspec") - .and_then(Value::as_object) - .and_then(|k| k.get("language")) - .and_then(Value::as_str) - && !lang.is_empty() - { - return Some(lang.to_ascii_lowercase()); - } - if let Some(lang) = meta - .get("language_info") - .and_then(Value::as_object) - .and_then(|k| k.get("name")) - .and_then(Value::as_str) - && !lang.is_empty() - { - return Some(lang.to_ascii_lowercase()); - } - } - None -} - -fn detect_json_indent(source: &str) -> String { - // Look for the first `\n` followed by whitespace inside the top-level object - // (i.e. after the opening `{`). This is a heuristic; Jupyter canonically - // uses a single space per level. - let Some(brace) = source.find('{') else { - return " ".to_string(); - }; - let rest = &source[brace + 1..]; - let Some(nl) = rest.find('\n') else { - return " ".to_string(); - }; - let after_nl = &rest[nl + 1..]; - let mut end = 0usize; - for ch in after_nl.chars() { - if ch == ' ' || ch == '\t' { - end += ch.len_utf8(); - } else { - break; - } - } - if end == 0 { - " ".to_string() - } else { - after_nl[..end].to_string() - } -} - -// ──────────────────────────────────────────────────────────────────── -// Virtual source assembly -// ──────────────────────────────────────────────────────────────────── - -/// Build the virtual source text from the current cell list. -/// -/// Each cell is preceded by a marker line and its body. Cells do not include -/// trailing blank separators — we rely purely on the marker line to delimit -/// adjacent cells so the reconstructed sources stay byte-identical to the -/// originals after whole-cell edits. -pub fn build_virtual_source(ctx: &NotebookContext) -> String { - let mut out = String::new(); - for (i, cell) in ctx.cells.iter().enumerate() { - out.push_str(&format_marker(i + 1, &cell.cell_type)); - out.push('\n'); - out.push_str(&cell.source); - if !cell.source.is_empty() && !cell.source.ends_with('\n') { - out.push('\n'); - } - } - out -} - -// ──────────────────────────────────────────────────────────────────── -// Chunk tree construction -// ──────────────────────────────────────────────────────────────────── - -/// Locate every cell marker in `source`, returning tuples of: -/// (`cell_number`, `marker_line_byte_start`, `content_byte_start`, -/// `content_byte_end`, `marker_line_number_1based`, -/// `content_start_line_1based`, -/// `content_end_line_1based_inclusive_or_zero_if_empty`) -struct CellRegion { - cell_num: usize, - cell_type: String, - marker_start: usize, // byte offset of the `#` starting the marker line - content_start: usize, // byte offset of the first byte of the cell body - content_end: usize, // byte offset one past the last byte of the cell body - marker_line: u32, // 1-based line number of the marker - content_line: u32, // 1-based line number of the first body line (or marker_line + 1) - content_end_line: u32, /* 1-based line number of the last body line (== content_line - 1 - * for empty bodies) */ -} - -/// Scan a virtual source text for cell markers and return the list of -/// regions. Assumes markers occur at the very start of their line. -fn scan_cells(virtual_source: &str) -> Vec { - let line_starts = line_start_offsets(virtual_source); - let mut regions: Vec = Vec::new(); - - for (line_idx, &line_start) in line_starts.iter().enumerate() { - let line_end = if line_idx + 1 < line_starts.len() { - // Exclude the trailing newline - line_starts[line_idx + 1] - 1 - } else { - virtual_source.len() - }; - let line = &virtual_source[line_start..line_end]; - if let Some((cell_num, cell_type)) = parse_marker_line(line) { - // Close the previous region if any. - if let Some(prev) = regions.last_mut() { - prev.content_end = line_start; - // Trim trailing newline from content_end if present (i.e. the body ended with - // \n). Actually we keep the newline: cell bodies end with \n except - // possibly the last. content_end_line = line of the last body byte. - if prev.content_end > prev.content_start { - let body_last_char_line = line_idx; // line_idx is 0-based, so this is the previous line - prev.content_end_line = body_last_char_line as u32; - } else { - prev.content_end_line = prev.content_line.saturating_sub(1); - } - } - let content_start = if line_idx + 1 < line_starts.len() { - line_starts[line_idx + 1] - } else { - virtual_source.len() - }; - regions.push(CellRegion { - cell_num, - cell_type: cell_type.to_string(), - marker_start: line_start, - content_start, - content_end: virtual_source.len(), // provisional, closed by the next marker - marker_line: (line_idx as u32) + 1, - content_line: (line_idx as u32) + 2, - content_end_line: 0, - }); - } - } - - // Close the last region. - if let Some(last) = regions.last_mut() { - last.content_end = virtual_source.len(); - if last.content_end > last.content_start { - // Count lines inside the body. - let body = &virtual_source[last.content_start..last.content_end]; - let body_lines = body.matches('\n').count(); - // If body doesn't end with '\n', the final partial line still counts. - let has_trailing_nl = body.ends_with('\n'); - let content_lines = if has_trailing_nl { - body_lines - } else { - body_lines + 1 - }; - if content_lines > 0 { - last.content_end_line = last.content_line + content_lines as u32 - 1; - } else { - last.content_end_line = last.content_line.saturating_sub(1); - } - } else { - last.content_end_line = last.content_line.saturating_sub(1); - } - } - - regions -} - -/// Build a chunk tree from a virtual source text. -/// -/// Re-scans the virtual source for cell markers, parses each cell body with -/// its language, and wraps the results in `cell_` parent chunks. This is -/// the entry point used by both the initial JSON-based parse (via -/// [`parse_notebook`] → `build_virtual_source` → this function) and the -/// post-edit rebuilds that operate directly on the mutated virtual source. -pub fn build_notebook_tree_from_virtual( - virtual_source: &str, - kernel_language: &str, -) -> Result { - let total_lines = total_line_count(virtual_source); - let root_checksum = chunk_checksum(virtual_source.as_bytes()); - let regions = scan_cells(virtual_source); - - // Accumulated chunk nodes. Index 0 is reserved for the synthetic root. - let mut chunks: Vec = Vec::with_capacity(1 + regions.len() * 2); - chunks.push(ChunkNode { - path: String::new(), - identifier: None, - kind: ChunkKind::Root, - leaf: false, - virtual_content: None, - parent_path: None, - children: Vec::new(), - signature: None, - start_line: u32::from(total_lines != 0), - end_line: total_lines as u32, - line_count: total_lines as u32, - start_byte: 0, - end_byte: virtual_source.len() as u32, - checksum_start_byte: 0, - prologue_end_byte: Some(0), - epilogue_start_byte: Some(virtual_source.len() as u32), - checksum: root_checksum.clone(), - error: false, - indent: 0, - indent_char: String::new(), - group: false, - }); - - let mut root_children: Vec = Vec::with_capacity(regions.len()); - - for region in ®ions { - let cell_path = format!("cell_{}", region.cell_num); - let cell_language_str = match region.cell_type.as_str() { - "code" => kernel_language.to_string(), - "markdown" => "markdown".to_string(), - _ => String::new(), - }; - - let body = &virtual_source[region.content_start..region.content_end]; - let body_has_content = !body.is_empty(); - let cell_checksum = chunk_checksum(body.as_bytes()); - - // Build sub-chunks by parsing the cell body in isolation. Offsets in - // the returned tree are relative to `body`; we translate them into - // virtual-source coordinates by adding `region.content_start` bytes - // and `region.content_line - 1` lines. - let mut cell_children_paths: Vec = Vec::new(); - if body_has_content { - let sub_tree = build_chunk_tree(body, cell_language_str.as_str()) - .map_err(|err| format!("Failed to parse cell_{} body: {err}", region.cell_num))?; - for sub_chunk in sub_tree.chunks.into_iter().skip(1) { - let translated_path = format!("{}.{}", cell_path, sub_chunk.path); - let translated_parent = match sub_chunk.parent_path.as_deref() { - Some("") | None => Some(cell_path.clone()), - Some(other) => Some(format!("{cell_path}.{other}")), - }; - let translated_children: Vec = sub_chunk - .children - .iter() - .map(|c| format!("{cell_path}.{c}")) - .collect(); - let shifted_start_byte = sub_chunk - .start_byte - .saturating_add(region.content_start as u32); - let shifted_end_byte = sub_chunk - .end_byte - .saturating_add(region.content_start as u32); - let line_shift = region.content_line.saturating_sub(1); - chunks.push(ChunkNode { - path: translated_path.clone(), - identifier: sub_chunk.identifier, - kind: sub_chunk.kind, - leaf: sub_chunk.leaf, - virtual_content: sub_chunk.virtual_content, - parent_path: translated_parent, - children: translated_children, - signature: sub_chunk.signature, - start_line: sub_chunk.start_line.saturating_add(line_shift), - end_line: sub_chunk.end_line.saturating_add(line_shift), - line_count: sub_chunk.line_count, - start_byte: shifted_start_byte, - end_byte: shifted_end_byte, - checksum_start_byte: sub_chunk - .checksum_start_byte - .saturating_add(region.content_start as u32), - prologue_end_byte: sub_chunk - .prologue_end_byte - .map(|b| b.saturating_add(region.content_start as u32)), - epilogue_start_byte: sub_chunk - .epilogue_start_byte - .map(|b| b.saturating_add(region.content_start as u32)), - checksum: sub_chunk.checksum, - error: sub_chunk.error, - indent: sub_chunk.indent, - indent_char: sub_chunk.indent_char, - group: false, - }); - } - for sub_path in sub_tree.root_children { - cell_children_paths.push(format!("{cell_path}.{sub_path}")); - } - } - - let cell_line_count = { - let body_lines = if body_has_content { - if body.ends_with('\n') { - body.matches('\n').count() - } else { - body.matches('\n').count() + 1 - } - } else { - 0 - }; - 1 + body_lines as u32 - }; - let cell_end_line = region.marker_line + cell_line_count.saturating_sub(1); - let cell_leaf = cell_children_paths.is_empty(); - chunks.push(ChunkNode { - path: cell_path.clone(), - identifier: Some(cell_path.clone()), - kind: ChunkKind::Cell, - leaf: cell_leaf, - virtual_content: None, - parent_path: Some(String::new()), - children: cell_children_paths, - signature: Some(format!("cell_{} ({})", region.cell_num, region.cell_type)), - start_line: region.marker_line, - end_line: cell_end_line, - line_count: cell_line_count, - start_byte: region.marker_start as u32, - end_byte: region.content_end as u32, - checksum_start_byte: region.content_start as u32, - prologue_end_byte: Some(region.content_start as u32), - epilogue_start_byte: Some(region.content_end as u32), - checksum: cell_checksum, - error: false, - indent: 0, - indent_char: String::new(), - group: false, - }); - root_children.push(cell_path); - } - - // Populate root children now that every cell is known. - if let Some(root) = chunks.get_mut(0) { - root.children.clone_from(&root_children); - } - - // Sort chunks so the cell parent always comes before its sub-chunks, - // matching the invariant that other paths rely on (render, edit - // scheduling, line-to-chunk lookup). Keep the root at index 0. - // The insertion order above places sub-chunks before the cell parent, so - // we need to reorder: for each cell region, move the cell parent ahead of - // its sub-chunks. - // - // Simpler: rebuild the chunks list by iterating cells, emitting the cell - // parent followed by its sub-chunks in path order. - let mut reordered: Vec = Vec::with_capacity(chunks.len()); - reordered.push(chunks.remove(0)); // root - - let mut remaining: Vec = chunks; - for cell_path in &root_children { - // Extract the cell parent first. - if let Some(pos) = remaining.iter().position(|c| &c.path == cell_path) { - reordered.push(remaining.remove(pos)); - } - // Then any descendants of this cell. - let prefix = format!("{cell_path}."); - let mut i = 0; - while i < remaining.len() { - if remaining[i].path.starts_with(&prefix) { - reordered.push(remaining.remove(i)); - } else { - i += 1; - } - } - } - // Anything left over (shouldn't happen, but be defensive). - reordered.extend(remaining); - - Ok(ChunkTree { - language: "ipynb".to_string(), - checksum: root_checksum, - line_count: total_lines as u32, - parse_errors: 0, - parse_error_lines: Vec::new(), - fallback: false, - root_path: String::new(), - root_children, - chunks: reordered, - }) -} - -fn total_line_count(source: &str) -> usize { - if source.is_empty() { - 0 - } else { - source.bytes().filter(|b| *b == b'\n').count() + 1 - } -} - -// ──────────────────────────────────────────────────────────────────── -// Virtual → JSON round-trip -// ──────────────────────────────────────────────────────────────────── - -/// Update a [`NotebookContext`] from a (possibly edited) virtual source, -/// then serialize it back to JSON. Cells that no longer appear in the -/// virtual source are dropped; cells whose markers survive get their -/// `source` field replaced with the current body. -/// -/// Returns the serialized JSON text ready to be written to disk. -pub fn notebook_to_json( - virtual_source: &str, - base_ctx: &NotebookContext, -) -> Result { - let mut ctx = base_ctx.clone(); - let regions = scan_cells(virtual_source); - - // Rebuild the cells array in the order markers appear in the virtual - // source. Look up each marker's original cell by 1-based cell_num so - // edits that reorder cells via sibling insertion continue to track the - // right metadata. - let mut new_cells: Vec = Vec::with_capacity(regions.len()); - for region in ®ions { - let body_slice = &virtual_source[region.content_start..region.content_end]; - // Trim the single trailing newline that the virtual source format - // adds so edits that replace an entire cell body don't grow by one - // line every round-trip. - let body = trim_virtual_body(body_slice); - - let original = ctx.cells.get(region.cell_num.saturating_sub(1)).cloned(); - let cell = match original { - Some(mut cell) => { - cell.source = body.to_string(); - cell.cell_type.clone_from(®ion.cell_type); - cell - }, - None => NotebookCell { - cell_type: region.cell_type.clone(), - source: body.to_string(), - metadata: Value::Object(Map::new()), - outputs: match region.cell_type.as_str() { - "code" => Some(Value::Array(Vec::new())), - _ => None, - }, - execution_count: match region.cell_type.as_str() { - "code" => Some(Value::Null), - _ => None, - }, - other: Map::new(), - source_was_string: false, - }, - }; - new_cells.push(cell); - } - ctx.cells = new_cells; - - let json = serialize_notebook(&ctx)?; - Ok(json) -} - -/// Strip a single trailing newline from `body`, if present. The virtual -/// source always terminates each cell body with `\n` to make the markers -/// start on a fresh line; we remove that byte so the cell's stored source -/// matches the semantic content. -fn trim_virtual_body(body: &str) -> &str { - body.strip_suffix('\n').unwrap_or(body) -} - -fn serialize_notebook(ctx: &NotebookContext) -> Result { - // Build the cells array first. - let mut cells_arr: Vec = Vec::with_capacity(ctx.cells.len()); - for cell in &ctx.cells { - cells_arr.push(cell_to_value(cell)); - } - - // Rebuild the top-level object preserving the original key order with - // `cells` injected at the position it originally occupied. If the input - // had no `cells` key (we wouldn't be here), we append. - let mut top = Map::new(); - let mut cells_inserted = false; - for (k, v) in &ctx.top_fields { - top.insert(k.clone(), v.clone()); - if k == "metadata" && !cells_inserted { - // Jupyter's canonical order is cells, metadata, nbformat, - // nbformat_minor. We preserve whatever we found. - } - } - // If the original document had `cells` somewhere, we want to re-insert - // it at roughly the same slot. Jupyter always writes `cells` first, so - // build a fresh Map in canonical order: cells then the preserved - // top_fields. - let mut final_top = Map::new(); - final_top.insert("cells".to_string(), Value::Array(cells_arr)); - cells_inserted = true; - for (k, v) in top { - if k != "cells" { - final_top.insert(k, v); - } - } - let _ = cells_inserted; - - let indent_bytes = ctx.indent.as_bytes().to_vec(); - let formatter = serde_json::ser::PrettyFormatter::with_indent(&indent_bytes); - let mut buf: Vec = Vec::with_capacity(1024); - { - let mut ser = serde_json::Serializer::with_formatter(&mut buf, formatter); - Value::Object(final_top) - .serialize(&mut ser) - .map_err(|err| format!("Failed to serialize notebook JSON: {err}"))?; - } - let mut text = String::from_utf8(buf) - .map_err(|err| format!("Serialized notebook is not valid UTF-8: {err}"))?; - if ctx.trailing_newline && !text.ends_with('\n') { - text.push('\n'); - } - Ok(text) -} - -fn cell_to_value(cell: &NotebookCell) -> Value { - let mut obj = Map::new(); - obj.insert("cell_type".to_string(), Value::String(cell.cell_type.clone())); - // Preserve `id` and similar fields that idiomatically appear before - // `metadata` in nbformat 4+. - for (k, v) in &cell.other { - if !matches!(k.as_str(), "metadata" | "outputs" | "execution_count" | "source") { - obj.insert(k.clone(), v.clone()); - } - } - obj.insert("metadata".to_string(), cell.metadata.clone()); - if cell.cell_type == "code" { - obj.insert( - "execution_count".to_string(), - cell.execution_count.clone().unwrap_or(Value::Null), - ); - obj.insert( - "outputs".to_string(), - cell - .outputs - .clone() - .unwrap_or_else(|| Value::Array(Vec::new())), - ); - } else { - if let Some(outputs) = &cell.outputs { - obj.insert("outputs".to_string(), outputs.clone()); - } - if let Some(ec) = &cell.execution_count { - obj.insert("execution_count".to_string(), ec.clone()); - } - } - obj.insert("source".to_string(), source_to_value(&cell.source, cell.source_was_string)); - Value::Object(obj) -} - -/// Convert a flat source string to the Jupyter `source` field representation. -/// -/// If the cell originally used a string (or a new cell was inserted), we keep -/// it as a string. Otherwise we split into the canonical `Vec` with -/// each element preserving its trailing `\n`. -fn source_to_value(source: &str, was_string: bool) -> Value { - if was_string { - return Value::String(source.to_string()); - } - if source.is_empty() { - return Value::Array(Vec::new()); - } - let mut parts: Vec = Vec::new(); - for line in source.split_inclusive('\n') { - parts.push(Value::String(line.to_string())); - } - Value::Array(parts) -} - -// ──────────────────────────────────────────────────────────────────── -// Shared helpers -// ──────────────────────────────────────────────────────────────────── - -/// Atomically-shareable notebook context; used by [`ChunkStateInner`] to -/// carry the notebook metadata through edit cycles. -pub type SharedNotebookContext = Arc; - -#[cfg(test)] -mod tests { - use serde_json::json; - - use super::*; - use crate::chunk::state::ChunkStateInner; - - fn sample_notebook() -> String { - let value = json!({ - "cells": [ - { - "cell_type": "code", - "source": ["def foo():\n", " return 1\n"], - "metadata": {}, - "outputs": [], - "execution_count": null - }, - { - "cell_type": "markdown", - "source": ["# Hello\n", "World\n"], - "metadata": {} - }, - { - "cell_type": "code", - "source": ["class Bar:\n", " def baz(self):\n", " pass\n"], - "metadata": {}, - "outputs": [], - "execution_count": 3 - } - ], - "metadata": { - "kernelspec": { - "language": "python" - } - }, - "nbformat": 4, - "nbformat_minor": 5 - }); - serde_json::to_string_pretty(&value).expect("static json") - } - - #[test] - fn parses_notebook_into_cells() { - let nb = parse_notebook(&sample_notebook()).expect("valid notebook"); - assert_eq!(nb.context.cells.len(), 3); - assert_eq!(nb.context.cells[0].cell_type, "code"); - assert_eq!(nb.context.cells[1].cell_type, "markdown"); - assert_eq!(nb.context.cells[2].cell_type, "code"); - assert_eq!(nb.context.kernel_language, "python"); - } - - #[test] - fn virtual_source_contains_all_cells() { - let nb = parse_notebook(&sample_notebook()).expect("valid notebook"); - let vs = &nb.virtual_source; - assert!(vs.contains("def foo():"), "cell 1 body missing"); - assert!(vs.contains("# Hello"), "cell 2 body missing"); - assert!(vs.contains("class Bar:"), "cell 3 body missing"); - assert!(vs.contains("# %%% oh-my-pi cell_1 [code]"), "cell_1 marker missing"); - assert!(vs.contains("# %%% oh-my-pi cell_2 [markdown]"), "cell_2 marker missing"); - assert!(vs.contains("# %%% oh-my-pi cell_3 [code]"), "cell_3 marker missing"); - } - - #[test] - fn builds_cell_level_chunks() { - let nb = parse_notebook(&sample_notebook()).expect("valid notebook"); - let tree = - build_notebook_tree_from_virtual(&nb.virtual_source, "python").expect("tree should build"); - assert_eq!( - tree.root_children, - vec!["cell_1", "cell_2", "cell_3"], - "root children should be the three cells" - ); - let cell1 = tree - .chunks - .iter() - .find(|c| c.path == "cell_1") - .expect("cell_1 chunk"); - assert!(!cell1.leaf, "code cell with a function should not be a leaf"); - assert!( - cell1 - .children - .iter() - .any(|p| p.starts_with("cell_1.fn_foo")), - "cell_1 should contain fn_foo, got {:?}", - cell1.children - ); - } - - #[test] - fn sub_chunk_paths_are_prefixed_with_cell() { - let nb = parse_notebook(&sample_notebook()).expect("valid notebook"); - let tree = - build_notebook_tree_from_virtual(&nb.virtual_source, "python").expect("tree should build"); - let cell3 = tree - .chunks - .iter() - .find(|c| c.path == "cell_3") - .expect("cell_3 chunk"); - assert!( - cell3 - .children - .iter() - .any(|p| p.starts_with("cell_3.cls_Bar")), - "cell_3 should contain cls_Bar, got {:?}", - cell3.children - ); - let bar_method = tree - .chunks - .iter() - .find(|c| c.path == "cell_3.cls_Bar.fn_baz"); - assert!(bar_method.is_some(), "cell_3.cls_Bar.fn_baz should exist"); - } - - #[test] - fn chunk_state_parse_ipynb_carries_notebook_context() { - let json = sample_notebook(); - let state = - ChunkStateInner::parse(json, "ipynb".to_string()).expect("ChunkState should parse ipynb"); - assert_eq!(state.language(), "ipynb"); - // Source is the virtual source, not the JSON - assert!(state.source().contains("# %%% oh-my-pi cell_1")); - // The notebook context is preserved for JSON round-trip - let ctx = state - .notebook - .as_ref() - .expect("notebook context should be set"); - let json_out = notebook_to_json(state.source(), ctx).expect("should serialize back to JSON"); - let reparsed: serde_json::Value = - serde_json::from_str(&json_out).expect("output should be valid JSON"); - let cells = reparsed["cells"].as_array().expect("cells array"); - assert_eq!(cells.len(), 3, "should still have 3 cells"); - assert_eq!(cells[0]["cell_type"], "code"); - assert_eq!(cells[2]["execution_count"], 3); - } - - #[test] - fn empty_notebook_produces_empty_tree() { - let json = r#"{"cells": [], "metadata": {}, "nbformat": 4, "nbformat_minor": 5}"#; - let state = ChunkStateInner::parse(json.to_string(), "ipynb".to_string()) - .expect("empty notebook should parse"); - assert!(state.tree().root_children.is_empty()); - } -} diff --git a/crates/pi-natives/src/chunk/ast_js_ts.rs b/crates/pi-natives/src/chunk/ast_js_ts.rs deleted file mode 100644 index cd0cc012f..000000000 --- a/crates/pi-natives/src/chunk/ast_js_ts.rs +++ /dev/null @@ -1,564 +0,0 @@ -//! JavaScript / TypeScript / TSX chunk classifier. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, WrapperSignature, - WrapperTransform, classify_with_defaults, first_wrapper_content_child, - promote_wrapper_candidate, semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct JsTsClassifier; - -fn recurse_internal_module(node: Node<'_>) -> Option> { - recurse_into(node, ChunkContext::ClassBody, &["body"], &["statement_block"]) -} - -static JSTS_TABLES: ClassifierTables = ClassifierTables { - root: &[ - semantic_rule( - "import_statement", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "import_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "function_declaration", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "function", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "function_expression", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "arrow_function", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "generator_function", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "generator_function_declaration", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "class_declaration", - ChunkKind::Class, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "class", - ChunkKind::Class, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "class_expression", - ChunkKind::Class, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "interface_declaration", - ChunkKind::Interface, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "enum_declaration", - ChunkKind::Enum, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "type_alias_declaration", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::None, - ), - ], - class: &[ - semantic_rule( - "constructor", - ChunkKind::Constructor, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "class_static_block", - ChunkKind::StaticInit, - RuleStyle::Named, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "type_alias_declaration", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::None, - ), - ], - function: &[], - structural_overrides: super::classify::StructuralOverrides::EMPTY, -}; - -impl LangClassifier for JsTsClassifier { - fn tables(&self) -> &'static ClassifierTables { - &JSTS_TABLES - } - - fn is_trivia(&self, kind: &str) -> bool { - // Whitespace/text runs between JSX elements carry no structure and - // should be absorbed as leading trivia of the next element (matching - // the existing comment-absorption semantics). - kind == "jsx_text" - } - - fn should_skip_child(&self, kind: &str) -> bool { - // JSX opening and closing elements are part of the enclosing - // `jsx_element` chunk's framing, not children in their own right. - // Skip them entirely when enumerating children so they don't pollute - // the chunk tree with noisy 1‑line entries and, crucially, so they - // don't get absorbed backward into the next real child. - matches!(kind, "jsx_opening_element" | "jsx_closing_element") - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_root_custom(node, source), - ChunkContext::ClassBody => classify_class_custom(node, source), - ChunkContext::FunctionBody => Some(classify_function_js(node, source)), - } - } -} - -fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - // ── Exports / decorators ── - "export_statement" => Some(classify_export_statement(ChunkContext::Root, node, source)), - "decorated_definition" => promote_wrapper_candidate( - &JsTsClassifier, - ChunkContext::Root, - node, - source, - WrapperTransform { signature: WrapperSignature::Wrapper, ..WrapperTransform::default() }, - ) - .or_else(|| Some(positional_candidate(node, ChunkKind::Block, source))), - - // ── Variables ── - "lexical_declaration" | "variable_declaration" => Some(classify_var_decl_js(node, source)), - - // ── Containers with custom recursion ── - "internal_module" => { - Some(container_candidate(node, ChunkKind::Module, source, recurse_internal_module(node))) - }, - - // ── Control flow at top level ── - "if_statement" | "switch_statement" | "switch_expression" | "try_statement" - | "for_statement" | "for_in_statement" | "for_of_statement" | "while_statement" - | "do_statement" | "with_statement" => Some(classify_function_js(node, source)), - - // ── Statements ── - "expression_statement" => { - // Unwrap `expression_statement` wrapping an `internal_module` (namespace). - let inner = named_children(node) - .into_iter() - .find(|c| c.kind() == "internal_module"); - if let Some(ns) = inner { - Some(container_candidate(ns, ChunkKind::Module, source, recurse_internal_module(ns))) - } else { - Some(group_candidate(node, ChunkKind::Statements, source)) - } - }, - - _ => None, - } -} - -fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - // ── Exports / decorators (re-exported members) ── - "export_statement" => Some(classify_export_statement(ChunkContext::ClassBody, node, source)), - "decorated_definition" => promote_wrapper_candidate( - &JsTsClassifier, - ChunkContext::ClassBody, - node, - source, - WrapperTransform { signature: WrapperSignature::Wrapper, ..WrapperTransform::default() }, - ) - .or_else(|| Some(positional_candidate(node, ChunkKind::Block, source))), - - // ── Variables ── - "lexical_declaration" | "variable_declaration" => Some(classify_var_decl_js(node, source)), - - // ── Methods ── - "method_definition" | "method_signature" | "abstract_method_signature" => { - let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string()); - if name == "constructor" { - Some(make_kind_chunk( - node, - ChunkKind::Constructor, - None, - source, - recurse_body(node, ChunkContext::FunctionBody), - )) - } else { - Some(make_kind_chunk( - node, - ChunkKind::Function, - Some(name), - source, - recurse_body(node, ChunkContext::FunctionBody), - )) - } - }, - - // ── Fields ── - "public_field_definition" - | "field_definition" - | "property_definition" - | "property_signature" - | "property_declaration" - | "abstract_class_field" => match extract_identifier(node, source) { - Some(name) => Some(make_kind_chunk(node, ChunkKind::Field, Some(name), source, None)), - None => Some(group_candidate(node, ChunkKind::Fields, source)), - }, - - // ── Enum members ── - "enum_assignment" | "enum_member_declaration" => match extract_identifier(node, source) { - Some(name) => Some(make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None)), - None => Some(group_candidate(node, ChunkKind::Variants, source)), - }, - - _ => None, - } -} - -/// Classify nodes inside a function body for JS/TS. -fn classify_function_js<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody); - match node.kind() { - // ── Control flow ── - "if_statement" => { - make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source) - }, - "switch_statement" | "switch_expression" => { - make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source) - }, - "try_statement" => { - make_candidate(node, ChunkKind::Try, None, NameStyle::Named, None, fn_recurse(), source) - }, - - // ── Loops ── - "for_statement" => { - make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source) - }, - "for_in_statement" => { - make_candidate(node, ChunkKind::ForIn, None, NameStyle::Named, None, fn_recurse(), source) - }, - "for_of_statement" => { - make_candidate(node, ChunkKind::ForOf, None, NameStyle::Named, None, fn_recurse(), source) - }, - "while_statement" => { - make_candidate(node, ChunkKind::While, None, NameStyle::Named, None, fn_recurse(), source) - }, - "do_statement" => { - make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source) - }, - - // ── Blocks ── - "with_statement" => { - make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source) - }, - - // ── Variables ── - "lexical_declaration" | "variable_declaration" => { - if let Some(name) = extract_single_declarator_name(node, source) { - make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None) - } else { - group_from_sanitized(node, source) - } - }, - - // ── Return statements ── - // A bare `return …` or `return (…)` creates a - // huge monolithic leaf chunk in React components. Recurse into the JSX - // so each child element inside the returned tree stays individually - // addressable. Callback-with-trailing-block patterns such as - // `return items.map(item => { … })` are handled by the shared - // call-with-callback promotion in `classify_with_defaults` via the - // `return_statement` arm below. - "return_statement" => classify_return_statement_js(node, source), - - // ── JSX elements ── - // Inside function bodies, JSX elements become container chunks with - // their tag name so React component trees are navigable instead of - // opaque walls of markup. - "jsx_element" => classify_jsx_element(node, source), - "jsx_self_closing_element" => classify_jsx_self_closing_element(node, source), - "jsx_fragment" => make_candidate( - node, - ChunkKind::Tag, - Some("fragment".to_string()), - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::FunctionBody)), - source, - ), - - // ── Expression statements (enable call-with-callback promotion) ── - "expression_statement" => group_candidate(node, ChunkKind::Statements, source), - - // ── Fallback ── - _ => group_from_sanitized(node, source), - } -} - -/// Classify a `jsx_element` as a container chunk named after its tag. -/// -/// The chunk recurses into itself so that nested JSX children are emitted as -/// sub-chunks. Structural JSX nodes (opening/closing elements, text, -/// attributes) are filtered out as trivia by the classifier's `is_trivia` -/// override, so only meaningful children (child elements, expression -/// containers) become chunks. -fn classify_jsx_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let tag_name = extract_jsx_tag_name(node, source); - let mut candidate = make_candidate( - node, - ChunkKind::Tag, - tag_name, - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::FunctionBody)), - source, - ); - // Force recursion for jsx_elements that span more than a single - // source line. Without this, a `
` wrapping a single - // near-equal-sized child fails `recursion_narrows_scope` and the - // whole subtree collapses into one opaque chunk. One-line elements - // keep natural collapse behavior so short inline JSX stays a leaf. - if candidate - .range_end_line - .saturating_sub(candidate.range_start_line) - > 0 - { - candidate.force_recurse = true; - } - candidate -} - -/// Classify a `return_statement`. -/// -/// Two patterns matter: -/// -/// 1. `return …` / `return (…)` — unwrap any -/// parentheses and recurse directly into the JSX tree so each nested JSX -/// element is individually addressable. -/// 2. `return items.map(item => { … })` — the shared call-with-trailing- -/// callback promoter turns this into a named expression container that -/// recurses into the callback body. We invoke it explicitly here because the -/// shared promotion in `classify_with_defaults` only runs on groupable -/// leaves, and `Return` is not groupable. -fn classify_return_statement_js<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - if let Some(expr) = named_children(node).into_iter().next() { - let target = unwrap_parenthesized(expr); - if matches!(target.kind(), "jsx_element" | "jsx_fragment" | "jsx_self_closing_element") { - let mut candidate = make_candidate( - node, - ChunkKind::Return, - None::, - NameStyle::Named, - signature_for_node(node, source), - Some(RecurseSpec { node: target, context: ChunkContext::FunctionBody }), - source, - ); - // The JSX tree may span nearly the entire return statement, which - // would fail the `recursion_narrows_scope` check. Force recursion - // so the JSX children are always individually addressable. - candidate.force_recurse = true; - return candidate; - } - } - if let Some(mut promoted) = try_promote_call_with_callback(node, source) { - // The callback body is the sole child of the return value, so it - // spans nearly the entire return statement. Force recursion to - // guarantee the callback internals are addressable. - promoted.force_recurse = true; - return promoted; - } - group_from_sanitized(node, source) -} - -/// Classify a `jsx_self_closing_element` as a leaf chunk named after its tag. -fn classify_jsx_self_closing_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let tag_name = extract_jsx_tag_name(node, source); - make_kind_chunk(node, ChunkKind::Tag, tag_name, source, None) -} - -/// Unwrap nested `parenthesized_expression` wrappers to reach the meaningful -/// inner expression. -fn unwrap_parenthesized(mut node: Node<'_>) -> Node<'_> { - while node.kind() == "parenthesized_expression" { - let Some(inner) = named_children(node).into_iter().next() else { - break; - }; - node = inner; - } - node -} - -/// Extract the tag name from a `jsx_element` or `jsx_self_closing_element`. -/// -/// The tag may be an identifier (`div`, `Link`), a member expression -/// (`Foo.Bar`), or a nested identifier. We sanitize the full text so path -/// segments remain valid identifiers. -fn extract_jsx_tag_name(node: Node<'_>, source: &str) -> Option { - let name_holder = match node.kind() { - "jsx_element" => child_by_kind(node, &["jsx_opening_element"])?, - "jsx_self_closing_element" => node, - _ => return None, - }; - let name_node = named_children(name_holder).into_iter().find(|child| { - matches!( - child.kind(), - "identifier" | "member_expression" | "nested_identifier" | "jsx_namespace_name" - ) - })?; - sanitize_identifier(node_text(source, name_node.start_byte(), name_node.end_byte())) -} - -/// Classify `const`/`let`/`var` declarations, promoting arrow functions -/// and class expressions to fn_/class_ chunks. -fn classify_var_decl_js<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - // Inline promotion logic — look for single variable_declarator with fn/class - // value. - let declarators: Vec> = named_children(node) - .into_iter() - .filter(|c| c.kind() == "variable_declarator") - .collect(); - if declarators.len() == 1 { - let decl = declarators[0]; - if let Some(value) = decl.child_by_field_name("value") { - let name = extract_identifier(decl, source).unwrap_or_else(|| "anonymous".to_string()); - match value.kind() { - "arrow_function" | "function_expression" | "function" => { - let recurse = recurse_body(value, ChunkContext::FunctionBody); - return make_kind_chunk(node, ChunkKind::Function, Some(name), source, recurse); - }, - "class" | "class_expression" => { - let recurse = recurse_class(value); - return make_container_chunk(node, ChunkKind::Class, Some(name), source, recurse); - }, - _ => {}, - } - } - } - // Not promoted — fall back to var_NAME or group. - if let Some(name) = extract_single_declarator_name(node, source) { - return make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None); - } - group_candidate(node, ChunkKind::Declarations, source) -} - -fn group_from_sanitized<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let sanitized = sanitize_node_kind(node.kind()); - let kind = ChunkKind::from_sanitized_kind(sanitized); - let identifier = if kind == ChunkKind::Chunk { - Some(sanitized.to_string()) - } else { - None - }; - make_candidate(node, kind, identifier, NameStyle::Group, None, None, source) -} - -/// Unwrap `export` / `export default` to classify the inner declaration. -/// -/// Wrapper promotion handles declaration-like exports automatically. -/// `export default …` remaps the promoted child to `default_export`, while -/// re-exports and bare expression exports still fall through to `stmts`. -fn classify_export_statement<'t>( - context: ChunkContext, - node: Node<'t>, - source: &str, -) -> RawChunkCandidate<'t> { - let header = normalized_header(source, node.start_byte(), node.end_byte()); - let is_default = header.starts_with("export default"); - - if let Some(candidate) = - promote_wrapper_candidate(&JsTsClassifier, context, node, source, WrapperTransform { - kind: is_default.then_some(ChunkKind::DefaultExport), - name_style: is_default.then_some(NameStyle::Named), - clear_identifier: is_default, - ..WrapperTransform::default() - }) { - return candidate; - } - - let Some(child) = first_wrapper_content_child(&JsTsClassifier, node) else { - return if is_default { - make_kind_chunk(node, ChunkKind::DefaultExport, None, source, None) - } else { - group_candidate(node, ChunkKind::Statements, source) - }; - }; - - if is_default { - return make_kind_chunk(node, ChunkKind::DefaultExport, None, source, None); - } - - match child.kind() { - "lexical_declaration" | "variable_declaration" => { - classify_with_defaults(&JsTsClassifier, context, child, source) - }, - _ => group_candidate(child, ChunkKind::Statements, source), - } -} diff --git a/crates/pi-natives/src/chunk/ast_just.rs b/crates/pi-natives/src/chunk/ast_just.rs deleted file mode 100644 index 304c346c0..000000000 --- a/crates/pi-natives/src/chunk/ast_just.rs +++ /dev/null @@ -1,129 +0,0 @@ -//! Language-specific chunk classifier for Just. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct JustClassifier; - -const JUST_FUNCTION_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "recipe_line", - ChunkKind::Cmd, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "shebang", - ChunkKind::Shebang, - RuleStyle::Named, - NamingMode::None, - RecurseMode::None, - ), -]; - -const JUST_TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: JUST_FUNCTION_RULES, - structural_overrides: StructuralOverrides { - extra_trivia: &[], - preserved_trivia: &[], - extra_root_wrappers: &["source_file"], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, -}; - -fn first_named_child(node: Node<'_>) -> Option> { - named_children(node).into_iter().next() -} - -fn first_named_child_of_kind<'t>(node: Node<'t>, kind: &str) -> Option> { - named_children(node) - .into_iter() - .find(|child| child.kind() == kind) -} - -fn child_text<'a>(source: &'a str, node: Node<'_>) -> &'a str { - node_text(source, node.start_byte(), node.end_byte()) -} - -/// `set shell := ...` uses a dedicated `shell` token instead of a named -/// identifier, so parse the assignment head text instead of relying on fields. -fn extract_setting_name(node: Node<'_>, source: &str) -> Option { - let header = child_text(source, node).lines().next()?.trim(); - let rest = header.strip_prefix("set ")?; - let name = rest.split_once(":=")?.0.trim(); - sanitize_identifier(name) -} - -fn extract_alias_name(node: Node<'_>, source: &str) -> Option { - first_named_child(node).and_then(|child| sanitize_identifier(child_text(source, child))) -} - -fn extract_recipe_name(node: Node<'_>, source: &str) -> Option { - let header = first_named_child_of_kind(node, "recipe_header")?; - first_named_child(header).and_then(|child| sanitize_identifier(child_text(source, child))) -} - -fn classify_just_root_node<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "setting" => { - let name = extract_setting_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - make_kind_chunk(node, ChunkKind::Setting, Some(name), source, None) - }, - "alias" => { - let name = extract_alias_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - make_kind_chunk(node, ChunkKind::Alias, Some(name), source, None) - }, - "recipe" => { - let name = extract_recipe_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - make_container_chunk( - node, - ChunkKind::Recipe, - Some(name), - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["recipe_body"]), - ) - }, - _ => return None, - }) -} - -fn classify_just_body_node<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - // Just recipe bodies are line-oriented; tree-sitter exposes shell lines as - // `recipe_line` leaves rather than a nested shell AST. - "recipe_line" => group_candidate(node, ChunkKind::Cmd, source), - "shebang" => make_kind_chunk(node, ChunkKind::Shebang, None, source, None), - _ => return None, - }) -} - -impl LangClassifier for JustClassifier { - fn tables(&self) -> &'static ClassifierTables { - &JUST_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_just_root_node(node, source), - ChunkContext::FunctionBody => classify_just_body_node(node, source), - ChunkContext::ClassBody => None, - } - } -} diff --git a/crates/pi-natives/src/chunk/ast_markup.rs b/crates/pi-natives/src/chunk/ast_markup.rs deleted file mode 100644 index ce24aeab5..000000000 --- a/crates/pi-natives/src/chunk/ast_markup.rs +++ /dev/null @@ -1,341 +0,0 @@ -//! Language-specific chunk classifiers for Markdown and Handlebars. - -use tree_sitter::Node; - -use super::{ - chunk_checksum, - classify::{ClassifierTables, LangClassifier}, - common::*, - kind::ChunkKind, - types::ChunkNode, -}; -use crate::language::SupportLang; - -pub struct MarkupClassifier; - -impl MarkupClassifier { - fn classify_section<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let name = extract_markdown_heading(node, source).unwrap_or_else(|| "anonymous".to_string()); - force_container(make_container_chunk( - node, - ChunkKind::Section, - Some(name), - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - )) - } - - fn classify_block_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let name = - extract_glimmer_block_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - force_container(make_container_chunk( - node, - ChunkKind::Block, - Some(name), - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - )) - } - - fn classify_mustache_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let name = - extract_glimmer_mustache_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - make_kind_chunk(node, ChunkKind::Mustache, Some(name), source, None) - } - - /// Classify HTML-like element nodes that appear inside handlebars blocks. - fn classify_element<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "element" | "script_element" | "style_element" | "element_node" => { - let name = - extract_element_tag_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - Some(force_container(make_container_chunk( - node, - ChunkKind::Tag, - Some(name), - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - ))) - }, - "text_node" => Some(group_candidate(node, ChunkKind::Text, source)), - _ => None, - } - } -} - -impl LangClassifier for MarkupClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: super::classify::StructuralOverrides::EMPTY, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - if !matches!(context, ChunkContext::Root | ChunkContext::ClassBody) { - return None; - } - match node.kind() { - "section" => Some(Self::classify_section(node, source)), - "fenced_code_block" => Some(classify_fenced_code_block(node, source)), - "html_block" => Some(classify_html_block(node, source)), - "block_statement" => Some(Self::classify_block_statement(node, source)), - "mustache_statement" => Some(Self::classify_mustache_statement(node, source)), - "element" | "script_element" | "style_element" | "element_node" | "text_node" => { - Self::classify_element(node, source) - }, - _ => None, - } - } - - fn post_process( - &self, - chunks: &mut Vec, - _root_children: &mut Vec, - source: &str, - ) { - add_markdown_table_row_chunks(chunks, source); - } -} - -const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> { - candidate.force_recurse = true; - candidate -} - -fn classify_fenced_code_block<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let embedded_language = fenced_code_language(node, source); - let identifier = embedded_language - .map(embedded_selector_token) - .map(str::to_string); - let candidate = - with_region_node(make_kind_chunk(node, ChunkKind::Code, identifier, source, None), None); - match (child_by_kind(node, &["code_fence_content"]), embedded_language) { - (Some(content_node), Some(language)) => { - with_injected_subtree(candidate, language, content_node) - }, - _ => candidate, - } -} - -fn classify_html_block<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let candidate = - with_region_node(make_kind_chunk(node, ChunkKind::Html, None, source, None), None); - with_injected_subtree(candidate, SupportLang::Html, node) -} - -fn add_markdown_table_row_chunks(chunks: &mut Vec, source: &str) { - let original_len = chunks.len(); - let mut additions = Vec::<(usize, Vec)>::new(); - for (index, chunk) in chunks.iter().enumerate().take(original_len) { - if !chunk.children.is_empty() || matches!(chunk.kind, ChunkKind::Code | ChunkKind::Html) { - continue; - } - let rows = markdown_table_rows_for_chunk(source, chunk); - if rows.len() < 2 - || !rows - .iter() - .any(|row| markdown_table_separator_row(row.text)) - { - continue; - } - let nodes = rows - .into_iter() - .enumerate() - .map(|(row_index, row)| { - let identifier = (row_index + 1).to_string(); - let path = format!("{}.row_{}", chunk.path, identifier); - let (indent, indent_char) = detect_indent(source, row.start_byte); - let row_source = source.get(row.start_byte..row.end_byte).unwrap_or_default(); - ChunkNode { - path, - identifier: Some(identifier), - kind: ChunkKind::Row, - leaf: true, - virtual_content: None, - parent_path: Some(chunk.path.clone()), - children: Vec::new(), - signature: Some(row.text.trim().to_owned()), - start_line: row.line, - end_line: row.line, - line_count: 1, - start_byte: row.start_byte as u32, - end_byte: row.end_byte as u32, - checksum_start_byte: row.start_byte as u32, - prologue_end_byte: None, - epilogue_start_byte: None, - checksum: chunk_checksum(row_source.as_bytes()), - error: false, - indent, - indent_char, - group: false, - } - }) - .collect(); - additions.push((index, nodes)); - } - - for (index, nodes) in additions { - let child_paths = nodes.iter().map(|node| node.path.clone()).collect(); - chunks[index].leaf = false; - chunks[index].children = child_paths; - chunks.extend(nodes); - } -} - -struct MarkdownTableRow<'a> { - line: u32, - start_byte: usize, - end_byte: usize, - text: &'a str, -} - -fn markdown_table_rows_for_chunk<'a>( - source: &'a str, - chunk: &ChunkNode, -) -> Vec> { - let line_offsets = source_line_offsets(source); - let mut rows = Vec::new(); - let mut saw_non_empty = false; - for line in chunk.start_line..=chunk.end_line { - let Some((start_byte, end_byte)) = line_bounds(source, &line_offsets, line) else { - continue; - }; - let text = source - .get(start_byte..end_byte) - .unwrap_or_default() - .trim_end_matches('\n'); - if text.trim().is_empty() { - continue; - } - saw_non_empty = true; - if !markdown_table_row(text) { - return Vec::new(); - } - rows.push(MarkdownTableRow { line, start_byte, end_byte, text }); - } - if saw_non_empty { rows } else { Vec::new() } -} - -fn source_line_offsets(source: &str) -> Vec { - let mut offsets = vec![0usize]; - for (index, ch) in source.char_indices() { - if ch == '\n' { - offsets.push(index + 1); - } - } - offsets -} - -fn line_bounds(source: &str, offsets: &[usize], line: u32) -> Option<(usize, usize)> { - if line == 0 { - return None; - } - let start = *offsets.get((line - 1) as usize)?; - let end = offsets.get(line as usize).copied().unwrap_or(source.len()); - Some((start, end)) -} - -fn markdown_table_row(line: &str) -> bool { - let trimmed = line.trim(); - trimmed.starts_with('|') && trimmed.ends_with('|') && trimmed.matches('|').count() >= 2 -} - -fn markdown_table_separator_row(line: &str) -> bool { - let trimmed = line.trim(); - trimmed.contains('-') - && trimmed - .chars() - .all(|ch| matches!(ch, '|' | '-' | ':' | ' ' | '\t')) -} - -/// Extract heading text from a Markdown `section` node's `atx_heading` or -/// `setext_heading` child. -fn extract_markdown_heading(node: Node<'_>, source: &str) -> Option { - named_children(node) - .into_iter() - .find(|child| child.kind() == "atx_heading" || child.kind() == "setext_heading") - .and_then(|heading| { - sanitize_identifier(node_text(source, heading.start_byte(), heading.end_byte())) - }) -} - -/// Extract name from a Handlebars `block_statement` via its -/// `block_statement_start` child. -fn extract_glimmer_block_name(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["block_statement_start"]).and_then(|start| { - start - .child_by_field_name("path") - .or_else(|| child_by_kind(start, &["identifier"])) - .and_then(|name| { - sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())) - }) - }) -} - -fn fenced_code_language(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["info_string"]) - .and_then(|info| child_by_kind(info, &["language"])) - .and_then(|lang| { - SupportLang::from_alias(node_text(source, lang.start_byte(), lang.end_byte())) - }) -} - -/// Extract name from a Handlebars `mustache_statement`: -/// tries `helper_invocation`'s helper field first, then direct -/// `identifier`/`path_expression`. -fn extract_glimmer_mustache_name(node: Node<'_>, source: &str) -> Option { - let children = named_children(node); - for child in children { - if child.kind() == "helper_invocation" - && let Some(helper) = child - .child_by_field_name("helper") - .or_else(|| child_by_kind(child, &["identifier", "path_expression"])) - { - return sanitize_identifier(node_text(source, helper.start_byte(), helper.end_byte())); - } - if matches!(child.kind(), "identifier" | "path_expression") { - return sanitize_identifier(node_text(source, child.start_byte(), child.end_byte())); - } - } - None -} - -/// Extract tag name from an HTML-like element node. -/// -/// Handles both standard HTML (`element` → `start_tag`/`self_closing_tag` → -/// `tag_name`) and Handlebars element nodes (`element_node` → -/// `element_node_start`/`element_node_void` → `tag_name`). -fn extract_element_tag_name(node: Node<'_>, source: &str) -> Option { - // Handlebars element_node uses element_node_start / element_node_void - if node.kind() == "element_node" { - return named_children(node).into_iter().find_map(|child| { - if child.kind() == "element_node_start" || child.kind() == "element_node_void" { - child_by_kind(child, &["tag_name"]).and_then(|tag| { - sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte())) - }) - } else { - None - } - }); - } - - // Standard HTML: element → start_tag / self_closing_tag → tag_name - named_children(node).into_iter().find_map(|child| { - if child.kind() == "start_tag" || child.kind() == "self_closing_tag" { - child_by_kind(child, &["tag_name"]).and_then(|tag| { - sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte())) - }) - } else { - None - } - }) -} diff --git a/crates/pi-natives/src/chunk/ast_misc.rs b/crates/pi-natives/src/chunk/ast_misc.rs deleted file mode 100644 index ee7177fb3..000000000 --- a/crates/pi-natives/src/chunk/ast_misc.rs +++ /dev/null @@ -1,1163 +0,0 @@ -//! Chunk classifiers for languages well-served by defaults: -//! Kotlin, Swift, PHP, Solidity, Julia, Odin, Verilog, Zig, Regex, Diff. -//! -//! This is the catch-all classifier: it handles every node kind that any of the -//! miscellaneous languages produce so that nothing silently falls through. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - WrapperSignature, WrapperTransform, promote_wrapper_candidate, semantic_rule, - }, - common::*, - defaults::classify_var_decl, - kind::ChunkKind, -}; - -pub struct MiscClassifier; - -fn sanitized_group_candidate<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let sanitized = sanitize_node_kind(node.kind()); - let kind = ChunkKind::from_sanitized_kind(sanitized); - // For unknown kinds that fall back to `Chunk`, preserve the original - // tree-sitter kind as the identifier so the path stays informative. - let identifier = if kind == ChunkKind::Chunk { - Some(sanitized.to_string()) - } else { - None - }; - make_candidate(node, kind, identifier, NameStyle::Group, None, None, source) -} - -// ── Root-level table rules ────────────────────────────────────────────────── -// -// Control flow kinds that previously delegated to classify_function are -// duplicated here so they resolve at root without a cross-context call. - -const MISC_ROOT_RULES: &[super::classify::SemanticRule] = &[ - // Imports / package headers - semantic_rule( - "import_statement", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "import_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "using_directive", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "using_statement", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "namespace_use_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "namespace_statement", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "import_list", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "import_header", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "package_header", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "package_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // Variables / assignments (simple group) - semantic_rule( - "assignment", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "property_declaration", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "state_variable_declaration", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // Statements - semantic_rule( - "expression_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "global_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "command", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "pipeline", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "function_call", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // Methods - semantic_rule( - "method_declaration", - ChunkKind::Method, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // Constructors - semantic_rule( - "constructor_definition", - ChunkKind::Constructor, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "constructor_declaration", - ChunkKind::Constructor, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "secondary_constructor", - ChunkKind::Constructor, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "init_declaration", - ChunkKind::Constructor, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "fallback_receive_definition", - ChunkKind::Constructor, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // Containers - semantic_rule( - "class_declaration", - ChunkKind::Class, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "class_definition", - ChunkKind::Class, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "interface_declaration", - ChunkKind::Iface, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "protocol_declaration", - ChunkKind::Iface, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "struct_declaration", - ChunkKind::Struct, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "object_declaration", - ChunkKind::Struct, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "enum_declaration", - ChunkKind::Enum, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "enum_definition", - ChunkKind::Enum, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "trait_definition", - ChunkKind::Trait, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "class", - ChunkKind::Trait, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "contract_declaration", - ChunkKind::Contract, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "library_declaration", - ChunkKind::Contract, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "trait_declaration", - ChunkKind::Contract, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - // Types / aliases - semantic_rule( - "type_alias_declaration", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "const_type_declaration", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "opaque_declaration", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - // Macros - semantic_rule( - "macro_definition", - ChunkKind::Macro, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "modifier_definition", - ChunkKind::Macro, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // Systems (Verilog etc.) - semantic_rule( - "covergroup_declaration", - ChunkKind::Group, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "checker_declaration", - ChunkKind::Group, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "module_declaration", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "union_declaration", - ChunkKind::Union, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - // Control flow at top level (same rules as function table) - semantic_rule( - "if_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "unless", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "guard_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "switch_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "switch_expression", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "case_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "expression_switch_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "type_switch_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "select_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "try_statement", - ChunkKind::Try, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "try_block", - ChunkKind::Try, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "foreach_statement", - ChunkKind::For, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "do_statement", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "with_statement", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), -]; - -// ── Class-level table rules ───────────────────────────────────────────────── - -const MISC_CLASS_RULES: &[super::classify::SemanticRule] = &[ - // Constructors - semantic_rule( - "constructor", - ChunkKind::Constructor, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "constructor_declaration", - ChunkKind::Constructor, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "secondary_constructor", - ChunkKind::Constructor, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "init_declaration", - ChunkKind::Constructor, - RuleStyle::Named, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // Method specs - semantic_rule( - "method_spec", - ChunkKind::Method, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::None, - ), - // Field / method lists - semantic_rule( - "field_declaration_list", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "method_spec_list", - ChunkKind::Methods, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // Static initializer - semantic_rule( - "class_static_block", - ChunkKind::StaticInit, - RuleStyle::Named, - NamingMode::None, - RecurseMode::None, - ), - // Types inside classes - semantic_rule( - "type_item", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::None, - ), - semantic_rule( - "type_alias_declaration", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::None, - ), - semantic_rule( - "type_alias", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::None, - ), - // Const / macro inside classes - semantic_rule( - "const_item", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "macro_invocation", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // Grouped field-like entries - semantic_rule( - "assignment", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "expression_statement", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "attribute", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule("pair", ChunkKind::Fields, RuleStyle::Group, NamingMode::None, RecurseMode::None), - semantic_rule( - "block_mapping_pair", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "flow_pair", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -// ── Function-level table rules ────────────────────────────────────────────── - -const MISC_FUNCTION_RULES: &[super::classify::SemanticRule] = &[ - // Control flow: conditionals - semantic_rule( - "if_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "unless", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "guard_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // Control flow: switches - semantic_rule( - "switch_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "switch_expression", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "case_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "case_match", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "expression_switch_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "type_switch_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "select_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "receive_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "yul_switch_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // Control flow: try/catch - semantic_rule( - "try_statement", - ChunkKind::Try, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "try_block", - ChunkKind::Try, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "catch_clause", - ChunkKind::Try, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "finally_clause", - ChunkKind::Try, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "assembly_statement", - ChunkKind::Try, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // Blocks - semantic_rule( - "do_statement", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "with_statement", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "do_block", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "subshell", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "async_block", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "unsafe_block", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "const_block", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "block_expression", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // Loops: foreach - semantic_rule( - "foreach_statement", - ChunkKind::For, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // Statements - semantic_rule( - "defer_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "go_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "send_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // Positional candidates - semantic_rule( - "elif_clause", - ChunkKind::Elif, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "except_clause", - ChunkKind::Except, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "when_statement", - ChunkKind::When, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "match_expression", - ChunkKind::Match, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "match_block", - ChunkKind::Match, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - // Loops / misc expressions - semantic_rule( - "loop_expression", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "while_expression", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "for_expression", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "errdefer_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "comptime_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "nosuspend_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "suspend_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "yul_if_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "yul_for_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), -]; - -const MISC_TABLES: ClassifierTables = ClassifierTables { - root: MISC_ROOT_RULES, - class: MISC_CLASS_RULES, - function: MISC_FUNCTION_RULES, - structural_overrides: StructuralOverrides::EMPTY, -}; - -impl LangClassifier for MiscClassifier { - fn tables(&self) -> &'static ClassifierTables { - &MISC_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_root_override(node, source), - ChunkContext::ClassBody => classify_class_override(node, source), - ChunkContext::FunctionBody => classify_function_override(node, source), - } - } -} - -/// Root-level overrides for arms that need custom logic (`classify_var_decl`, -/// conditional identifier extraction, or custom recurse fallbacks). -fn classify_root_override<'t>(node: Node<'t>, source: &str) -> Option> { - let fn_recurse = || { - recurse_body(node, ChunkContext::FunctionBody) - .or_else(|| recurse_into(node, ChunkContext::FunctionBody, &["body"], &["block"])) - }; - let module_recurse = || { - recurse_class(node).or_else(|| { - recurse_into(node, ChunkContext::ClassBody, &["body"], &[ - "compound_statement", - "statement_block", - "declaration_list", - "block", - ]) - }) - }; - Some(match node.kind() { - // classify_var_decl delegation - "lexical_declaration" | "variable_declaration" => classify_var_decl(node, source), - - // Conditional identifier extraction - "const_declaration" | "var_declaration" => match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None), - None => group_candidate(node, ChunkKind::Declarations, source), - }, - - // Custom recurse fallback (recurse_into for body/block) - "function_declaration" - | "function_definition" - | "procedure_declaration" - | "overloaded_procedure_declaration" - | "test_declaration" => named_candidate(node, ChunkKind::Function, source, fn_recurse()), - - // Custom recurse fallback (module_recurse with compound_statement etc.) - "namespace_declaration" - | "namespace_definition" - | "module_definition" - | "extension_definition" => { - container_candidate(node, ChunkKind::Module, source, module_recurse()) - }, - - // Conditional loop kinds: looks_like_python_statement changes the ChunkKind - "for_statement" | "for_in_statement" | "for_of_statement" => { - let fn_recurse = recurse_body(node, ChunkContext::FunctionBody); - let kind = if looks_like_python_statement(node, source) { - ChunkKind::Loop - } else { - match node.kind() { - "for_statement" => ChunkKind::For, - "for_in_statement" => ChunkKind::ForIn, - "for_of_statement" => ChunkKind::ForOf, - _ => unreachable!(), - } - }; - make_candidate(node, kind, None, NameStyle::Named, None, fn_recurse, source) - }, - - // Conditional: looks_like_python_statement - "while_statement" => { - let kind = if looks_like_python_statement(node, source) { - ChunkKind::Loop - } else { - ChunkKind::While - }; - make_candidate( - node, - kind, - None, - NameStyle::Named, - None, - recurse_body(node, ChunkContext::FunctionBody), - source, - ) - }, - - _ => return None, - }) -} - -/// Class-level overrides for arms with conditional logic (constructor name -/// checks, identifier extraction fallbacks, decorated definitions). -fn classify_class_override<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - // Conditional: name == "constructor" changes the ChunkKind - "method_definition" - | "method_signature" - | "abstract_method_signature" - | "method_declaration" - | "function_declaration" - | "function_definition" - | "function_item" - | "procedure_declaration" - | "protocol_function_declaration" - | "method" - | "singleton_method" => { - let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string()); - if name == "constructor" { - make_kind_chunk( - node, - ChunkKind::Constructor, - None, - source, - recurse_body(node, ChunkContext::FunctionBody), - ) - } else { - make_kind_chunk( - node, - ChunkKind::Function, - Some(name), - source, - recurse_body(node, ChunkContext::FunctionBody), - ) - } - }, - - // Conditional identifier extraction with group fallback - "public_field_definition" - | "field_definition" - | "property_definition" - | "property_signature" - | "property_declaration" - | "protocol_property_declaration" - | "abstract_class_field" - | "const_declaration" - | "constant_declaration" - | "event_field_declaration" => match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None), - None => group_candidate(node, ChunkKind::Fields, source), - }, - - // Conditional identifier extraction - "enum_assignment" - | "enum_member_declaration" - | "enum_constant" - | "enum_entry" - | "enum_variant" => match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None), - None => group_candidate(node, ChunkKind::Variants, source), - }, - - // Conditional identifier extraction - "field_declaration" | "embedded_field" | "container_field" | "binding" => { - match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None), - None => group_candidate(node, ChunkKind::Fields, source), - } - }, - - "decorated_definition" => promote_wrapper_candidate( - &MiscClassifier, - ChunkContext::ClassBody, - node, - source, - WrapperTransform { signature: WrapperSignature::Wrapper, ..WrapperTransform::default() }, - )?, - - _ => return None, - }) -} - -/// Function-level overrides for arms with conditional logic -/// (Python-like detection, span-based variable handling). -fn classify_function_override<'t>(node: Node<'t>, source: &str) -> Option> { - let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody); - Some(match node.kind() { - // Conditional: looks_like_python_statement changes the ChunkKind - "for_statement" | "for_in_statement" | "for_of_statement" => { - let kind = if looks_like_python_statement(node, source) { - ChunkKind::Loop - } else { - match node.kind() { - "for_statement" => ChunkKind::For, - "for_in_statement" => ChunkKind::ForIn, - "for_of_statement" => ChunkKind::ForOf, - _ => unreachable!(), - } - }; - make_candidate(node, kind, None, NameStyle::Named, None, fn_recurse(), source) - }, - - // Conditional: looks_like_python_statement - "while_statement" => { - let kind = if looks_like_python_statement(node, source) { - ChunkKind::Loop - } else { - ChunkKind::While - }; - make_candidate(node, kind, None, NameStyle::Named, None, fn_recurse(), source) - }, - - // Conditional span/name logic - "lexical_declaration" - | "variable_declaration" - | "const_declaration" - | "var_declaration" - | "short_var_declaration" - | "let_declaration" => { - let span = line_span(node.start_position().row + 1, node.end_position().row + 1); - if span > 1 { - if let Some(name) = extract_single_declarator_name(node, source) { - make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None) - } else { - sanitized_group_candidate(node, source) - } - } else { - sanitized_group_candidate(node, source) - } - }, - - _ => return None, - }) -} diff --git a/crates/pi-natives/src/chunk/ast_nix_hcl.rs b/crates/pi-natives/src/chunk/ast_nix_hcl.rs deleted file mode 100644 index e374e9d56..000000000 --- a/crates/pi-natives/src/chunk/ast_nix_hcl.rs +++ /dev/null @@ -1,228 +0,0 @@ -//! Language-specific chunk classifiers for Nix and HCL (Terraform). - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier, StructuralOverrides}, - common::*, - kind::ChunkKind, -}; - -pub struct NixHclClassifier; - -const NIX_HCL_TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: StructuralOverrides { - extra_trivia: &[], - preserved_trivia: &[], - extra_root_wrappers: &["body"], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, -}; - -/// Extract a structured name from an HCL `block` node. -/// -/// Shape: `block_type label1 label2 … { body }` where labels are `string_lit`. -/// Returns e.g. `resource_aws_instance_web` for `resource "aws_instance" "web" -/// { … }`. -fn extract_hcl_block_name(node: Node<'_>, source: &str) -> Option { - let mut children = named_children(node).into_iter(); - let block_type = children.next()?; - let mut parts = - vec![node_text(source, block_type.start_byte(), block_type.end_byte()).to_string()]; - for child in children { - if child.kind() == "string_lit" { - let text = unquote_text(node_text(source, child.start_byte(), child.end_byte())); - if !text.is_empty() { - parts.push(text); - } - continue; - } - if child.kind() == "body" || child.kind() == "block_end" || child.kind() == "block_start" { - continue; - } - let text = node_text(source, child.start_byte(), child.end_byte()); - if !text.is_empty() { - parts.push(text.to_string()); - } - } - sanitize_identifier(parts.join("_").as_str()) -} - -/// Extract the attrpath name from a Nix `binding` node. -fn extract_nix_binding_name(node: Node<'_>, source: &str) -> Option { - node.child_by_field_name("attrpath").and_then(|attrpath| { - sanitize_identifier(node_text(source, attrpath.start_byte(), attrpath.end_byte())) - }) -} - -fn recurse_nix_attrset(node: Node<'_>) -> Option> { - recurse_into(node, ChunkContext::ClassBody, &[], &["binding_set"]) -} - -fn recurse_nix_binding_value(node: Node<'_>) -> Option> { - let expression = node.child_by_field_name("expression")?; - if matches!( - expression.kind(), - "attrset_expression" | "let_attrset_expression" | "rec_attrset_expression" - ) { - return recurse_nix_attrset(expression); - } - recurse_value_container(node) -} - -fn classify_nix_binding<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let name = extract_nix_binding_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - let expression = node.child_by_field_name("expression"); - if let Some(expression) = expression - && matches!( - expression.kind(), - "attrset_expression" | "let_attrset_expression" | "rec_attrset_expression" - ) { - return make_container_chunk( - node, - ChunkKind::Attr, - Some(name), - source, - recurse_nix_attrset(expression), - ); - } - make_kind_chunk(node, ChunkKind::Attr, Some(name), source, recurse_nix_binding_value(node)) -} - -impl LangClassifier for NixHclClassifier { - fn tables(&self) -> &'static ClassifierTables { - &NIX_HCL_TABLES - } - - fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option> { - match node.kind() { - // Nix top-level attrsets should recurse into their binding_set so the file exposes - // structural attr chunks instead of a single opaque attrset_expr leaf. - "attrset_expression" | "let_attrset_expression" | "rec_attrset_expression" => { - Some(make_candidate( - node, - ChunkKind::Attrs, - None, - NameStyle::Named, - signature_for_node(node, source), - recurse_nix_attrset(node), - source, - )) - }, - // Older tree-sitter-nix revisions used `attribute`; current grammars expose `binding`. - "attribute" | "binding" => Some(classify_nix_binding(node, source)), - // HCL top-level block, or diff hunk fallback - "block" => { - if let Some(name) = extract_hcl_block_name(node, source) { - Some(make_container_chunk( - node, - ChunkKind::Block, - Some(name), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["body"]), - )) - } else { - Some(group_candidate(node, ChunkKind::Hunks, source)) - } - }, - // Nix expressions - "function_expression" | "let_expression" => Some(named_candidate( - node, - ChunkKind::Expression, - source, - recurse_value_container(node), - )), - // Nix inherit - "inherit" => Some(group_candidate(node, ChunkKind::Imports, source)), - // Variable/assignment declarations - "variable_declaration" | "assignment" => { - Some(group_candidate(node, ChunkKind::Declarations, source)) - }, - // HCL top-level block types - "provider" | "resource" | "data" | "locals" | "variable" | "output" | "module" => { - let kind = match node.kind() { - "locals" => ChunkKind::BlockLocals, - "variable" => ChunkKind::Variable, - "module" => ChunkKind::Module, - _ => ChunkKind::Block, - }; - Some(make_candidate( - node, - kind, - prefixed_name(sanitize_node_kind(node.kind()), node, source), - NameStyle::Named, - signature_for_node(node, source), - recurse_into(node, ChunkContext::ClassBody, &[], &["body"]), - source, - )) - }, - _ => None, - } - } - - fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option> { - match node.kind() { - // Nested HCL block — only promote if it has an identifiable block name - "block" => extract_hcl_block_name(node, source).map(|name| { - make_container_chunk( - node, - ChunkKind::Block, - Some(name), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["body"]), - ) - }), - // Nested Nix attrset values recurse into their binding_set just like top-level ones. - "attrset_expression" | "let_attrset_expression" | "rec_attrset_expression" => { - Some(make_candidate( - node, - ChunkKind::Attrs, - None, - NameStyle::Named, - signature_for_node(node, source), - recurse_nix_attrset(node), - source, - )) - }, - // Nix binding_set is a transparent wrapper around individual bindings. - // Without this, the binding_set becomes an opaque leaf chunk hiding - // all bindings inside a single massive block. - "binding_set" => Some(make_candidate( - node, - ChunkKind::Attrs, - None, - NameStyle::Named, - None, - Some(RecurseSpec { node, context: ChunkContext::ClassBody }), - source, - )), - // Nix let_expression inside a class-body context — recurse into its - // binding_set so individual bindings are addressable. - "let_expression" => Some(make_candidate( - node, - ChunkKind::Expression, - None, - NameStyle::Named, - signature_for_node(node, source), - recurse_value_container(node), - source, - )), - // Nested Nix binding - "binding" => Some(classify_nix_binding(node, source)), - _ => None, - } - } - - fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option> { - match node.kind() { - // Nix control flow - "if_expression" => Some(positional_candidate(node, ChunkKind::If, source)), - "let_expression" => Some(positional_candidate(node, ChunkKind::Block, source)), - _ => None, - } - } -} diff --git a/crates/pi-natives/src/chunk/ast_ocaml.rs b/crates/pi-natives/src/chunk/ast_ocaml.rs deleted file mode 100644 index 004620bb8..000000000 --- a/crates/pi-natives/src/chunk/ast_ocaml.rs +++ /dev/null @@ -1,236 +0,0 @@ -//! OCaml-specific chunk classifier. - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier, StructuralOverrides}, - common::*, - kind::ChunkKind, -}; - -pub struct OcamlClassifier; - -impl LangClassifier for OcamlClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: StructuralOverrides::EMPTY, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_ocaml_item(node, source), - ChunkContext::ClassBody => classify_class(node, source), - ChunkContext::FunctionBody => classify_function(node, source), - } - } -} - -fn classify_class<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "method_definition" => Some(make_kind_chunk( - node, - ChunkKind::Function, - ocaml_named_text(node, source, &["method_name"]), - source, - ocaml_method_recurse(node), - )), - "method_specification" => Some(make_kind_chunk( - node, - ChunkKind::Function, - ocaml_named_text(node, source, &["method_name"]), - source, - None, - )), - "instance_variable_definition" => { - Some(match ocaml_named_text(node, source, &["instance_variable_name"]) { - Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None), - None => group_candidate(node, ChunkKind::Fields, source), - }) - }, - _ => classify_ocaml_item(node, source), - } -} - -fn classify_function<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "function_expression" | "match_expression" => Some(make_candidate( - node, - ChunkKind::Match, - None, - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::FunctionBody)), - source, - )), - "match_case" => Some(make_candidate( - node, - ChunkKind::Case, - None, - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::FunctionBody)), - source, - )), - "let_expression" => Some(make_candidate( - node, - ChunkKind::Let, - None, - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::FunctionBody)), - source, - )), - _ => classify_ocaml_item(node, source), - } -} - -fn classify_ocaml_item<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "open_module" => group_candidate(node, ChunkKind::Imports, source), - "module_definition" => make_container_chunk( - node, - ChunkKind::Module, - ocaml_named_text(node, source, &["module_name"]), - source, - ocaml_module_recurse(node), - ), - "module_type_definition" => make_candidate( - node, - ChunkKind::Interface, - format!("modtype_{}", ocaml_named_text(node, source, &["module_type_name"])?), - NameStyle::Named, - signature_for_node(node, source), - ocaml_module_type_recurse(node), - source, - ), - "class_definition" => make_container_chunk( - node, - ChunkKind::Class, - ocaml_named_text(node, source, &["class_name"]), - source, - ocaml_class_recurse(node), - ), - "class_type_definition" => make_candidate( - node, - ChunkKind::Iface, - format!("classtype_{}", ocaml_named_text(node, source, &["class_type_name"])?), - NameStyle::Named, - signature_for_node(node, source), - ocaml_class_type_recurse(node), - source, - ), - "type_definition" => make_kind_chunk( - node, - ChunkKind::Type, - ocaml_named_text(node, source, &["type_constructor"]), - source, - None, - ), - "exception_definition" => make_candidate( - node, - ChunkKind::Constructor, - format!("exception_{}", ocaml_named_text(node, source, &["constructor_name"])?), - NameStyle::Named, - signature_for_node(node, source), - None, - source, - ), - "value_definition" => classify_ocaml_value_definition(node, source)?, - "value_specification" => make_kind_chunk( - node, - ChunkKind::Val, - ocaml_named_text(node, source, &["value_name"]), - source, - None, - ), - _ => return None, - }) -} - -fn classify_ocaml_value_definition<'t>( - node: Node<'t>, - source: &str, -) -> Option> { - let name = ocaml_named_text(node, source, &["value_name"])?; - let recurse = ocaml_value_recurse(node); - if ocaml_value_definition_is_function(node) { - Some(make_kind_chunk(node, ChunkKind::Function, Some(name), source, recurse)) - } else { - Some(make_kind_chunk(node, ChunkKind::Val, Some(name), source, recurse)) - } -} - -fn ocaml_named_text(node: Node<'_>, source: &str, kinds: &[&str]) -> Option { - find_named_text(node, source, kinds).and_then(sanitize_identifier) -} - -fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> { - if kinds.iter().any(|kind| node.kind() == *kind) { - return Some(node_text(source, node.start_byte(), node.end_byte())); - } - for child in named_children(node) { - if let Some(text) = find_named_text(child, source, kinds) { - return Some(text); - } - } - None -} - -fn ocaml_module_recurse(node: Node<'_>) -> Option> { - recurse_into(node, ChunkContext::ClassBody, &[], &["module_binding"]).and_then(|binding| { - recurse_into(binding.node, ChunkContext::ClassBody, &["body"], &["structure", "signature"]) - }) -} - -fn ocaml_module_type_recurse(node: Node<'_>) -> Option> { - recurse_into(node, ChunkContext::ClassBody, &["body"], &["signature"]) -} - -fn ocaml_class_recurse(node: Node<'_>) -> Option> { - recurse_into(node, ChunkContext::ClassBody, &[], &["class_binding"]).and_then(|binding| { - recurse_into(binding.node, ChunkContext::ClassBody, &["body"], &["object_expression"]) - }) -} - -fn ocaml_class_type_recurse(node: Node<'_>) -> Option> { - recurse_into(node, ChunkContext::ClassBody, &[], &["class_type_binding"]).and_then(|binding| { - recurse_into(binding.node, ChunkContext::ClassBody, &["body"], &["class_body_type"]) - }) -} - -fn ocaml_method_recurse(node: Node<'_>) -> Option> { - recurse_into(node, ChunkContext::FunctionBody, &["body"], &[ - "function_expression", - "match_expression", - "let_expression", - ]) -} - -fn ocaml_value_recurse(node: Node<'_>) -> Option> { - recurse_into(node, ChunkContext::FunctionBody, &[], &["let_binding"]).and_then(|binding| { - named_children(binding.node) - .into_iter() - .find(|child| { - matches!(child.kind(), "function_expression" | "match_expression" | "let_expression") - }) - .map(|child| RecurseSpec { node: child, context: ChunkContext::FunctionBody }) - }) -} - -fn ocaml_value_definition_is_function(node: Node<'_>) -> bool { - recurse_into(node, ChunkContext::FunctionBody, &[], &["let_binding"]).is_some_and(|binding| { - named_children(binding.node) - .into_iter() - .any(|child| matches!(child.kind(), "parameter" | "function_expression")) - }) -} diff --git a/crates/pi-natives/src/chunk/ast_perl.rs b/crates/pi-natives/src/chunk/ast_perl.rs deleted file mode 100644 index afe7a9d0a..000000000 --- a/crates/pi-natives/src/chunk/ast_perl.rs +++ /dev/null @@ -1,137 +0,0 @@ -//! Language-specific chunk classifier for Perl. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct PerlClassifier; - -const PERL_SHARED_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "use_statement", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "conditional_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "for_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "loop_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), -]; - -const PERL_TABLES: ClassifierTables = ClassifierTables { - root: PERL_SHARED_RULES, - class: &[], - function: PERL_SHARED_RULES, - structural_overrides: StructuralOverrides { - extra_trivia: &[], - preserved_trivia: &[], - extra_root_wrappers: &["statement_list"], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, -}; - -impl LangClassifier for PerlClassifier { - fn tables(&self) -> &'static ClassifierTables { - &PERL_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root | ChunkContext::FunctionBody => classify_perl_node(node, source), - ChunkContext::ClassBody => None, - } - } -} - -fn classify_perl_node<'t>(node: Node<'t>, source: &str) -> Option> { - let body_recurse = || recurse_into(node, ChunkContext::FunctionBody, &["body"], &["block"]); - - Some(match node.kind() { - "package_statement" => { - make_kind_chunk(node, ChunkKind::Module, Some(perl_name(node, source)?), source, None) - }, - "subroutine_declaration_statement" => make_kind_chunk( - node, - ChunkKind::Function, - Some(perl_name(node, source)?), - source, - body_recurse(), - ), - "expression_statement" => classify_perl_statement(node, source), - _ => return None, - }) -} - -fn classify_perl_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - if perl_declares_variable(node) { - group_candidate(node, ChunkKind::Declarations, source) - } else { - group_candidate(node, ChunkKind::Statements, source) - } -} - -fn perl_declares_variable(node: Node<'_>) -> bool { - if node.kind() == "variable_declaration" { - return true; - } - - if node.kind() == "assignment_expression" - && named_children(node) - .into_iter() - .any(|child| child.kind() == "variable_declaration") - { - return true; - } - - named_children(node).into_iter().any(perl_declares_variable) -} - -fn perl_name(node: Node<'_>, source: &str) -> Option { - find_named_text(node, source, &["bareword", "package", "varname"]).and_then(sanitize_identifier) -} - -fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> { - if kinds.iter().any(|kind| node.kind() == *kind) { - return Some(node_text(source, node.start_byte(), node.end_byte())); - } - - for child in named_children(node) { - if let Some(text) = find_named_text(child, source, kinds) { - return Some(text); - } - } - - None -} diff --git a/crates/pi-natives/src/chunk/ast_powershell.rs b/crates/pi-natives/src/chunk/ast_powershell.rs deleted file mode 100644 index a6aa20636..000000000 --- a/crates/pi-natives/src/chunk/ast_powershell.rs +++ /dev/null @@ -1,290 +0,0 @@ -//! PowerShell-specific chunk classifier. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct PowershellClassifier; - -static POWERSHELL_TABLES: ClassifierTables = ClassifierTables { - root: &[ - semantic_rule( - "param_block", - ChunkKind::Parameters, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "flow_control_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - ], - class: &[], - function: &[ - semantic_rule( - "class_method_parameter_list", - ChunkKind::Parameters, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "param_block", - ChunkKind::Parameters, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "flow_control_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - ], - structural_overrides: StructuralOverrides { - extra_trivia: &[ - "function_name", - "simple_name", - "type_literal", - "switch_condition", - ], - preserved_trivia: &[], - extra_root_wrappers: &[], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, -}; - -impl LangClassifier for PowershellClassifier { - fn tables(&self) -> &'static ClassifierTables { - &POWERSHELL_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_root_custom(node, source), - ChunkContext::ClassBody => classify_class_custom(node, source), - ChunkContext::FunctionBody => classify_function_custom(node, source), - } - } -} - -fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "statement_list" => make_container_chunk( - node, - ChunkKind::Body, - None, - source, - Some(recurse_self(node, ChunkContext::Root)), - ), - "class_statement" => make_container_chunk( - node, - ChunkKind::Class, - Some(powershell_name(node, source)?), - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - ), - "function_statement" => make_container_chunk( - node, - ChunkKind::Function, - Some(powershell_name(node, source)?), - source, - Some(recurse_self(node, ChunkContext::FunctionBody)), - ), - "pipeline" => classify_powershell_pipeline(node, source), - "switch_statement" | "if_statement" | "foreach_statement" => { - return classify_function_custom(node, source); - }, - _ => return None, - }) -} - -fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "class_property_definition" => match powershell_name(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None), - None => group_candidate(node, ChunkKind::Fields, source), - }, - "class_method_definition" => classify_class_method(node, source)?, - _ => return None, - }) -} - -fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "script_block" => make_container_chunk( - node, - block_kind_for_parent(node), - None, - source, - Some(recurse_self(node, ChunkContext::FunctionBody)), - ), - "script_block_body" | "statement_block" => make_container_chunk( - node, - ChunkKind::Block, - None, - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_list"]), - ), - "pipeline" => classify_powershell_pipeline(node, source), - "if_statement" => make_container_chunk( - node, - ChunkKind::If, - None, - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_block"]), - ), - "foreach_statement" => make_container_chunk( - node, - ChunkKind::Loop, - None, - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_block"]), - ), - "switch_statement" => make_container_chunk( - node, - ChunkKind::Switch, - None, - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["switch_body"]), - ), - "switch_clauses" => make_container_chunk( - node, - ChunkKind::Cases, - None, - source, - Some(recurse_self(node, ChunkContext::FunctionBody)), - ), - "switch_clause" => make_container_chunk( - node, - ChunkKind::Case, - None, - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_block"]), - ), - _ => return None, - }) -} - -fn classify_class_method<'t>(node: Node<'t>, source: &str) -> Option> { - let name = powershell_name(node, source)?; - let class_name = powershell_name(node.parent()?, source)?; - let (kind, identifier) = if name == "new" || name == class_name { - (ChunkKind::Constructor, None) - } else { - (ChunkKind::Function, Some(name)) - }; - - Some(make_container_chunk( - node, - kind, - identifier, - source, - Some(recurse_self(node, ChunkContext::FunctionBody)), - )) -} - -fn classify_powershell_pipeline<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - if let Some(command_name) = powershell_command_name(node, source) - && matches!(command_name.as_str(), "using" | "using-module" | "Import-Module") - { - return group_candidate(node, ChunkKind::Imports, source); - } - - if let Some((name, script_block)) = assigned_script_block(node, source) { - return make_container_chunk_from( - node, - node, - ChunkKind::Block, - Some(name), - source, - Some(recurse_self(script_block, ChunkContext::FunctionBody)), - ); - } - - if child_by_kind(node, &["assignment_expression"]).is_some() { - group_candidate(node, ChunkKind::Declarations, source) - } else { - group_candidate(node, ChunkKind::Statements, source) - } -} - -fn assigned_script_block<'t>(node: Node<'t>, source: &str) -> Option<(String, Node<'t>)> { - let assignment = child_by_kind(node, &["assignment_expression"])?; - let lhs = child_by_kind(assignment, &["left_assignment_expression"])?; - let name = sanitize_identifier( - node_text(source, lhs.start_byte(), lhs.end_byte()).trim_start_matches('$'), - )?; - let script_block = named_children(assignment) - .into_iter() - .filter(|child| child.kind() != "left_assignment_expression") - .find_map(find_script_block)?; - Some((name, script_block)) -} - -fn find_script_block(node: Node<'_>) -> Option> { - if node.kind() == "script_block" { - return Some(node); - } - for child in named_children(node) { - if let Some(script_block) = find_script_block(child) { - return Some(script_block); - } - } - None -} - -fn block_kind_for_parent(node: Node<'_>) -> ChunkKind { - match node.parent().map(|parent| parent.kind()) { - Some("function_statement" | "class_method_definition") => ChunkKind::Body, - _ => ChunkKind::Block, - } -} - -fn powershell_name(node: Node<'_>, source: &str) -> Option { - find_named_text(node, source, &[ - "function_name", - "simple_name", - "member_name", - "type_identifier", - "variable", - ]) - .and_then(|text| sanitize_identifier(text.trim_start_matches('$'))) -} - -fn powershell_command_name(node: Node<'_>, source: &str) -> Option { - find_named_text(node, source, &["command_name"]).and_then(sanitize_identifier) -} - -fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> { - if kinds.iter().any(|kind| node.kind() == *kind) { - return Some(node_text(source, node.start_byte(), node.end_byte())); - } - - for child in named_children(node) { - if let Some(text) = find_named_text(child, source, kinds) { - return Some(text); - } - } - - None -} diff --git a/crates/pi-natives/src/chunk/ast_proto.rs b/crates/pi-natives/src/chunk/ast_proto.rs deleted file mode 100644 index d1c8f9191..000000000 --- a/crates/pi-natives/src/chunk/ast_proto.rs +++ /dev/null @@ -1,193 +0,0 @@ -//! Chunk classifier for Protocol Buffers. -//! -//! Mirror the grammar's declaration structure directly: the root owns headers, -//! imports, options, messages, enums, and services; message bodies own fields, -//! oneofs, and nested messages/enums; services own rpc declarations and service -//! options; rpc blocks may contain rpc-scoped options. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct ProtoClassifier; - -const PROTO_ROOT_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "syntax", - ChunkKind::Headers, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "package", - ChunkKind::Headers, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "import", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "option", - ChunkKind::Options, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const PROTO_CLASS_RULES: &[super::classify::SemanticRule] = &[semantic_rule( - "option", - ChunkKind::Options, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, -)]; - -const PROTO_TABLES: ClassifierTables = ClassifierTables { - root: PROTO_ROOT_RULES, - class: PROTO_CLASS_RULES, - function: &[], - structural_overrides: super::classify::StructuralOverrides::EMPTY, -}; - -impl LangClassifier for ProtoClassifier { - fn tables(&self) -> &'static ClassifierTables { - &PROTO_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_proto_root(node, source), - ChunkContext::ClassBody => classify_proto_class(node, source), - ChunkContext::FunctionBody => None, - } - } -} - -fn classify_proto_root<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "message" => make_named_proto_chunk( - node, - ChunkKind::Type, - format!("msg_{}", proto_name(node, source)?), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["message_body"]), - ), - "enum" => make_container_chunk( - node, - ChunkKind::Enum, - proto_name(node, source), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["enum_body"]), - ), - "service" => make_named_proto_chunk( - node, - ChunkKind::Interface, - format!("service_{}", proto_name(node, source)?), - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - ), - _ => return None, - }) -} - -fn classify_proto_class<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "field" if is_proto_message_field(node) => { - make_kind_chunk(node, ChunkKind::Field, proto_name(node, source), source, None) - }, - "oneof" => make_named_proto_chunk( - node, - ChunkKind::Either, - format!("oneof_{}", proto_name(node, source)?), - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - ), - "oneof_field" => { - make_kind_chunk(node, ChunkKind::Field, proto_name(node, source), source, None) - }, - "message" => make_named_proto_chunk( - node, - ChunkKind::Type, - format!("msg_{}", proto_name(node, source)?), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["message_body"]), - ), - "enum" => make_container_chunk( - node, - ChunkKind::Enum, - proto_name(node, source), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["enum_body"]), - ), - "enum_field" => { - make_kind_chunk(node, ChunkKind::Variant, proto_name(node, source), source, None) - }, - "rpc" => make_named_proto_chunk( - node, - ChunkKind::Proc, - format!("rpc_{}", proto_name(node, source)?), - source, - proto_rpc_recurse(node), - ), - _ => return None, - }) -} - -fn make_named_proto_chunk<'t>( - node: Node<'t>, - kind: ChunkKind, - identifier: impl Into>, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'t> { - make_candidate( - node, - kind, - identifier, - NameStyle::Named, - signature_for_node(node, source), - recurse, - source, - ) -} - -fn is_proto_message_field(node: Node<'_>) -> bool { - node - .parent() - .is_some_and(|parent| parent.kind() == "message_body") -} - -fn proto_rpc_recurse(node: Node<'_>) -> Option> { - let has_nested_option = named_children(node) - .into_iter() - .any(|child| child.kind() == "option"); - if has_nested_option { - Some(recurse_self(node, ChunkContext::ClassBody)) - } else { - None - } -} - -fn proto_name(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["message_name", "enum_name", "service_name", "rpc_name", "identifier"]) - .and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))) -} diff --git a/crates/pi-natives/src/chunk/ast_python.rs b/crates/pi-natives/src/chunk/ast_python.rs deleted file mode 100644 index a7fd324e8..000000000 --- a/crates/pi-natives/src/chunk/ast_python.rs +++ /dev/null @@ -1,244 +0,0 @@ -//! Language-specific chunk classifiers for Python and Starlark. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, WrapperSignature, - WrapperTransform, promote_wrapper_candidate, semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct PythonClassifier; - -const ROOT_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "import_statement", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "import_from_statement", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "assignment", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "function_definition", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "class_definition", - ChunkKind::Class, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "if_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "for_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "while_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "try_statement", - ChunkKind::Try, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "with_statement", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "expression_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "global_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const CLASS_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "expression_statement", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "assignment", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "type_alias_statement", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::None, - ), -]; - -const FUNCTION_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "if_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "for_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "while_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "try_statement", - ChunkKind::Try, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "with_statement", - ChunkKind::Block, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "elif_clause", - ChunkKind::Elif, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "except_clause", - ChunkKind::Except, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "match_statement", - ChunkKind::Match, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), -]; - -const PYTHON_TABLES: ClassifierTables = ClassifierTables { - root: ROOT_RULES, - class: CLASS_RULES, - function: FUNCTION_RULES, - structural_overrides: super::classify::StructuralOverrides::EMPTY, -}; - -impl LangClassifier for PythonClassifier { - fn tables(&self) -> &'static ClassifierTables { - &PYTHON_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root | ChunkContext::ClassBody if node.kind() == "decorated_definition" => { - promote_wrapper_candidate(self, context, node, source, WrapperTransform { - signature: WrapperSignature::Wrapper, - ..WrapperTransform::default() - }) - .or_else(|| Some(positional_candidate(node, ChunkKind::Block, source))) - }, - ChunkContext::ClassBody if node.kind() == "function_definition" => { - Some(classify_class_method(node, source)) - }, - _ => None, - } - } - - fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option> { - let _ = source; - Some(group_candidate(node, ChunkKind::Statements, source)) - } -} - -fn classify_class_method<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string()); - let kind = if name == "__init__" || name == "__new__" { - ChunkKind::Constructor - } else { - ChunkKind::Function - }; - let identifier = if kind == ChunkKind::Constructor { - None - } else { - Some(name) - }; - make_kind_chunk( - node, - kind, - identifier, - source, - resolve_recurse(node, ChunkContext::FunctionBody), - ) -} diff --git a/crates/pi-natives/src/chunk/ast_r.rs b/crates/pi-natives/src/chunk/ast_r.rs deleted file mode 100644 index 9f62eb6e1..000000000 --- a/crates/pi-natives/src/chunk/ast_r.rs +++ /dev/null @@ -1,173 +0,0 @@ -//! R-specific chunk classifier. - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier}, - common::*, - kind::ChunkKind, -}; - -pub struct RClassifier; - -static R_TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: super::classify::StructuralOverrides::EMPTY, -}; - -impl LangClassifier for RClassifier { - fn tables(&self) -> &'static ClassifierTables { - &R_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_root_custom(node, source), - ChunkContext::FunctionBody => classify_function_custom(node, source), - _ => None, - } - } -} - -fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - // ── Imports ── - "call" if is_import_call(node, source) => group_candidate(node, ChunkKind::Imports, source), - "call" => group_candidate(node, ChunkKind::Statements, source), - - // ── Function / value assignments ── - "binary_operator" => classify_assignment(node, source, ChunkScope::Root)?, - - // ── Control flow at script scope ── - "if_statement" => control_candidate(node, ChunkKind::If, source, recurse_if(node)), - "for_statement" | "while_statement" | "repeat_statement" => { - control_candidate(node, ChunkKind::Loop, source, recurse_loop(node)) - }, - - // ── Bare expressions ── - "identifier" | "subset" | "subset2" | "extract_operator" => { - group_candidate(node, ChunkKind::Statements, source) - }, - - _ => return None, - }) -} - -fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - // ── Local assignments ── - "binary_operator" => classify_assignment(node, source, ChunkScope::Function)?, - - // ── Control flow ── - "if_statement" => control_candidate(node, ChunkKind::If, source, recurse_if(node)), - "for_statement" | "while_statement" | "repeat_statement" => { - control_candidate(node, ChunkKind::Loop, source, recurse_loop(node)) - }, - - // ── Calls / bare expressions ── - "call" | "identifier" | "subset" | "subset2" | "extract_operator" | "break" | "next" - | "return" => group_candidate(node, ChunkKind::Statements, source), - - _ => return None, - }) -} - -#[derive(Clone, Copy)] -enum ChunkScope { - Root, - Function, -} - -fn classify_assignment<'t>( - node: Node<'t>, - source: &str, - scope: ChunkScope, -) -> Option> { - let (lhs, rhs) = assignment_sides(node, source)?; - - if rhs.kind() == "function_definition" { - let name = simple_lhs_name(lhs, source).unwrap_or_else(|| "anonymous".to_string()); - return Some(make_kind_chunk_from( - node, - rhs, - ChunkKind::Function, - Some(name), - source, - recurse_body(rhs, ChunkContext::FunctionBody), - )); - } - - match (scope, simple_lhs_name(lhs, source)) { - (ChunkScope::Root, Some(name)) => { - Some(make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)) - }, - (ChunkScope::Function, Some(name)) if spans_multiple_lines(node) => { - Some(make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)) - }, - _ => Some(group_candidate( - node, - match scope { - ChunkScope::Root => ChunkKind::Declarations, - ChunkScope::Function => ChunkKind::Statements, - }, - source, - )), - } -} - -fn assignment_sides<'t>(node: Node<'t>, source: &str) -> Option<(Node<'t>, Node<'t>)> { - if node.kind() != "binary_operator" { - return None; - } - - let operator = node.child_by_field_name("operator")?; - let operator_text = node_text(source, operator.start_byte(), operator.end_byte()); - if !matches!(operator_text, "<-" | "<<-" | "=") { - return None; - } - - Some((node.child_by_field_name("lhs")?, node.child_by_field_name("rhs")?)) -} - -fn simple_lhs_name(lhs: Node<'_>, source: &str) -> Option { - (lhs.kind() == "identifier") - .then(|| extract_identifier(lhs, source)) - .flatten() -} - -fn is_import_call(node: Node<'_>, source: &str) -> bool { - matches!( - extract_identifier(node, source).as_deref(), - Some("library" | "require" | "requireNamespace" | "source") - ) -} - -fn recurse_if(node: Node<'_>) -> Option> { - recurse_into(node, ChunkContext::FunctionBody, &["consequence", "alternative"], &[ - "braced_expression", - ]) -} - -fn recurse_loop(node: Node<'_>) -> Option> { - recurse_into(node, ChunkContext::FunctionBody, &["body"], &["braced_expression"]) -} - -fn control_candidate<'t>( - node: Node<'t>, - kind: ChunkKind, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'t> { - make_candidate(node, kind, None, NameStyle::Named, None, recurse, source) -} - -fn spans_multiple_lines(node: Node<'_>) -> bool { - node.start_position().row != node.end_position().row -} diff --git a/crates/pi-natives/src/chunk/ast_ruby_lua.rs b/crates/pi-natives/src/chunk/ast_ruby_lua.rs deleted file mode 100644 index 2f289d8f1..000000000 --- a/crates/pi-natives/src/chunk/ast_ruby_lua.rs +++ /dev/null @@ -1,253 +0,0 @@ -//! Language-specific chunk classifiers for Ruby and Lua. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct RubyLuaClassifier; - -const RUBY_LUA_ROOT_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "function_definition", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "method", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "singleton_method", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "class", - ChunkKind::Class, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "module", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "if_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "unless", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "while_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "for_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "assignment", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "expression_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "function_call", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const RUBY_LUA_CLASS_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "class", - ChunkKind::Class, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "module", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "assignment", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "call", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "command", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "identifier", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const RUBY_LUA_FUNCTION_RULES: &[super::classify::SemanticRule] = &[ - semantic_rule( - "if_statement", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "unless", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "case_statement", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "case_match", - ChunkKind::Switch, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "while_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "for_statement", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "assignment", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const RUBY_LUA_TABLES: ClassifierTables = ClassifierTables { - root: RUBY_LUA_ROOT_RULES, - class: RUBY_LUA_CLASS_RULES, - function: RUBY_LUA_FUNCTION_RULES, - structural_overrides: StructuralOverrides { - extra_trivia: &[], - preserved_trivia: &[], - extra_root_wrappers: &[], - preserved_root_wrappers: &["module"], - absorbable_attrs: &[], - }, -}; - -impl LangClassifier for RubyLuaClassifier { - fn tables(&self) -> &'static ClassifierTables { - &RUBY_LUA_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match (context, node.kind()) { - (ChunkContext::Root, "command" | "call") => { - let target = extract_identifier(node, source); - Some(match target.as_deref() { - Some("require" | "require_relative" | "load" | "autoload") => { - group_candidate(node, ChunkKind::Imports, source) - }, - _ => group_candidate(node, ChunkKind::Statements, source), - }) - }, - (ChunkContext::ClassBody, "method" | "singleton_method") => { - let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string()); - let kind = if name == "initialize" { - ChunkKind::Constructor - } else { - ChunkKind::Function - }; - let identifier = (kind != ChunkKind::Constructor).then_some(name); - Some(make_kind_chunk( - node, - kind, - identifier, - source, - resolve_recurse(node, ChunkContext::FunctionBody), - )) - }, - _ => None, - } - } -} diff --git a/crates/pi-natives/src/chunk/ast_rust.rs b/crates/pi-natives/src/chunk/ast_rust.rs deleted file mode 100644 index bff402c17..000000000 --- a/crates/pi-natives/src/chunk/ast_rust.rs +++ /dev/null @@ -1,404 +0,0 @@ -//! Rust-specific chunk classifier. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::*, - kind::ChunkKind, -}; - -pub struct RustClassifier; - -const ROOT_RULES: &[super::classify::SemanticRule] = &[ - // ── Imports ── - semantic_rule( - "use_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "extern_crate_declaration", - ChunkKind::Imports, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // ── Functions ── - semantic_rule( - "function_item", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "function_definition", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // ── Containers ── - semantic_rule( - "struct_item", - ChunkKind::Struct, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "enum_item", - ChunkKind::Enum, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "trait_item", - ChunkKind::Trait, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "mod_item", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - semantic_rule( - "foreign_block", - ChunkKind::Module, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - // ── Types ── - semantic_rule( - "type_item", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::ClassBody), - ), - // ── Macros ── - semantic_rule( - "macro_definition", - ChunkKind::Macro, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "macro_rule", - ChunkKind::Macro, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // ── Statics / consts ── - semantic_rule( - "static_item", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "const_item", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // ── Attributes ── - semantic_rule( - "inner_attribute_item", - ChunkKind::Attrs, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - // ── Expression statements ── - semantic_rule( - "expression_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const CLASS_RULES: &[super::classify::SemanticRule] = &[ - // ── Functions (methods in impl/trait) ── - semantic_rule( - "function_item", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - semantic_rule( - "function_definition", - ChunkKind::Function, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::Auto(ChunkContext::FunctionBody), - ), - // ── Types ── - semantic_rule( - "type_item", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::None, - ), - semantic_rule( - "type_alias", - ChunkKind::Type, - RuleStyle::Named, - NamingMode::AutoIdentifier, - RecurseMode::None, - ), - // ── Consts / macros in class body ── - semantic_rule( - "const_item", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "macro_invocation", - ChunkKind::Fields, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const FUNCTION_RULES: &[super::classify::SemanticRule] = &[ - // ── Control flow ── - semantic_rule( - "match_expression", - ChunkKind::Match, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "loop_expression", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "while_expression", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "for_expression", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - // ── Expression statements ── - semantic_rule( - "expression_statement", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), -]; - -const RUST_TABLES: ClassifierTables = ClassifierTables { - root: ROOT_RULES, - class: CLASS_RULES, - function: FUNCTION_RULES, - structural_overrides: StructuralOverrides::EMPTY, -}; - -impl LangClassifier for RustClassifier { - fn tables(&self) -> &'static ClassifierTables { - &RUST_TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_root_custom(node, source), - ChunkContext::ClassBody => classify_class_custom(node, source), - ChunkContext::FunctionBody => classify_function_custom(node, source), - } - } -} - -fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - // ── Impl blocks (custom name extraction) ── - "impl_item" => { - let name = extract_impl_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - Some(make_container_chunk( - node, - ChunkKind::Impl, - Some(name), - source, - recurse_into(node, ChunkContext::ClassBody, &["body"], &["declaration_list"]), - )) - }, - - // ── Variables (conditional auto-id vs group) ── - "let_declaration" => Some(match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None), - None => group_candidate(node, ChunkKind::Declarations, source), - }), - - _ => None, - } -} - -fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - // ── Fields (conditional auto-id vs group) ── - "field_declaration" => Some(match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None), - None => group_candidate(node, ChunkKind::Fields, source), - }), - - // ── Enum variants (conditional auto-id vs group) ── - "enum_variant" => Some(match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None), - None => group_candidate(node, ChunkKind::Variants, source), - }), - - // ── Attributes (explicitly return None — absorbed by framework) ── - "attribute_item" => None, - - _ => None, - } -} - -fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option> { - let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody); - match node.kind() { - // ── Control flow with recurse ── - "if_expression" => Some(make_candidate( - node, - ChunkKind::If, - None, - NameStyle::Named, - None, - fn_recurse(), - source, - )), - - // ── Blocks ── - "unsafe_block" | "async_block" | "const_block" | "block_expression" => Some(make_candidate( - node, - ChunkKind::Block, - None, - NameStyle::Named, - None, - fn_recurse(), - source, - )), - - // ── Variables (conditional line span) ── - "let_declaration" => { - let span = line_span(node.start_position().row + 1, node.end_position().row + 1); - Some(if span > 1 { - match extract_identifier(node, source) { - Some(name) => make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None), - None => group_candidate(node, ChunkKind::Let, source), - } - } else { - group_candidate(node, ChunkKind::Let, source) - }) - }, - - _ => None, - } -} - -/// Extract the name for an `impl` block. -/// -/// - Plain impl: `impl Foo` → `"Foo"` -/// - Trait impl: `impl Trait for Foo` → `"Trait_for_Foo"` -/// - Scoped trait: `impl fmt::Display for Foo` → `"Display_for_Foo"` -fn extract_impl_name(node: Node<'_>, source: &str) -> Option { - // Collect ALL children (including anonymous keywords like `for`). - let all_children: Vec> = (0..node.child_count()) - .filter_map(|i| node.child(i)) - .collect(); - - // Find the `for` keyword position. - let for_index = all_children - .iter() - .position(|c| node_text(source, c.start_byte(), c.end_byte()) == "for"); - - if let Some(fi) = for_index { - // Trait impl: trait name before `for`, type name after `for`. - let trait_node = all_children[..fi].iter().rev().find(|c| { - matches!(c.kind(), "type_identifier" | "scoped_type_identifier" | "generic_type") - }); - let type_node = all_children[fi + 1..].iter().find(|c| { - matches!(c.kind(), "type_identifier" | "scoped_type_identifier" | "generic_type") - }); - - if let (Some(tn), Some(ty)) = (trait_node, type_node) { - let trait_name = extract_last_type_identifier(*tn, source) - .or_else(|| sanitize_identifier(node_text(source, tn.start_byte(), tn.end_byte())))?; - let type_name = extract_last_type_identifier(*ty, source) - .or_else(|| sanitize_identifier(node_text(source, ty.start_byte(), ty.end_byte())))?; - return Some(format!("{trait_name}_for_{type_name}")); - } - } - - // Plain impl: take the last type_identifier. - let type_ids: Vec> = named_children(node) - .into_iter() - .filter(|c| c.kind() == "type_identifier") - .collect(); - type_ids - .last() - .and_then(|n| sanitize_identifier(node_text(source, n.start_byte(), n.end_byte()))) -} - -/// Recursively find the innermost `type_identifier` from a type node. -/// -/// Handles scoped types like `fmt::Display` by traversing into -/// `scoped_type_identifier` and `generic_type` children. -fn extract_last_type_identifier(node: Node<'_>, source: &str) -> Option { - if node.kind() == "type_identifier" { - return sanitize_identifier(node_text(source, node.start_byte(), node.end_byte())); - } - - let mut result = None; - for child in named_children(node) { - if child.kind() == "type_identifier" { - result = sanitize_identifier(node_text(source, child.start_byte(), child.end_byte())); - } else if matches!(child.kind(), "scoped_type_identifier" | "generic_type") - && let Some(inner) = extract_last_type_identifier(child, source) - { - result = Some(inner); - } - } - result -} diff --git a/crates/pi-natives/src/chunk/ast_sql.rs b/crates/pi-natives/src/chunk/ast_sql.rs deleted file mode 100644 index 1398b17a6..000000000 --- a/crates/pi-natives/src/chunk/ast_sql.rs +++ /dev/null @@ -1,282 +0,0 @@ -//! SQL-specific chunk classifier. - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier, StructuralOverrides}, - common::*, - kind::ChunkKind, -}; - -pub struct SqlClassifier; - -impl LangClassifier for SqlClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: StructuralOverrides { - extra_trivia: &["empty_statement", "dollar_quote", "keyword_from"], - preserved_trivia: &[], - extra_root_wrappers: &[], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_sql_root(node, source), - ChunkContext::ClassBody => classify_sql_class(node, source), - ChunkContext::FunctionBody => classify_sql_function(node, source), - } - } -} - -fn classify_sql_root<'t>(node: Node<'t>, source: &str) -> Option> { - if node.kind() == "statement" { - return classify_sql_statement_root(node, source); - } - - classify_sql_root_node(node, node, source).or_else(|| classify_sql_query_node(node, source)) -} - -fn classify_sql_statement_root<'t>(node: Node<'t>, source: &str) -> Option> { - let children = named_children(node); - if children.len() == 1 { - return classify_sql_root_node(node, children[0], source) - .or_else(|| classify_sql_query_node(children[0], source)); - } - - if children.iter().any(|child| is_sql_query_kind(child.kind())) { - return Some(make_named_sql_chunk( - node, - ChunkKind::Query, - None, - source, - Some(recurse_self(node, ChunkContext::FunctionBody)), - )); - } - - None -} - -fn classify_sql_root_node<'t>( - range_node: Node<'t>, - node: Node<'t>, - source: &str, -) -> Option> { - Some(match node.kind() { - "create_schema" => make_kind_chunk_from( - range_node, - node, - ChunkKind::Schema, - extract_sql_identifier(node, source), - source, - None, - ), - "create_table" => make_container_chunk_from( - range_node, - node, - ChunkKind::Table, - extract_sql_object_name(node, source), - source, - recurse_into(node, ChunkContext::ClassBody, &[], &["column_definitions"]), - ), - "create_view" => make_named_sql_chunk_from( - range_node, - node, - ChunkKind::Query, - format!( - "view_{}", - extract_sql_object_name(node, source).unwrap_or_else(|| "anonymous".to_string()) - ), - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["create_query"]), - ), - "create_materialized_view" => make_named_sql_chunk_from( - range_node, - node, - ChunkKind::Query, - format!( - "matview_{}", - extract_sql_object_name(node, source).unwrap_or_else(|| "anonymous".to_string()) - ), - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["create_query"]), - ), - "create_function" => make_container_chunk_from( - range_node, - node, - ChunkKind::Function, - extract_sql_object_name(node, source), - source, - recurse_sql_function_query(node), - ), - "create_trigger" => make_named_sql_chunk_from( - range_node, - node, - ChunkKind::Function, - format!( - "trigger_{}", - extract_sql_object_name(node, source).unwrap_or_else(|| "anonymous".to_string()) - ), - source, - None, - ), - "create_index" => make_named_sql_chunk_from( - range_node, - node, - ChunkKind::Key, - format!( - "index_{}", - extract_sql_identifier(node, source).unwrap_or_else(|| "anonymous".to_string()) - ), - source, - None, - ), - _ => return None, - }) -} - -fn classify_sql_class<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "column_definition" => { - make_kind_chunk(node, ChunkKind::Field, extract_sql_identifier(node, source), source, None) - }, - _ => return None, - }) -} - -fn classify_sql_function<'t>(node: Node<'t>, source: &str) -> Option> { - if node.kind() == "statement" { - return Some(make_named_sql_chunk( - node, - ChunkKind::Query, - None, - source, - Some(recurse_self(node, ChunkContext::FunctionBody)), - )); - } - - classify_sql_query_node(node, source) -} - -fn classify_sql_query_node<'t>(node: Node<'t>, source: &str) -> Option> { - Some(match node.kind() { - "insert" => group_candidate(node, ChunkKind::Statements, source), - "keyword_with" => group_candidate(node, ChunkKind::With, source), - "cte" => make_named_sql_chunk( - node, - ChunkKind::With, - format!( - "cte_{}", - extract_sql_identifier(node, source).unwrap_or_else(|| "anonymous".to_string()) - ), - source, - recurse_into(node, ChunkContext::FunctionBody, &[], &["statement"]), - ), - "select" => positional_candidate(node, ChunkKind::Select, source), - "from" => make_named_sql_chunk( - node, - ChunkKind::Query, - "from".to_string(), - source, - Some(recurse_self(node, ChunkContext::FunctionBody)), - ), - "relation" => group_candidate(node, ChunkKind::Relations, source), - "join" => positional_candidate(node, ChunkKind::Join, source), - "where" => positional_candidate(node, ChunkKind::Where, source), - "group_by" => positional_candidate(node, ChunkKind::GroupBy, source), - "order_by" => positional_candidate(node, ChunkKind::OrderBy, source), - _ => return None, - }) -} - -fn make_named_sql_chunk<'t>( - node: Node<'t>, - kind: ChunkKind, - identifier: impl Into>, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'t> { - make_candidate( - node, - kind, - identifier, - NameStyle::Named, - signature_for_node(node, source), - recurse, - source, - ) -} - -fn make_named_sql_chunk_from<'t>( - range_node: Node<'t>, - signature_node: Node<'t>, - kind: ChunkKind, - identifier: impl Into>, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'t> { - make_candidate( - range_node, - kind, - identifier, - NameStyle::Named, - signature_for_node(signature_node, source), - recurse, - source, - ) -} - -fn recurse_sql_function_query(node: Node<'_>) -> Option> { - let body = child_by_kind(node, &["function_body"])?; - recurse_into(body, ChunkContext::FunctionBody, &[], &["statement"]) -} - -fn is_sql_query_kind(kind: &str) -> bool { - matches!( - kind, - "insert" - | "keyword_with" - | "cte" - | "select" - | "from" - | "where" - | "group_by" - | "order_by" - | "join" - ) -} - -fn extract_sql_identifier(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["identifier"]) - .and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))) -} - -fn extract_sql_object_name(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["object_reference"]).and_then(|name| last_identifier(name, source)) -} - -fn last_identifier(node: Node<'_>, source: &str) -> Option { - if node.kind() == "identifier" { - return sanitize_identifier(node_text(source, node.start_byte(), node.end_byte())); - } - - for child in named_children(node).into_iter().rev() { - if let Some(identifier) = last_identifier(child, source) { - return Some(identifier); - } - } - - None -} diff --git a/crates/pi-natives/src/chunk/ast_svelte.rs b/crates/pi-natives/src/chunk/ast_svelte.rs deleted file mode 100644 index 6a76e0fbe..000000000 --- a/crates/pi-natives/src/chunk/ast_svelte.rs +++ /dev/null @@ -1,299 +0,0 @@ -//! Language-specific chunk classifier for Svelte. - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier, StructuralOverrides}, - common::*, - kind::ChunkKind, -}; -use crate::language::SupportLang; - -pub struct SvelteClassifier; - -impl LangClassifier for SvelteClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: StructuralOverrides { - extra_trivia: &[], - preserved_trivia: &[], - extra_root_wrappers: &["document"], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - let include_plain_elements = matches!(context, ChunkContext::Root); - classify_svelte_node(node, source, include_plain_elements) - } -} - -fn classify_svelte_node<'t>( - node: Node<'t>, - source: &str, - include_plain_elements: bool, -) -> Option> { - match node.kind() { - "script_element" => Some(classify_script_element(node, source)), - "style_element" => Some(classify_style_element(node, source)), - "snippet_statement" => Some(classify_snippet_statement(node, source)), - "if_statement" => Some(classify_if_statement(node, source)), - "else_if_statement" => Some(classify_else_if_statement(node, source)), - "else_statement" => Some(classify_else_statement(node, source)), - "each_statement" => Some(classify_each_statement(node, source)), - "await_statement" => Some(classify_await_statement(node, source)), - "then_statement" => Some(classify_then_statement(node, source)), - "catch_statement" => Some(classify_catch_statement(node, source)), - "render_expr" => Some(classify_render_expr(node, source)), - "html_interpolation" => Some(group_candidate(node, ChunkKind::Html, source)), - "interpolation" => Some(group_candidate(node, ChunkKind::Interpolation, source)), - "expression" => Some(group_candidate(node, ChunkKind::Expression, source)), - "element" if include_plain_elements || element_has_structure(node) => { - classify_element(node, source) - }, - _ => None, - } -} - -fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let kind = if has_attribute(node, "module", source) - || attribute_value(node, "context", source).as_deref() == Some("module") - { - ChunkKind::ScriptModule - } else { - ChunkKind::Script - }; - classify_raw_text_block(node, kind, source, SupportLang::TypeScript) -} - -fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let kind = if has_attribute(node, "scoped", source) { - ChunkKind::StyleScoped - } else { - ChunkKind::Style - }; - classify_raw_text_block(node, kind, source, SupportLang::Css) -} - -fn classify_snippet_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let identifier = child_by_kind(node, &["snippet_start_expr"]) - .and_then(|start| child_by_kind(start, &["snippet_name"])) - .and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))); - force_container(make_container_chunk( - node, - ChunkKind::Snippet, - identifier, - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - )) -} - -fn classify_if_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let identifier = block_expr_identifier(node, source, "if_start_expr", &["raw_text_expr"]); - force_container(make_container_chunk( - node, - ChunkKind::If, - identifier, - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - )) -} - -fn classify_else_if_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let identifier = block_expr_identifier(node, source, "else_if_expr", &["raw_text_expr"]) - .map_or_else(|| "if".to_string(), |expr| format!("if_{expr}")); - make_named_container_chunk(node, ChunkKind::Else, identifier, source) -} - -fn classify_else_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - force_container(make_container_chunk( - node, - ChunkKind::Else, - None, - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - )) -} - -fn classify_each_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let expr = block_expr_identifier(node, source, "each_start_expr", &["raw_text_each"]); - let id = expr.map_or_else(|| "each".to_string(), |expr| format!("each_{expr}")); - make_named_container_chunk(node, ChunkKind::Loop, id, source) -} - -fn classify_await_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let expr = block_expr_identifier(node, source, "await_start_expr", &["raw_text_expr"]); - let id = expr.map_or_else(|| "await".to_string(), |expr| format!("await_{expr}")); - make_named_container_chunk(node, ChunkKind::With, id, source) -} - -fn classify_then_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let expr = block_expr_identifier(node, source, "then_expr", &["raw_text_expr"]); - let id = expr.map_or_else(|| "then".to_string(), |expr| format!("then_{expr}")); - make_named_container_chunk(node, ChunkKind::After, id, source) -} - -fn classify_catch_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let identifier = block_expr_identifier(node, source, "catch_expr", &["raw_text_expr"]); - force_container(make_container_chunk( - node, - ChunkKind::Catch, - identifier, - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - )) -} - -fn classify_render_expr<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let identifier = child_by_kind(node, &["snippet_name"]) - .and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))); - make_kind_chunk(node, ChunkKind::Render, identifier, source, None) -} - -fn classify_element<'t>(node: Node<'t>, source: &str) -> Option> { - let tag_name = extract_markup_tag_name(node, source)?; - Some(force_container(make_container_chunk( - node, - ChunkKind::Tag, - Some(tag_name), - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - ))) -} - -fn make_named_container_chunk<'t>( - node: Node<'t>, - kind: ChunkKind, - identifier: impl Into>, - source: &str, -) -> RawChunkCandidate<'t> { - force_container(make_candidate( - node, - kind, - identifier, - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::ClassBody)), - source, - )) -} - -const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> { - candidate.force_recurse = true; - candidate -} - -fn classify_raw_text_block<'t>( - node: Node<'t>, - kind: ChunkKind, - source: &str, - default_language: SupportLang, -) -> RawChunkCandidate<'t> { - let Some(content_node) = child_by_kind(node, &["raw_text"]) else { - return positional_candidate(node, kind, source); - }; - let candidate = with_region_node(positional_candidate(node, kind, source), Some(content_node)); - match resolve_embedded_language(node, source, default_language) { - Some(language) => with_injected_subtree(candidate, language, content_node), - None => candidate, - } -} - -fn block_expr_identifier( - node: Node<'_>, - source: &str, - header_kind: &str, - expr_kinds: &[&str], -) -> Option { - child_by_kind(node, &[header_kind]) - .and_then(|header| child_by_kind(header, expr_kinds)) - .and_then(|expr| sanitize_identifier(node_text(source, expr.start_byte(), expr.end_byte()))) -} - -fn element_has_structure(node: Node<'_>) -> bool { - named_children(node).into_iter().any(|child| { - matches!( - child.kind(), - "snippet_statement" - | "if_statement" - | "else_if_statement" - | "else_statement" - | "each_statement" - | "await_statement" - | "then_statement" - | "catch_statement" - | "render_expr" - | "html_interpolation" - | "interpolation" - | "expression" - | "element" - ) - }) -} - -fn extract_markup_tag_name(node: Node<'_>, source: &str) -> Option { - start_like(node) - .and_then(|start| child_by_kind(start, &["tag_name"])) - .and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte()))) -} - -fn has_attribute(node: Node<'_>, name: &str, source: &str) -> bool { - start_like(node) - .into_iter() - .flat_map(named_children) - .filter(|child| child.kind() == "attribute") - .filter_map(|attr| extract_attribute_name(attr, source)) - .any(|attr_name| attr_name == name) -} - -fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option { - let start = start_like(node)?; - for child in named_children(start) { - if child.kind() != "attribute" { - continue; - } - if extract_attribute_name(child, source).as_deref() != Some(name) { - continue; - } - if let Some(value) = child_by_kind(child, &["attribute_value", "quoted_attribute_value"]) { - return sanitize_identifier(&unquote_text(node_text( - source, - value.start_byte(), - value.end_byte(), - ))); - } - return Some(name.to_string()); - } - None -} - -fn resolve_embedded_language( - node: Node<'_>, - source: &str, - default_language: SupportLang, -) -> Option { - if let Some(language) = attribute_value(node, "lang", source) { - return SupportLang::from_alias(language.as_str()); - } - Some(default_language) -} - -fn extract_attribute_name(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["attribute_name"]) - .and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))) -} - -fn start_like(node: Node<'_>) -> Option> { - child_by_kind(node, &["start_tag", "self_closing_tag"]) -} diff --git a/crates/pi-natives/src/chunk/ast_tlaplus.rs b/crates/pi-natives/src/chunk/ast_tlaplus.rs deleted file mode 100644 index 380e1d860..000000000 --- a/crates/pi-natives/src/chunk/ast_tlaplus.rs +++ /dev/null @@ -1,387 +0,0 @@ -//! Language-specific chunk classification for TLA+ / `PlusCal`. -//! -//! The shared defaults are too noisy for TLA+: the parser exposes a top-level -//! `module` wrapper, `PlusCal` algorithms live inside block comments, and the -//! generated translation section introduces operator definitions we do not want -//! to surface in chunked read/edit views. - -use tree_sitter::Node; - -use super::{ - classify::{ - ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides, - semantic_rule, - }, - common::{ - ChunkContext, RawChunkCandidate, RecurseSpec, child_by_kind, extract_identifier, - make_container_chunk, make_container_chunk_from, make_kind_chunk, recurse_self, - sanitize_identifier, - }, - kind::ChunkKind, - types::ChunkNode, -}; - -pub struct TlaplusClassifier; - -impl LangClassifier for TlaplusClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[ - semantic_rule( - "variable_declaration", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "constant_declaration", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "recursive_declaration", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - ], - class: &[semantic_rule( - "pcal_var_decls", - ChunkKind::Declarations, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - )], - function: &[ - semantic_rule( - "pcal_if", - ChunkKind::If, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "pcal_while", - ChunkKind::Loop, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "pcal_either", - ChunkKind::Either, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "pcal_with", - ChunkKind::With, - RuleStyle::Positional, - NamingMode::None, - RecurseMode::None, - ), - semantic_rule( - "pcal_assign", - ChunkKind::Statements, - RuleStyle::Group, - NamingMode::None, - RecurseMode::None, - ), - ], - structural_overrides: StructuralOverrides { - extra_trivia: &[ - "header_line", - "double_line", - "extends", - "pcal_algorithm_start", - ], - preserved_trivia: &["block_comment"], - extra_root_wrappers: &[], - preserved_root_wrappers: &["module"], - absorbable_attrs: &[], - }, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_root_custom(node, source), - ChunkContext::ClassBody => classify_class_custom(node, source), - _ => None, - } - } - - fn preserve_children( - &self, - parent: &RawChunkCandidate<'_>, - _children: &[RawChunkCandidate<'_>], - ) -> bool { - matches!( - parent.kind, - ChunkKind::Module | ChunkKind::Algo | ChunkKind::Proc | ChunkKind::Process - ) - } - - fn post_process( - &self, - chunks: &mut Vec, - root_children: &mut Vec, - source: &str, - ) { - let ranges = translation_ranges(source); - if ranges.is_empty() { - return; - } - - let removed_by_range = ranges - .iter() - .map(|range| { - chunks - .iter() - .filter(|chunk| { - !chunk.path.is_empty() && chunk_wholly_inside_translation_fence(chunk, *range) - }) - .cloned() - .collect::>() - }) - .collect::>(); - let removed_paths = removed_by_range - .iter() - .flatten() - .map(|chunk| chunk.path.clone()) - .collect::>(); - if removed_paths.is_empty() { - return; - } - - chunks.retain(|chunk| !removed_paths.iter().any(|removed| removed == &chunk.path)); - for chunk in chunks.iter_mut() { - chunk - .children - .retain(|child| !removed_paths.iter().any(|removed| removed == child)); - } - root_children.retain(|child| !removed_paths.iter().any(|removed| removed == child)); - - for (index, range) in ranges.iter().copied().enumerate() { - let removed_chunks = &removed_by_range[index]; - if removed_chunks.is_empty() { - continue; - } - let synthetic = translation_chunk(range, removed_chunks[0].parent_path.clone(), source); - let synthetic_path = synthetic.path.clone(); - if let Some(parent_path) = synthetic.parent_path.as_ref() { - if let Some(parent) = chunks.iter_mut().find(|chunk| chunk.path == *parent_path) { - parent.children.push(synthetic_path); - } - } else { - root_children.push(synthetic_path); - } - chunks.push(synthetic); - } - } -} - -fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "module" => Some(make_container_chunk( - node, - ChunkKind::Module, - tla_identifier(node, source), - source, - Some(recurse_self(node, ChunkContext::Root)), - )), - "operator_definition" => Some(make_kind_chunk( - node, - ChunkKind::Operator, - tla_identifier(node, source), - source, - None, - )), - "module_definition" => Some(make_container_chunk( - node, - ChunkKind::Module, - tla_identifier(node, source), - source, - Some(recurse_self(node, ChunkContext::Root)), - )), - "pcal_algorithm" => Some(make_container_chunk( - node, - ChunkKind::Algo, - tla_identifier(node, source), - source, - recurse_child(node, "pcal_algorithm_body", ChunkContext::ClassBody), - )), - "block_comment" => child_by_kind(node, &["pcal_algorithm"]).map(|algorithm| { - make_container_chunk_from( - node, - algorithm, - ChunkKind::Algo, - tla_identifier(algorithm, source), - source, - recurse_child(algorithm, "pcal_algorithm_body", ChunkContext::ClassBody), - ) - }), - _ => None, - } -} - -fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "pcal_procedure" => Some(make_container_chunk( - node, - ChunkKind::Proc, - tla_identifier(node, source), - source, - recurse_child(node, "pcal_algorithm_body", ChunkContext::ClassBody), - )), - "pcal_process" => Some(make_container_chunk( - node, - ChunkKind::Process, - tla_identifier(node, source), - source, - recurse_child(node, "pcal_algorithm_body", ChunkContext::ClassBody), - )), - _ => None, - } -} - -fn tla_identifier(node: Node<'_>, source: &str) -> Option { - extract_identifier(node, source).or_else(|| { - child_by_kind(node, &["identifier"]) - .and_then(|child| sanitize_identifier(child.utf8_text(source.as_bytes()).ok()?)) - }) -} - -fn recurse_child<'tree>( - node: Node<'tree>, - kind: &'static str, - context: ChunkContext, -) -> Option> { - child_by_kind(node, &[kind]).map(|child| RecurseSpec { node: child, context }) -} - -#[derive(Clone, Copy)] -struct TranslationRange { - start_line: u32, - end_line: u32, -} - -fn translation_ranges(source: &str) -> Vec { - let lines = source.split('\n').collect::>(); - let mut ranges = Vec::new(); - let mut current_start: Option = None; - - for (index, line) in lines.iter().enumerate() { - let line_no = index as u32 + 1; - let trimmed = line.trim(); - if trimmed == r"\* BEGIN TRANSLATION" { - current_start = Some(line_no); - continue; - } - if trimmed == r"\* END TRANSLATION" - && let Some(start_line) = current_start.take() - { - push_translation_range(&mut ranges, start_line, line_no, lines.len() as u32); - } - } - - if let Some(start_line) = current_start { - push_translation_range(&mut ranges, start_line, lines.len() as u32, lines.len() as u32); - } - - ranges -} - -fn push_translation_range( - ranges: &mut Vec, - start_line: u32, - end_line: u32, - total_lines: u32, -) { - let clamped_end = end_line.min(total_lines.max(start_line)); - ranges.push(TranslationRange { start_line, end_line: clamped_end }); -} - -/// True when the chunk's span lies entirely inside the `\* BEGIN` … `\* END` -/// translation fence. We must not treat broad containers (e.g. the `module` -/// chunk spanning the whole file) as translation-only, or the module node is -/// removed and children become orphaned. -const fn chunk_wholly_inside_translation_fence(chunk: &ChunkNode, range: TranslationRange) -> bool { - chunk.start_line >= range.start_line && chunk.end_line <= range.end_line -} - -fn translation_chunk( - range: TranslationRange, - parent_path: Option, - source: &str, -) -> ChunkNode { - let path = match &parent_path { - Some(parent) => format!("{parent}.translation_{}", range.start_line), - None => format!("translation_{}", range.start_line), - }; - let (start_byte, end_byte) = byte_range_for_lines(source, range.start_line, range.end_line); - let checksum = super::chunk_checksum(&source.as_bytes()[start_byte as usize..end_byte as usize]); - ChunkNode { - path, - identifier: Some(range.start_line.to_string()), - kind: ChunkKind::Translation, - leaf: true, - virtual_content: None, - parent_path, - children: Vec::new(), - signature: Some("translation block".to_string()), - start_line: range.start_line, - end_line: range.end_line, - line_count: range.end_line.saturating_sub(range.start_line) + 1, - start_byte, - end_byte, - checksum_start_byte: start_byte, - prologue_end_byte: None, - epilogue_start_byte: None, - checksum, - error: false, - indent: 0, - indent_char: String::new(), - group: false, - } -} - -fn byte_range_for_lines(source: &str, start_line: u32, end_line: u32) -> (u32, u32) { - let mut start_byte = 0usize; - let mut current_line = 1u32; - for (byte_index, byte) in source.bytes().enumerate() { - if current_line == start_line { - start_byte = byte_index; - break; - } - if byte == b'\n' { - current_line += 1; - start_byte = byte_index + 1; - } - } - - let mut end_byte = source.len(); - current_line = 1; - for (byte_index, byte) in source.bytes().enumerate() { - if current_line > end_line { - end_byte = byte_index; - break; - } - if byte == b'\n' { - current_line += 1; - } - } - - (start_byte as u32, end_byte as u32) -} diff --git a/crates/pi-natives/src/chunk/ast_vue.rs b/crates/pi-natives/src/chunk/ast_vue.rs deleted file mode 100644 index a16e57d2f..000000000 --- a/crates/pi-natives/src/chunk/ast_vue.rs +++ /dev/null @@ -1,318 +0,0 @@ -//! Language-specific chunk classifier for Vue single-file components. - -use tree_sitter::Node; - -use super::{ - classify::{ClassifierTables, LangClassifier, StructuralOverrides}, - common::*, - kind::ChunkKind, -}; -use crate::language::SupportLang; - -pub struct VueClassifier; - -impl LangClassifier for VueClassifier { - fn tables(&self) -> &'static ClassifierTables { - static TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: StructuralOverrides { - extra_trivia: &[], - preserved_trivia: &[], - extra_root_wrappers: &["document"], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }, - }; - &TABLES - } - - fn classify_override<'t>( - &self, - context: ChunkContext, - node: Node<'t>, - source: &str, - ) -> Option> { - match context { - ChunkContext::Root => classify_root_node(node, source), - ChunkContext::ClassBody | ChunkContext::FunctionBody => classify_nested_node(node, source), - } - } -} - -fn classify_root_node<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "template_element" => Some(classify_template_element(node, source)), - "script_element" => Some(classify_script_element(node, source)), - "style_element" => Some(classify_style_element(node, source)), - // Vue custom blocks (for example ) currently parse as plain `element` - // nodes at the document root, so infer custom-block semantics from position. - "element" => Some(classify_custom_block(node, source)), - _ => None, - } -} - -fn classify_nested_node<'t>(node: Node<'t>, source: &str) -> Option> { - match node.kind() { - "template_element" => Some(classify_template_element(node, source)), - "element" => classify_element(node, source), - "start_tag" => classify_start_tag(node, source), - "directive_attribute" => Some(classify_directive_attribute(node, source)), - "attribute" => Some(classify_attribute(node, source)), - "interpolation" => Some(make_kind_chunk(node, ChunkKind::Expression, None, source, None)), - "text" => Some(group_candidate(node, ChunkKind::Text, source)), - "raw_text" => Some(group_candidate(node, ChunkKind::Text, source)), - _ => None, - } -} - -fn classify_template_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let recurse = Some(recurse_self(node, ChunkContext::ClassBody)); - if let Some(slot_name) = extract_slot_name(node, source) { - force_container(make_container_chunk(node, ChunkKind::Slot, Some(slot_name), source, recurse)) - } else { - force_container(make_container_chunk(node, ChunkKind::Template, None, source, recurse)) - } -} - -fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let kind = if has_attribute(node, "setup", source) { - ChunkKind::ScriptSetup - } else if attribute_value(node, "context", source).as_deref() == Some("module") { - ChunkKind::ScriptModule - } else { - ChunkKind::Script - }; - classify_raw_text_block(node, kind, source, SupportLang::JavaScript) -} - -fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let kind = if has_attribute(node, "scoped", source) { - ChunkKind::StyleScoped - } else { - ChunkKind::Style - }; - classify_raw_text_block(node, kind, source, SupportLang::Css) -} - -fn classify_custom_block<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let tag_name = extract_markup_tag_name(node, source).unwrap_or_else(|| "anonymous".to_string()); - make_named_container_chunk(node, ChunkKind::Custom, tag_name, source) -} - -fn classify_element<'t>(node: Node<'t>, source: &str) -> Option> { - let tag_name = extract_markup_tag_name(node, source)?; - Some(force_container(make_container_chunk( - node, - ChunkKind::Tag, - Some(tag_name), - source, - Some(recurse_self(node, ChunkContext::ClassBody)), - ))) -} - -fn classify_start_tag<'t>(node: Node<'t>, source: &str) -> Option> { - if !named_children(node) - .into_iter() - .any(|child| matches!(child.kind(), "attribute" | "directive_attribute")) - { - return None; - } - let tag_name = child_by_kind(node, &["tag_name"]) - .and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte()))) - .unwrap_or_else(|| "anonymous".to_string()); - Some(make_named_container_chunk(node, ChunkKind::Attrs, tag_name, source)) -} - -fn classify_attribute<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let name = child_by_kind(node, &["attribute_name"]) - .and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))) - .unwrap_or_else(|| "attr".to_string()); - make_kind_chunk(node, ChunkKind::Attr, Some(name), source, None) -} - -fn classify_directive_attribute<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> { - let raw = node_text(source, node.start_byte(), node.end_byte()).trim(); - let directive_name = - extract_directive_name(node, source).unwrap_or_else(|| "directive".to_string()); - let modifier_suffix = extract_directive_modifiers(node, source) - .filter(|mods| !mods.is_empty()) - .map(|mods| format!("_{mods}")) - .unwrap_or_default(); - if raw.starts_with('@') { - make_named_leaf_chunk( - node, - ChunkKind::Directive, - format!("on_{directive_name}{modifier_suffix}"), - source, - ) - } else if raw.starts_with(':') { - make_named_leaf_chunk( - node, - ChunkKind::Directive, - format!("bind_{directive_name}{modifier_suffix}"), - source, - ) - } else if raw.starts_with('#') { - make_kind_chunk( - node, - ChunkKind::Slot, - Some(format!("{directive_name}{modifier_suffix}")), - source, - None, - ) - } else { - make_named_leaf_chunk( - node, - ChunkKind::Directive, - format!("dir_{directive_name}{modifier_suffix}"), - source, - ) - } -} - -fn make_named_leaf_chunk<'t>( - node: Node<'t>, - kind: ChunkKind, - identifier: impl Into>, - source: &str, -) -> RawChunkCandidate<'t> { - make_candidate( - node, - kind, - identifier, - NameStyle::Named, - signature_for_node(node, source), - None, - source, - ) -} - -fn make_named_container_chunk<'t>( - node: Node<'t>, - kind: ChunkKind, - identifier: impl Into>, - source: &str, -) -> RawChunkCandidate<'t> { - force_container(make_candidate( - node, - kind, - identifier, - NameStyle::Named, - signature_for_node(node, source), - Some(recurse_self(node, ChunkContext::ClassBody)), - source, - )) -} - -const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> { - candidate.force_recurse = true; - candidate -} - -fn classify_raw_text_block<'t>( - node: Node<'t>, - kind: ChunkKind, - source: &str, - default_language: SupportLang, -) -> RawChunkCandidate<'t> { - let Some(content_node) = child_by_kind(node, &["raw_text"]) else { - return positional_candidate(node, kind, source); - }; - let candidate = with_region_node(positional_candidate(node, kind, source), Some(content_node)); - match resolve_embedded_language(node, source, default_language) { - Some(language) => with_injected_subtree(candidate, language, content_node), - None => candidate, - } -} - -fn extract_markup_tag_name(node: Node<'_>, source: &str) -> Option { - start_like(node) - .and_then(|start| child_by_kind(start, &["tag_name"])) - .and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte()))) -} - -fn extract_slot_name(node: Node<'_>, source: &str) -> Option { - let start = start_like(node)?; - named_children(start) - .into_iter() - .find(|child| { - node_text(source, child.start_byte(), child.end_byte()) - .trim() - .starts_with('#') - }) - .and_then(|child| extract_directive_name(child, source)) -} - -fn extract_directive_name(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["directive_name", "directive_value"]) - .and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))) -} - -fn extract_directive_modifiers(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["directive_modifiers"]) - .and_then(|mods| sanitize_identifier(node_text(source, mods.start_byte(), mods.end_byte()))) -} - -fn has_attribute(node: Node<'_>, name: &str, source: &str) -> bool { - start_like(node) - .into_iter() - .flat_map(named_children) - .filter(|child| matches!(child.kind(), "attribute" | "directive_attribute")) - .filter_map(|attr| extract_attribute_name(attr, source)) - .any(|attr_name| attr_name == name) -} - -fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option { - let start = start_like(node)?; - for child in named_children(start) { - if !matches!(child.kind(), "attribute" | "directive_attribute") { - continue; - } - if extract_attribute_name(child, source).as_deref() != Some(name) { - continue; - } - if let Some(value) = - child_by_kind(child, &["attribute_value", "quoted_attribute_value", "directive_value"]) - { - return sanitize_identifier(&unquote_text(node_text( - source, - value.start_byte(), - value.end_byte(), - ))); - } - return Some(name.to_string()); - } - None -} - -fn resolve_embedded_language( - node: Node<'_>, - source: &str, - default_language: SupportLang, -) -> Option { - if let Some(language) = attribute_value(node, "lang", source) { - return SupportLang::from_alias(language.as_str()); - } - Some(default_language) -} - -fn extract_attribute_name(node: Node<'_>, source: &str) -> Option { - child_by_kind(node, &["attribute_name", "directive_name"]) - .and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))) - .or_else(|| { - if node_text(source, node.start_byte(), node.end_byte()) - .trim() - .starts_with('#') - { - extract_directive_name(node, source) - } else { - None - } - }) -} - -fn start_like(node: Node<'_>) -> Option> { - child_by_kind(node, &["start_tag", "self_closing_tag"]) -} diff --git a/crates/pi-natives/src/chunk/atom_list.rs b/crates/pi-natives/src/chunk/atom_list.rs deleted file mode 100644 index 867abf954..000000000 --- a/crates/pi-natives/src/chunk/atom_list.rs +++ /dev/null @@ -1,145 +0,0 @@ -use std::collections::{HashMap, HashSet}; - -use super::schema; - -type AtomSet = HashSet<&'static str>; - -static ATOM_NODES: std::sync::LazyLock> = - std::sync::LazyLock::new(|| { - HashMap::from([ - ("astro", HashSet::from(["frontmatter"])), - ("bash", HashSet::from(["string", "raw_string", "heredoc_body", "simple_expansion"])), - ("c", HashSet::from(["string_literal", "char_literal"])), - ("clojure", HashSet::from(["kwd_lit", "regex_lit"])), - ("cmake", HashSet::from(["argument"])), - ("cpp", HashSet::from(["string_literal", "char_literal"])), - ( - "csharp", - HashSet::from([ - "string_literal", - "verbatim_string_literal", - "character_literal", - "modifier", - ]), - ), - ("css", HashSet::from(["integer_value", "float_value", "color_value", "string_value"])), - ("elixir", HashSet::from(["string_constant_expr"])), - ("go", HashSet::from(["interpreted_string_literal", "raw_string_literal"])), - ( - "haskell", - HashSet::from([ - "qualified_variable", - "qualified_module", - "qualified_constructor", - "strict_type", - ]), - ), - ("hcl", HashSet::from(["string_lit", "heredoc_template"])), - ( - "html", - HashSet::from(["doctype", "quoted_attribute_value", "raw_text", "tag_name", "text"]), - ), - ( - "java", - HashSet::from([ - "string_literal", - "boolean_type", - "integral_type", - "floating_point_type", - "void_type", - ]), - ), - ("json", HashSet::from(["string"])), - ( - "julia", - HashSet::from([ - "string_literal", - "prefixed_string_literal", - "command_literal", - "character_literal", - ]), - ), - ( - "kotlin", - HashSet::from([ - "nullable_type", - "string_literal", - "line_string_literal", - "character_literal", - ]), - ), - ("lua", HashSet::from(["string"])), - ("make", HashSet::from(["shell_text", "text"])), - ("nix", HashSet::from(["string_expression", "indented_string_expression"])), - ("objc", HashSet::from(["string_literal"])), - ( - "perl", - HashSet::from([ - "string_single_quoted", - "string_double_quoted", - "comments", - "command_qx_quoted", - "pattern_matcher_m", - "regex_pattern_qr", - "transliteration_tr_or_y", - "substitution_pattern_s", - "scalar_variable", - "array_variable", - "hash_variable", - "hash_access_variable", - ]), - ), - ("php", HashSet::from(["string", "encapsed_string"])), - ("protobuf", HashSet::from(["string"])), - ("python", HashSet::from(["string"])), - ("r", HashSet::from(["string", "special"])), - ("ruby", HashSet::from(["string", "heredoc_body", "regex"])), - ("rust", HashSet::from(["char_literal", "string_literal", "raw_string_literal"])), - ("scala", HashSet::from(["string", "template_string", "interpolated_string_expression"])), - ("solidity", HashSet::from(["string", "hex_string_literal", "unicode_string_literal"])), - ("sql", HashSet::from(["string", "identifier"])), - ("swift", HashSet::from(["line_string_literal"])), - ("toml", HashSet::from(["string", "quoted_key"])), - ("tsx", HashSet::from(["string", "template_string"])), - ("typescript", HashSet::from(["string", "template_string", "regex", "predefined_type"])), - ("xml", HashSet::from(["AttValue", "XMLDecl"])), - ( - "yaml", - HashSet::from([ - "string_scalar", - "double_quote_scalar", - "single_quote_scalar", - "block_scalar", - ]), - ), - ("verilog", HashSet::from(["integral_number"])), - ("zig", HashSet::from(["string"])), - ]) - }); - -pub fn is_atom_node(language: &str, kind: &str) -> bool { - ATOM_NODES - .get(language) - .is_some_and(|atom_nodes| atom_nodes.contains(kind)) -} - -pub fn is_atom_node_current(kind: &str) -> bool { - schema::current_language().is_some_and(|language| is_atom_node(language, kind)) -} - -#[cfg(test)] -mod tests { - use super::is_atom_node; - - #[test] - fn nix_binding_set_is_not_an_atom() { - assert!(!is_atom_node("nix", "binding_set")); - assert!(is_atom_node("nix", "string_expression")); - } - - #[test] - fn typescript_predefined_types_stay_atomic() { - assert!(is_atom_node("typescript", "predefined_type")); - assert!(!is_atom_node("typescript", "class_declaration")); - } -} diff --git a/crates/pi-natives/src/chunk/classify.rs b/crates/pi-natives/src/chunk/classify.rs deleted file mode 100644 index 3f0a72cdc..000000000 --- a/crates/pi-natives/src/chunk/classify.rs +++ /dev/null @@ -1,509 +0,0 @@ -//! Per-language chunk classification trait. -//! -//! Languages now provide semantic tables plus a narrow override hook for -//! genuinely custom behavior. - -use tree_sitter::Node; - -use super::{ - common::{ - ChunkContext, NameStyle, RawChunkCandidate, extract_identifier, is_absorbable_attribute, - is_trivia_node, make_candidate, named_children, recurse_self, resolve_recurse, - resolve_value_container, sanitize_node_kind, signature_for_node, - try_promote_call_with_callback, - }, - defaults, - kind::ChunkKind, - schema, -}; -use crate::chunk::types::ChunkNode; - -#[derive(Clone, Copy, Debug)] -pub enum RuleStyle { - Named, - Group, - Positional, -} - -#[derive(Clone, Copy, Debug)] -pub enum NamingMode { - AutoIdentifier, - None, - SanitizedKind, -} - -#[derive(Clone, Copy, Debug)] -pub enum RecurseMode { - None, - Auto(ChunkContext), - SelfNode(ChunkContext), - ValueContainer, -} - -#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] -pub enum WrapperSignature { - #[default] - Child, - Wrapper, -} - -#[derive(Clone, Copy, Debug, Default)] -pub struct WrapperTransform { - pub kind: Option, - pub name_style: Option, - pub clear_identifier: bool, - pub signature: WrapperSignature, -} - -#[derive(Clone, Copy, Debug)] -pub struct SemanticRule { - pub ts_kind: &'static str, - pub chunk_kind: ChunkKind, - pub style: RuleStyle, - pub naming: NamingMode, - pub recurse: RecurseMode, -} - -pub const fn semantic_rule( - ts_kind: &'static str, - chunk_kind: ChunkKind, - style: RuleStyle, - naming: NamingMode, - recurse: RecurseMode, -) -> SemanticRule { - SemanticRule { ts_kind, chunk_kind, style, naming, recurse } -} - -#[derive(Clone, Copy, Debug)] -pub struct StructuralOverrides { - pub extra_trivia: &'static [&'static str], - pub preserved_trivia: &'static [&'static str], - pub extra_root_wrappers: &'static [&'static str], - pub preserved_root_wrappers: &'static [&'static str], - pub absorbable_attrs: &'static [&'static str], -} - -impl StructuralOverrides { - pub const EMPTY: Self = Self { - extra_trivia: &[], - preserved_trivia: &[], - extra_root_wrappers: &[], - preserved_root_wrappers: &[], - absorbable_attrs: &[], - }; - - pub fn is_extra_trivia(&self, kind: &str) -> bool { - self.extra_trivia.contains(&kind) - } - - pub fn preserves_trivia(&self, kind: &str) -> bool { - self.preserved_trivia.contains(&kind) - } - - pub fn is_extra_root_wrapper(&self, kind: &str) -> bool { - self.extra_root_wrappers.contains(&kind) - } - - pub fn preserves_root_wrapper(&self, kind: &str) -> bool { - self.preserved_root_wrappers.contains(&kind) - } - - pub fn is_absorbable_attr(&self, kind: &str) -> bool { - self.absorbable_attrs.contains(&kind) - } -} - -#[derive(Clone, Copy, Debug)] -pub struct ClassifierTables { - pub root: &'static [SemanticRule], - pub class: &'static [SemanticRule], - pub function: &'static [SemanticRule], - pub structural_overrides: StructuralOverrides, -} - -pub const EMPTY_CLASSIFIER_TABLES: ClassifierTables = ClassifierTables { - root: &[], - class: &[], - function: &[], - structural_overrides: StructuralOverrides::EMPTY, -}; - -pub trait LangClassifier { - fn tables(&self) -> &'static ClassifierTables { - &EMPTY_CLASSIFIER_TABLES - } - - fn classify_root<'t>(&self, _node: Node<'t>, _source: &str) -> Option> { - None - } - - fn classify_class<'t>(&self, _node: Node<'t>, _source: &str) -> Option> { - None - } - - fn classify_function<'t>( - &self, - _node: Node<'t>, - _source: &str, - ) -> Option> { - None - } - - fn is_root_wrapper(&self, _kind: &str) -> bool { - false - } - - fn preserve_root_wrapper(&self, _kind: &str) -> bool { - false - } - - fn preserve_trivia(&self, _kind: &str) -> bool { - false - } - - fn is_trivia(&self, _kind: &str) -> bool { - false - } - - fn is_absorbable_attr(&self, _kind: &str) -> bool { - false - } - - /// Return true to drop a named child from `collect_children_for_context` - /// without turning it into a chunk and without absorbing its byte range - /// into the next sibling via `attach_leading_trivia`. - /// - /// Use this for structural framing nodes (JSX opening/closing elements, - /// framework fragment markers, etc.) that have no meaningful chunk of - /// their own and should NOT extend the following chunk's span backward. - /// Prefer `is_trivia` for comment-like nodes that should be absorbed as - /// leading context of the next chunk. - fn should_skip_child(&self, _kind: &str) -> bool { - false - } - - fn classify_override<'t>( - &self, - _context: ChunkContext, - _node: Node<'t>, - _source: &str, - ) -> Option> { - None - } - - fn preserve_children( - &self, - _parent: &RawChunkCandidate<'_>, - _children: &[RawChunkCandidate<'_>], - ) -> bool { - false - } - - fn post_process( - &self, - _chunks: &mut Vec, - _root_children: &mut Vec, - _source: &str, - ) { - } -} - -pub fn structural_overrides(classifier: &dyn LangClassifier) -> StructuralOverrides { - classifier.tables().structural_overrides -} - -pub fn classify_with_tables<'tree>( - classifier: &dyn LangClassifier, - context: ChunkContext, - node: Node<'tree>, - source: &str, -) -> Option> { - if let Some(candidate) = classifier.classify_override(context, node, source) { - return Some(candidate); - } - - find_rule(classifier.tables(), context, node.kind()) - .map(|rule| build_candidate_from_rule(node, source, *rule)) - .or_else(|| match context { - ChunkContext::Root => classifier.classify_root(node, source), - ChunkContext::ClassBody => classifier.classify_class(node, source), - ChunkContext::FunctionBody => classifier.classify_function(node, source), - }) -} - -pub fn classify_with_defaults<'tree>( - classifier: &dyn LangClassifier, - context: ChunkContext, - node: Node<'tree>, - source: &str, -) -> RawChunkCandidate<'tree> { - if node.is_error() || node.kind() == "ERROR" { - return make_candidate(node, ChunkKind::Error, None, NameStyle::Error, None, None, source); - } - - let candidate = match context { - ChunkContext::Root => classify_with_tables(classifier, context, node, source) - .unwrap_or_else(|| defaults::classify_root_default(node, source)), - ChunkContext::ClassBody => classify_with_tables(classifier, context, node, source) - .unwrap_or_else(|| defaults::classify_class_default(node, source)), - ChunkContext::FunctionBody => classify_with_tables(classifier, context, node, source) - .unwrap_or_else(|| defaults::classify_function_default(node, source)), - }; - - // If the classifier produced a groupable leaf (no recurse), try to - // promote call-with-trailing-callback patterns into named container - // chunks. This handles `describe(...)`, `t.Run(...)`, etc. across - // all languages without per-language opt-in. - if candidate.recurse.is_none() - && candidate.groupable - && let Some(promoted) = try_promote_call_with_callback(node, source) - { - return promoted; - } - - candidate -} - -pub fn first_wrapper_content_child<'tree>( - classifier: &dyn LangClassifier, - node: Node<'tree>, -) -> Option> { - if let Some(child) = schema_wrapper_child(node) { - return Some(child); - } - - let overrides = structural_overrides(classifier); - named_children(node) - .into_iter() - .find(|child| !is_wrapper_metadata_child(*child, classifier, overrides)) -} - -pub fn promote_wrapper_candidate<'tree>( - classifier: &dyn LangClassifier, - context: ChunkContext, - node: Node<'tree>, - source: &str, - transform: WrapperTransform, -) -> Option> { - let (child, candidate) = promotable_wrapper_child(classifier, context, node, source)?; - let signature_node = match transform.signature { - WrapperSignature::Child => child, - WrapperSignature::Wrapper => node, - }; - let kind = transform.kind.unwrap_or(candidate.kind); - let name_style = transform.name_style.unwrap_or(candidate.name_style); - let identifier = if transform.clear_identifier { - None - } else { - candidate.identifier - }; - - Some(make_candidate( - node, - kind, - identifier, - name_style, - signature_for_node(signature_node, source), - candidate.recurse, - source, - )) -} - -pub fn build_candidate_from_rule<'tree>( - node: Node<'tree>, - source: &str, - rule: SemanticRule, -) -> RawChunkCandidate<'tree> { - let identifier = match rule.naming { - NamingMode::AutoIdentifier => extract_identifier(node, source), - NamingMode::None => None, - NamingMode::SanitizedKind => Some(sanitize_node_kind(node.kind()).to_string()), - }; - - let recurse = match rule.recurse { - RecurseMode::None => None, - RecurseMode::Auto(context) => resolve_recurse(node, context), - RecurseMode::SelfNode(context) => Some(recurse_self(node, context)), - RecurseMode::ValueContainer => resolve_value_container(node), - }; - - match rule.style { - RuleStyle::Named => make_candidate( - node, - rule.chunk_kind, - identifier, - NameStyle::Named, - signature_for_node(node, source), - recurse, - source, - ), - RuleStyle::Group => { - make_candidate(node, rule.chunk_kind, identifier, NameStyle::Group, None, recurse, source) - }, - RuleStyle::Positional => make_candidate( - node, - rule.chunk_kind, - None::, - NameStyle::Named, - None, - recurse, - source, - ), - } -} - -fn find_rule( - tables: &ClassifierTables, - context: ChunkContext, - kind: &str, -) -> Option<&'static SemanticRule> { - let rules = match context { - ChunkContext::Root => tables.root, - ChunkContext::ClassBody => tables.class, - ChunkContext::FunctionBody => tables.function, - }; - - rules.iter().find(|rule| rule.ts_kind == kind) -} - -fn promotable_wrapper_child<'tree>( - classifier: &dyn LangClassifier, - context: ChunkContext, - node: Node<'tree>, - source: &str, -) -> Option<(Node<'tree>, RawChunkCandidate<'tree>)> { - if let Some(child) = schema_wrapper_child(node) { - let candidate = classify_with_defaults(classifier, context, child, source); - if is_promotable_wrapper_candidate(child, &candidate) { - return Some((child, candidate)); - } - } - - let overrides = structural_overrides(classifier); - let mut promoted = named_children(node).into_iter().filter_map(|child| { - if is_wrapper_metadata_child(child, classifier, overrides) { - return None; - } - - let candidate = classify_with_defaults(classifier, context, child, source); - is_promotable_wrapper_candidate(child, &candidate).then_some((child, candidate)) - }); - - let promoted_child = promoted.next()?; - if promoted.next().is_some() { - return None; - } - Some(promoted_child) -} - -fn is_wrapper_metadata_child( - node: Node<'_>, - classifier: &dyn LangClassifier, - overrides: StructuralOverrides, -) -> bool { - let kind = node.kind(); - ((is_trivia_node(node) || classifier.is_trivia(kind)) - && !overrides.preserves_trivia(kind) - && !classifier.preserve_trivia(kind)) - || (overrides.is_extra_trivia(kind) - && !overrides.preserves_trivia(kind) - && !classifier.preserve_trivia(kind)) - || is_absorbable_attribute(kind) - || overrides.is_absorbable_attr(kind) - || classifier.is_absorbable_attr(kind) -} - -fn is_promotable_wrapper_candidate(node: Node<'_>, candidate: &RawChunkCandidate<'_>) -> bool { - if matches!(candidate.kind, ChunkKind::Error | ChunkKind::Chunk | ChunkKind::Statements) { - return false; - } - - candidate.identifier.is_some() - || candidate.recurse.is_some() - || candidate.kind.traits().container - || node.kind().ends_with("_definition") - || node.kind().ends_with("_declaration") -} - -fn schema_wrapper_child(node: Node<'_>) -> Option> { - let schema = schema::schema_for_current(node.kind())?; - for field in &schema.promotion_fields { - if let Some(child) = node.child_by_field_name(field) { - return Some(child); - } - } - None -} - -/// Resolve a [`LangClassifier`] for the given language. -pub fn classifier_for(lang: &str) -> &'static dyn LangClassifier { - match lang { - "astro" => &super::ast_astro::AstroClassifier, - // JS / TS family - "javascript" | "js" | "jsx" | "typescript" | "ts" | "tsx" => { - &super::ast_js_ts::JsTsClassifier - }, - // Python / Starlark - "python" | "starlark" => &super::ast_python::PythonClassifier, - // Rust - "rust" => &super::ast_rust::RustClassifier, - // Go - "go" | "golang" => &super::ast_go::GoClassifier, - // C / C++ / Objective-C - "c" | "cpp" | "c++" | "objc" | "objective-c" => &super::ast_c_cpp_objc::CCppClassifier, - // C# / Java - "csharp" | "java" => &super::ast_csharp_java::CSharpJavaClassifier, - // Clojure - "clojure" => &super::ast_clojure::ClojureClassifier, - // CMake - "cmake" => &super::ast_cmake::CMakeClassifier, - // CSS - "css" => &super::ast_css::CssClassifier, - // Data formats - "json" | "toml" | "yaml" => &super::ast_data_formats::DataFormatsClassifier, - // Dockerfile - "dockerfile" => &super::ast_dockerfile::DockerfileClassifier, - // Elixir - "elixir" => &super::ast_elixir::ElixirClassifier, - // Erlang - "erlang" => &super::ast_erlang::ErlangClassifier, - // GraphQL - "graphql" => &super::ast_graphql::GraphqlClassifier, - // Haskell / Scala - "haskell" | "scala" => &super::ast_haskell_scala::HaskellScalaClassifier, - // HTML / XML - "html" | "xml" => &super::ast_html_xml::HtmlXmlClassifier, - // INI - "ini" => &super::ast_ini::IniClassifier, - // Just - "just" => &super::ast_just::JustClassifier, - // Markdown / Handlebars - "markdown" | "handlebars" => &super::ast_markup::MarkupClassifier, - // Nix / HCL - "nix" | "hcl" => &super::ast_nix_hcl::NixHclClassifier, - // OCaml - "ocaml" => &super::ast_ocaml::OcamlClassifier, - // Perl - "perl" => &super::ast_perl::PerlClassifier, - // PowerShell - "powershell" => &super::ast_powershell::PowershellClassifier, - // Protobuf - "protobuf" | "proto" => &super::ast_proto::ProtoClassifier, - // R - "r" => &super::ast_r::RClassifier, - // Ruby / Lua - "ruby" | "lua" => &super::ast_ruby_lua::RubyLuaClassifier, - // SQL - "sql" => &super::ast_sql::SqlClassifier, - // Svelte - "svelte" => &super::ast_svelte::SvelteClassifier, - // TLA+ / PlusCal - "tlaplus" | "pluscal" | "pcal" | "tla" | "tla+" => &super::ast_tlaplus::TlaplusClassifier, - // Bash / Make / Diff - "bash" | "make" | "diff" => &super::ast_bash_make_diff::ShellBuildClassifier, - // Vue - "vue" => &super::ast_vue::VueClassifier, - // Everything else (Kotlin, Swift, PHP, Solidity, etc.) - _ => &super::ast_misc::MiscClassifier, - } -} diff --git a/crates/pi-natives/src/chunk/common.rs b/crates/pi-natives/src/chunk/common.rs deleted file mode 100644 index 515261b51..000000000 --- a/crates/pi-natives/src/chunk/common.rs +++ /dev/null @@ -1,903 +0,0 @@ -//! Shared helpers for chunk classification. -//! -//! These are the building blocks that per-language classifiers use to construct -//! [`RawChunkCandidate`] values. They are also used by the default (shared) -//! classification in [`super::defaults`]. - -use tree_sitter::Node; - -use super::{ - kind::{ChunkKind, SummaryStyle}, - shape, - types::ChunkNode, -}; -use crate::{env_uint, language::SupportLang}; - -// ── Configuration (environment overrides) ──────────────────────────────── -env_uint! { - // Configured leaf threshold. - pub static LEAF_THRESHOLD: usize = "PI_CHUNK_LEAF_THRESHOLD" or 8 => [1, usize::MAX]; - // Configured max chunk lines. - pub static MAX_CHUNK_LINES: usize = "PI_CHUNK_MAX_LINES" or 25 => [1, usize::MAX]; - // Configured min recurse savings. - pub static MIN_RECURSE_SAVINGS: usize = "PI_CHUNK_MIN_SAVINGS" or 4 => [1, usize::MAX]; -} - -// ── Internal types ─────────────────────────────────────────────────────── - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum ChunkContext { - Root, - ClassBody, - FunctionBody, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum NameStyle { - Named, - Group, - Error, -} - -#[derive(Clone, Copy, Debug)] -pub struct RecurseSpec<'tree> { - pub node: Node<'tree>, - pub context: ChunkContext, -} - -#[derive(Clone, Copy, Debug)] -pub struct InjectedChunkSpec<'tree> { - pub language: SupportLang, - pub content_node: Node<'tree>, -} - -#[derive(Clone, Debug)] -pub struct RawChunkCandidate<'tree> { - pub identifier: Option, - pub kind: ChunkKind, - pub name_style: NameStyle, - pub range_start_byte: usize, - pub range_end_byte: usize, - /// Start byte for `chunk_checksum`; stays at the primary node's start while - /// `range_start_byte` may be extended backward to include leading - /// attributes/comments. - pub checksum_start_byte: usize, - pub range_start_line: usize, - pub range_end_line: usize, - pub signature: Option, - pub error: bool, - pub groupable: bool, - pub has_leading_comment: bool, - pub force_recurse: bool, - pub region_node: Option>, - pub injected: Option>, - pub recurse: Option>, -} - -#[derive(Default)] -pub struct ChunkAccumulator { - pub chunks: Vec, -} - -// ── Candidate constructors ─────────────────────────────────────────────── - -/// Convert a tree-sitter `end_position` into a 1-indexed line number. -/// -/// Tree-sitter byte ranges are half-open, so `end_position` points to the byte -/// immediately *after* the last byte of the node. When that byte lands at -/// column 0 of a new row, the node's last byte actually sits on the previous -/// row and the 1-indexed last line is exactly `end.row` — not `end.row + 1`. -/// This matters for grammars whose container nodes terminate on the start of -/// the next sibling (tree-sitter-markdown sections, tree-sitter-toml tables, -/// etc.): the naive `end.row + 1` would claim the sibling's heading line and -/// make `replace_range_by_lines` clobber it. -const fn end_row_as_line(start: tree_sitter::Point, end: tree_sitter::Point) -> usize { - if end.column == 0 && end.row > start.row { - end.row - } else { - end.row + 1 - } -} - -pub fn make_candidate<'tree>( - node: Node<'tree>, - kind: ChunkKind, - identifier: impl Into>, - name_style: NameStyle, - signature: Option, - recurse: Option>, - source: &str, -) -> RawChunkCandidate<'tree> { - let identifier = identifier.into(); - let start = node.start_position(); - let end = node.end_position(); - let summary = summary_for_node(node, kind, identifier.as_deref(), signature.as_deref(), source); - let start_byte = node.start_byte(); - RawChunkCandidate { - identifier, - kind, - name_style, - range_start_byte: start_byte, - range_end_byte: node.end_byte(), - checksum_start_byte: start_byte, - range_start_line: start.row + 1, - range_end_line: end_row_as_line(start, end), - signature: summary, - error: kind == ChunkKind::Error, - groupable: kind.traits().groupable, - has_leading_comment: false, - force_recurse: kind.traits().container || (recurse.is_some() && node.has_error()), - region_node: recurse.map(|spec| spec.node), - injected: None, - recurse, - } -} - -pub fn group_candidate<'tree>( - node: Node<'tree>, - kind: ChunkKind, - source: &str, -) -> RawChunkCandidate<'tree> { - make_candidate(node, kind, None, NameStyle::Group, None, None, source) -} - -pub fn positional_candidate<'tree>( - node: Node<'tree>, - kind: ChunkKind, - source: &str, -) -> RawChunkCandidate<'tree> { - make_candidate(node, kind, None, NameStyle::Named, None, None, source) -} - -pub fn named_candidate<'tree>( - node: Node<'tree>, - kind: ChunkKind, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'tree> { - make_kind_chunk(node, kind, extract_identifier(node, source), source, recurse) -} - -pub fn container_candidate<'tree>( - node: Node<'tree>, - kind: ChunkKind, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'tree> { - make_kind_chunk(node, kind, extract_identifier(node, source), source, recurse) -} - -pub fn make_kind_chunk<'tree>( - node: Node<'tree>, - kind: ChunkKind, - identifier: Option, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'tree> { - make_candidate( - node, - kind, - identifier, - NameStyle::Named, - signature_for_node(node, source), - recurse, - source, - ) -} - -pub fn make_kind_chunk_from<'tree>( - range_node: Node<'tree>, - signature_node: Node<'tree>, - kind: ChunkKind, - identifier: Option, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'tree> { - make_candidate( - range_node, - kind, - identifier, - NameStyle::Named, - signature_for_node(signature_node, source), - recurse, - source, - ) -} - -pub fn make_container_chunk<'tree>( - node: Node<'tree>, - kind: ChunkKind, - identifier: Option, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'tree> { - make_candidate( - node, - kind, - identifier, - NameStyle::Named, - signature_for_node(node, source), - recurse, - source, - ) -} - -pub fn make_container_chunk_from<'tree>( - range_node: Node<'tree>, - signature_node: Node<'tree>, - kind: ChunkKind, - identifier: Option, - source: &str, - recurse: Option>, -) -> RawChunkCandidate<'tree> { - make_candidate( - range_node, - kind, - identifier, - NameStyle::Named, - signature_for_node(signature_node, source), - recurse, - source, - ) -} - -/// Derive a "`prefix_identifier`" name from a node. -pub fn prefixed_name(prefix: &str, node: Node<'_>, source: &str) -> String { - let identifier = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string()); - format!("{prefix}_{identifier}") -} - -pub const fn with_region_node<'tree>( - mut candidate: RawChunkCandidate<'tree>, - region_node: Option>, -) -> RawChunkCandidate<'tree> { - candidate.region_node = region_node; - candidate -} - -pub const fn with_injected_subtree<'tree>( - mut candidate: RawChunkCandidate<'tree>, - language: SupportLang, - content_node: Node<'tree>, -) -> RawChunkCandidate<'tree> { - candidate.injected = Some(InjectedChunkSpec { language, content_node }); - candidate -} - -pub const fn embedded_selector_token(language: SupportLang) -> &'static str { - match language { - SupportLang::Bash => "bash", - SupportLang::C => "c", - SupportLang::Cmake => "cmake", - SupportLang::Cpp => "cpp", - SupportLang::CSharp => "cs", - SupportLang::Css => "css", - SupportLang::Go => "go", - SupportLang::Html => "html", - SupportLang::Java => "java", - SupportLang::JavaScript => "js", - SupportLang::Json => "json", - SupportLang::Kotlin => "kt", - SupportLang::Lua => "lua", - SupportLang::Markdown => "md", - SupportLang::Php => "php", - SupportLang::Python => "py", - SupportLang::Ruby => "rb", - SupportLang::Rust => "rust", - SupportLang::Scala => "scala", - SupportLang::Sql => "sql", - SupportLang::Swift => "swift", - SupportLang::Toml => "toml", - SupportLang::Tsx => "tsx", - SupportLang::TypeScript => "ts", - SupportLang::Yaml => "yaml", - _ => language.canonical_name(), - } -} - -// ── Inferred / catch-all candidates ────────────────────────────────────── - -/// Derive a semantic name from a node's kind and/or identifier. -pub fn infer_named_candidate<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> { - let kind_name = sanitize_node_kind(node.kind()); - let kind = ChunkKind::from_sanitized_kind(kind_name); - auto_classify(node, kind, source) -} - -pub fn auto_classify<'tree>( - node: Node<'tree>, - kind: ChunkKind, - source: &str, -) -> RawChunkCandidate<'tree> { - make_kind_chunk( - node, - kind, - extract_identifier(node, source), - source, - auto_recurse_for_kind(node, kind), - ) -} - -// ── Tree navigation helpers ────────────────────────────────────────────── - -pub fn named_children(node: Node<'_>) -> Vec> { - let mut children = Vec::new(); - for index in 0..node.child_count() { - if let Some(child) = node.child(index) - && (child.is_named() || child.is_error() || child.kind() == "ERROR") - { - children.push(child); - } - } - children -} - -pub fn child_by_kind<'tree>(node: Node<'tree>, kinds: &[&str]) -> Option> { - named_children(node) - .into_iter() - .find(|child| kinds.iter().any(|kind| child.kind() == *kind)) -} - -pub fn child_by_field_or_kind<'tree>( - node: Node<'tree>, - fields: &[&str], - kinds: &[&str], -) -> Option> { - for field in fields { - if let Some(child) = node.child_by_field_name(field) { - return Some(child); - } - } - child_by_kind(node, kinds) -} - -pub fn resolve_recurse(node: Node<'_>, context: ChunkContext) -> Option> { - shape::recurse_target(node).map(|child| RecurseSpec { node: child, context }) -} - -pub fn resolve_value_container(node: Node<'_>) -> Option> { - shape::value_container_target(node) - .map(|child| RecurseSpec { node: child, context: ChunkContext::ClassBody }) -} - -fn auto_recurse_for_kind(node: Node<'_>, kind: ChunkKind) -> Option> { - let context = match kind { - ChunkKind::Constructor - | ChunkKind::Function - | ChunkKind::Macro - | ChunkKind::Method - | ChunkKind::Proc - | ChunkKind::Recipe => ChunkContext::FunctionBody, - _ if kind.traits().container => ChunkContext::ClassBody, - _ => return None, - }; - - resolve_recurse(node, context) -} - -pub fn compute_body_inner_boundaries( - source: &str, - body_start: usize, - body_end: usize, -) -> (usize, usize) { - let bounded_start = body_start.min(source.len()); - let bounded_end = body_end.min(source.len()).max(bounded_start); - let slice = &source[bounded_start..bounded_end]; - - let Some((first_non_ws_rel, first_non_ws)) = slice - .char_indices() - .find(|(_, ch)| !matches!(ch, ' ' | '\t' | '\n' | '\r')) - else { - return (bounded_start, bounded_end); - }; - let Some((last_non_ws_rel, last_non_ws)) = slice - .char_indices() - .rev() - .find(|(_, ch)| !matches!(ch, ' ' | '\t' | '\n' | '\r')) - else { - return (bounded_start, bounded_end); - }; - - let has_delimiters = matches!((first_non_ws, last_non_ws), ('{', '}') | ('(', ')') | ('[', ']')); - if !has_delimiters { - let line_start = source[..bounded_start].rfind('\n').map_or(0, |pos| pos + 1); - let leading_indent = &source[line_start..bounded_start]; - if !leading_indent.is_empty() && leading_indent.chars().all(|ch| matches!(ch, ' ' | '\t')) { - let mut inner_end = bounded_end; - let trailing = &source[bounded_end..]; - if let Some(rel_newline) = trailing.find('\n') { - if trailing[..rel_newline] - .chars() - .all(|ch| matches!(ch, ' ' | '\t' | '\r')) - { - inner_end = bounded_end + rel_newline + 1; - } - } else if trailing.chars().all(|ch| matches!(ch, ' ' | '\t' | '\r')) { - inner_end = source.len(); - } - return (line_start, inner_end); - } - return (bounded_start, bounded_end); - } - - let mut inner_start = bounded_start + first_non_ws_rel + first_non_ws.len_utf8(); - if source[inner_start..].starts_with("\r\n") { - inner_start += 2; - } else if source[inner_start..].starts_with('\n') { - inner_start += 1; - } - - // Epilogue starts at the beginning of the line containing the closing - // delimiter so that the closing line's indentation is part of the epilogue, - // not the body. - let close_abs = bounded_start + last_non_ws_rel; - let inner_end = source[..close_abs] - .rfind('\n') - .map_or(close_abs, |nl| nl + 1); - (inner_start.min(bounded_end), inner_end.max(inner_start).min(bounded_end)) -} - -// ── Recurse helpers ────────────────────────────────────────────────────── - -pub fn recurse_into<'tree>( - node: Node<'tree>, - context: ChunkContext, - fields: &[&str], - kinds: &[&str], -) -> Option> { - child_by_field_or_kind(node, fields, kinds).map(|child| RecurseSpec { node: child, context }) -} - -pub const fn recurse_self(node: Node<'_>, context: ChunkContext) -> RecurseSpec<'_> { - RecurseSpec { node, context } -} - -pub fn recurse_body(node: Node<'_>, context: ChunkContext) -> Option> { - resolve_recurse(node, context) -} - -pub fn recurse_class(node: Node<'_>) -> Option> { - resolve_recurse(node, ChunkContext::ClassBody) -} - -pub fn recurse_enum(node: Node<'_>) -> Option> { - resolve_recurse(node, ChunkContext::ClassBody) -} - -pub fn recurse_value_container(node: Node<'_>) -> Option> { - resolve_value_container(node) -} - -/// Try to promote a node that wraps a call expression with a trailing -/// callback/block argument. Returns a named chunk candidate with `recurse` -/// pointing into the callback body. -/// -/// This is language-agnostic: it uses structural shape detection to find -/// call-with-callback patterns in any language (JS `describe(...)`, Go -/// `t.Run(...)`, Rust `tokio::spawn(async { ... })`, etc.). -pub fn try_promote_call_with_callback<'tree>( - node: Node<'tree>, - source: &str, -) -> Option> { - let (func_node, body) = shape::trailing_callback_body(node)?; - - // Extract a name from the call target (e.g. `describe`, `describe.serial`, - // `app.use`). Sanitize the raw source text (dots become underscores). - let name = sanitize_identifier(node_text(source, func_node.start_byte(), func_node.end_byte())); - - let recurse = Some(RecurseSpec { node: body, context: ChunkContext::FunctionBody }); - - Some(make_kind_chunk(node, ChunkKind::Expression, name, source, recurse)) -} - -// ── Identifier extraction ──────────────────────────────────────────────── - -pub fn extract_identifier(node: Node<'_>, source: &str) -> Option { - if node.kind() == "constructor" { - return Some("constructor".to_string()); - } - - if let Some(name_node) = shape::identifier_node(node) { - return sanitize_identifier(node_text(source, name_node.start_byte(), name_node.end_byte())); - } - - None -} - -/// Extract the name of a single-declarator binding like `const FOO = ...`. -pub fn extract_single_declarator_name(node: Node<'_>, source: &str) -> Option { - let declarators: Vec> = named_children(node) - .into_iter() - .filter(|c| c.kind() == "variable_declarator") - .collect(); - if declarators.len() != 1 { - return None; - } - extract_identifier(declarators[0], source) -} - -// ── Text helpers ───────────────────────────────────────────────────────── - -pub fn node_text(source: &str, start_byte: usize, end_byte: usize) -> &str { - source.get(start_byte..end_byte).unwrap_or("") -} - -pub fn sanitize_identifier(text: &str) -> Option { - let mut out = String::new(); - let mut previous_was_underscore = false; - - for ch in text.chars() { - if ch.is_alphanumeric() || ch == '_' || ch == '$' { - out.push(ch); - previous_was_underscore = false; - continue; - } - - if !previous_was_underscore { - out.push('_'); - previous_was_underscore = true; - } - } - - let sanitized = out.trim_matches('_').to_string(); - if sanitized.is_empty() { - None - } else { - Some(sanitized) - } -} - -pub fn unquote_text(text: &str) -> String { - text.trim().trim_matches('"').trim_matches('\'').to_string() -} - -pub fn sanitize_node_kind(kind: &str) -> &str { - let mut kind_stripped = kind; - for suffix in [ - "_instruction", - "_statement", - "_declaration", - "_definition", - "_item", - "ession", // _expression -> _expr - ] { - if let Some(stripped) = kind_stripped.strip_suffix(suffix) { - kind_stripped = stripped; - } - } - if kind_stripped.is_empty() { - kind - } else { - kind_stripped - } -} - -pub fn normalized_header(source: &str, start_byte: usize, end_byte: usize) -> String { - let slice = node_text(source, start_byte, end_byte); - let mut header = String::new(); - - for line in slice.lines().take(4) { - let trimmed = line.trim(); - if trimmed.is_empty() { - continue; - } - if !header.is_empty() { - header.push(' '); - } - header.push_str(trimmed); - if trimmed.contains('{') || trimmed.ends_with(';') { - break; - } - } - - collapse_whitespace(header.as_str()) -} - -pub fn collapse_whitespace(text: &str) -> String { - let mut out = String::new(); - let mut pending_space = false; - for ch in text.chars() { - if ch.is_whitespace() { - pending_space = true; - continue; - } - if pending_space && !out.is_empty() { - out.push(' '); - } - out.push(ch); - pending_space = false; - } - out -} - -// ── Signature helpers ──────────────────────────────────────────────────── - -pub fn signature_for_node(node: Node<'_>, source: &str) -> Option { - let raw = if let Some(end_byte) = shape::signature_end_byte(node) { - node_text(source, node.start_byte(), end_byte) - } else { - node_text(source, node.start_byte(), node.end_byte()) - }; - - let sig = collapse_whitespace(raw.trim()); - let sig = sig - .trim_end_matches('{') - .trim_end_matches(':') - .trim_end_matches(';') - .trim(); - if sig.is_empty() { - None - } else { - Some(sig.to_string()) - } -} - -// ── Summary / canonical naming ─────────────────────────────────────────── - -fn normalize_summary_text(summary: &str) -> Option { - let summary = collapse_whitespace(summary.trim()) - .trim_end_matches('{') - .trim_end_matches(':') - .trim_end_matches(';') - .trim() - .to_string(); - if summary.is_empty() { - None - } else { - Some(summary) - } -} - -fn summarize_function_node( - kind: ChunkKind, - identifier: Option<&str>, - raw_signature: &str, -) -> String { - let name = identifier.unwrap_or_else(|| kind.prefix()); - let tail = go_method_signature(raw_signature, identifier) - .or_else(|| function_signature(raw_signature)) - .or_else(|| python_function_signature(raw_signature)) - .or_else(|| rust_function_signature(raw_signature)) - .unwrap_or_else(|| raw_signature.to_string()); - let tail = tail.replacen("): ", ") → ", 1); - format!("{} {name}{tail}", kind.prefix()) -} - -fn summarize_variable_node( - node: Node<'_>, - kind: ChunkKind, - identifier: Option<&str>, - source: &str, -) -> Option { - let header = normalized_header(source, node.start_byte(), node.end_byte()); - let keyword = header.split_whitespace().next()?; - let name = identifier.unwrap_or_else(|| kind.prefix()); - Some(format!("{keyword} {name}")) -} - -fn summarize_statement_node(node: Node<'_>, source: &str) -> Option { - normalize_summary_text(normalized_header(source, node.start_byte(), node.end_byte()).as_str()) -} - -pub fn summary_for_node( - node: Node<'_>, - kind: ChunkKind, - identifier: Option<&str>, - raw_signature: Option<&str>, - source: &str, -) -> Option { - match kind.traits().summary { - SummaryStyle::Imports => return Some("imports".to_string()), - SummaryStyle::Function => { - if let Some(signature) = raw_signature { - return Some(summarize_function_node(kind, identifier, signature)); - } - }, - SummaryStyle::Variable => { - return summarize_variable_node(node, kind, identifier, source); - }, - SummaryStyle::Default => {}, - } - if matches!( - node.kind(), - "for_statement" - | "for_in_statement" - | "for_of_statement" - | "if_statement" - | "return_statement" - | "expression_statement" - | "call_expression" - | "call" - | "function_call" - ) { - return summarize_statement_node(node, source); - } - raw_signature - .and_then(normalize_summary_text) - .or_else(|| summarize_statement_node(node, source)) -} - -fn function_signature(header: &str) -> Option { - let start = header.find('(')?; - let end = header.rfind('{').unwrap_or(header.len()); - let signature = header.get(start..end)?.trim().trim_end_matches(';').trim(); - if signature.is_empty() { - None - } else { - Some(signature.to_string()) - } -} - -fn go_method_signature(header: &str, identifier: Option<&str>) -> Option { - let name = identifier?; - let declaration = header - .trim() - .trim_end_matches('{') - .trim_end_matches(';') - .trim(); - let rest = declaration.strip_prefix("func")?.trim_start(); - if !rest.starts_with('(') { - return None; - } - let receiver_end = find_matching_paren(rest, 0)?; - let after_receiver = rest.get(receiver_end + 1..)?.trim_start(); - let after_name = after_receiver.strip_prefix(name)?.trim_start(); - if !after_name.starts_with('(') { - return None; - } - Some(after_name.to_string()) -} - -fn find_matching_paren(text: &str, open_index: usize) -> Option { - let mut depth = 0usize; - for (index, ch) in text - .char_indices() - .skip_while(|(index, _)| *index < open_index) - { - match ch { - '(' => depth = depth.saturating_add(1), - ')' => { - depth = depth.checked_sub(1)?; - if depth == 0 { - return Some(index); - } - }, - _ => {}, - } - } - None -} - -fn python_function_signature(header: &str) -> Option { - let start = header.find('(')?; - let end = header.rfind(':').unwrap_or(header.len()); - let signature = header.get(start..end)?.trim(); - if signature.is_empty() { - None - } else { - Some(signature.to_string()) - } -} - -fn rust_function_signature(header: &str) -> Option { - let start = header.find('(')?; - let end = header.rfind('{').unwrap_or(header.len()); - let mut sig = header.get(start..end)?.trim(); - if let Some(idx) = sig.find(" where ") { - sig = sig.get(..idx)?.trim(); - } - if sig.is_empty() { - None - } else { - Some(sig.to_string()) - } -} - -// ── Trivia and attribute detection ─────────────────────────────────────── - -pub fn is_trivia_node(node: Node<'_>) -> bool { - shape::is_generic_trivia(node) -} - -pub fn is_absorbable_attribute(kind: &str) -> bool { - shape::is_generic_absorbable_attr(kind) -} - -// ── Other helpers ──────────────────────────────────────────────────────── - -pub fn looks_like_python_statement(node: Node<'_>, source: &str) -> bool { - let header = normalized_header(source, node.start_byte(), node.end_byte()); - header.contains(':') && !header.contains('{') -} - -pub fn detect_indent(source: &str, start_byte: usize) -> (u32, String) { - let line_start = source.as_bytes()[..start_byte] - .iter() - .rposition(|&b| b == b'\n') - .map_or(0, |pos| pos + 1); - let line_prefix = &source[line_start..start_byte]; - let mut cols = 0u32; - let mut ch = String::new(); - for byte in line_prefix.bytes() { - match byte { - b'\t' => { - cols += 1; - if ch.is_empty() { - ch = "\t".to_string(); - } - }, - b' ' => { - cols += 1; - if ch.is_empty() { - ch = " ".to_string(); - } - }, - _ => break, - } - } - (cols, ch) -} - -pub fn is_root_wrapper_node(node: Node<'_>) -> bool { - shape::is_root_wrapper_node(node) -} - -pub const fn line_span(start_line: usize, end_line: usize) -> usize { - end_line.saturating_sub(start_line) + 1 -} - -pub fn total_line_count(source: &str) -> usize { - if source.is_empty() { - 0 - } else { - source.bytes().filter(|byte| *byte == b'\n').count() + 1 - } -} - -pub fn first_scalar_child(node: Node<'_>) -> Option> { - named_children(node).into_iter().find(|child| { - !matches!( - child.kind(), - "block_node" - | "flow_node" - | "block_mapping" - | "flow_mapping" - | "block_sequence" - | "flow_sequence" - ) - }) -} - -#[cfg(test)] -mod tests { - use super::compute_body_inner_boundaries; - - #[test] - fn compute_body_inner_boundaries_handles_brace_and_indent_bodies() { - let ts = "function main() {\n\treturn 1;\n}\n"; - let ts_start = ts.find('{').expect("open brace"); - let ts_end = ts.rfind('}').expect("close brace") + 1; - let (ts_inner_start, ts_inner_end) = compute_body_inner_boundaries(ts, ts_start, ts_end); - assert_eq!(&ts[ts_inner_start..ts_inner_end], "\treturn 1;\n"); - - let rust = "fn main() {\n println!(\"hi\");\n}\n"; - let rust_start = rust.find('{').expect("open brace"); - let rust_end = rust.rfind('}').expect("close brace") + 1; - let (rust_inner_start, rust_inner_end) = - compute_body_inner_boundaries(rust, rust_start, rust_end); - assert_eq!(&rust[rust_inner_start..rust_inner_end], " println!(\"hi\");\n"); - - let go = "func main() {\n\treturn\n}\n"; - let go_start = go.find('{').expect("open brace"); - let go_end = go.rfind('}').expect("close brace") + 1; - let (go_inner_start, go_inner_end) = compute_body_inner_boundaries(go, go_start, go_end); - assert_eq!(&go[go_inner_start..go_inner_end], "\treturn\n"); - - let py = "def main():\n return 1\n"; - let py_body_start = py.find(" return 1").expect("body start"); - let py_body_end = py_body_start + " return 1".len(); - let (py_inner_start, py_inner_end) = - compute_body_inner_boundaries(py, py_body_start, py_body_end); - assert_eq!(&py[py_inner_start..py_inner_end], " return 1"); - } -} diff --git a/crates/pi-natives/src/chunk/conflict.rs b/crates/pi-natives/src/chunk/conflict.rs deleted file mode 100644 index 4595a93a6..000000000 --- a/crates/pi-natives/src/chunk/conflict.rs +++ /dev/null @@ -1,690 +0,0 @@ -use std::collections::{HashMap, HashSet}; - -use super::{ - chunk_checksum, - common::{detect_indent, total_line_count}, - kind::ChunkKind, - line_start_offsets, - state::ConflictMeta, - types::{ChunkNode, ChunkTree}, -}; - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct ConflictRegion { - pub ours_start_line: usize, - pub ours_end_line: usize, - pub theirs_start_line: usize, - pub theirs_end_line: usize, - pub marker_start_line: usize, - pub marker_end_line: usize, - pub ours_content: String, - pub theirs_content: String, - pub base_content: Option, - pub base_label: Option, - pub ours_label: String, - pub theirs_label: String, -} - -#[derive(Clone, Debug)] -pub struct CleanResult { - pub source: String, - pub conflicts: Vec, - pub ours_byte_ranges: Vec<(usize, usize)>, -} - -#[derive(Clone)] -struct PendingConflict { - path: Option, - parent_path: Option, - ours_start_byte: usize, - ours_end_byte: usize, - ours_content: String, - theirs_content: String, - base_content: Option, - base_label: Option, - ours_label: String, - theirs_label: String, -} - -pub fn has_conflict_markers(source: &str) -> bool { - source.contains("<<<<<<<") && source.contains("=======") && source.contains(">>>>>>>") -} - -pub fn detect_conflicts(source: &str) -> Vec { - let lines = source_lines(source); - let mut conflicts = Vec::new(); - let mut index = 0usize; - - while index < lines.len() { - let Some(ours_label) = marker_label(lines[index], "<<<<<<<") else { - index += 1; - continue; - }; - let marker_start_line = index + 1; - let ours_start_index = index + 1; - let mut separator_index = None; - let mut base_marker_index = None; - let mut base_label = None; - let mut cursor = ours_start_index; - - while cursor < lines.len() { - if let Some(label) = marker_label(lines[cursor], "|||||||") { - base_marker_index = Some(cursor); - base_label = Some(label); - cursor += 1; - while cursor < lines.len() { - if is_marker_line(lines[cursor], "=======") { - separator_index = Some(cursor); - break; - } - cursor += 1; - } - break; - } - if is_marker_line(lines[cursor], "=======") { - separator_index = Some(cursor); - break; - } - cursor += 1; - } - - let Some(separator_index) = separator_index else { - index += 1; - continue; - }; - - let theirs_start_index = separator_index + 1; - let mut end_index = theirs_start_index; - while end_index < lines.len() && marker_label(lines[end_index], ">>>>>>>").is_none() { - end_index += 1; - } - let Some(theirs_label) = lines - .get(end_index) - .and_then(|line| marker_label(line, ">>>>>>>")) - else { - index += 1; - continue; - }; - - let ours_end_index = base_marker_index.unwrap_or(separator_index); - let base_start_index = base_marker_index.map_or(separator_index, |value| value + 1); - let base_end_index = separator_index; - - let (ours_start_line, ours_end_line) = content_line_range(ours_start_index, ours_end_index); - let (theirs_start_line, theirs_end_line) = content_line_range(theirs_start_index, end_index); - - conflicts.push(ConflictRegion { - ours_start_line, - ours_end_line, - theirs_start_line, - theirs_end_line, - marker_start_line, - marker_end_line: end_index + 1, - ours_content: join_lines(&lines[ours_start_index..ours_end_index]), - theirs_content: join_lines(&lines[theirs_start_index..end_index]), - base_content: base_marker_index - .map(|_| join_lines(&lines[base_start_index..base_end_index])), - base_label, - ours_label, - theirs_label, - }); - index = end_index + 1; - } - - conflicts -} - -pub fn accept_ours(source: &str, conflicts: &[ConflictRegion]) -> CleanResult { - if conflicts.is_empty() { - return CleanResult { - source: source.to_owned(), - conflicts: Vec::new(), - ours_byte_ranges: Vec::new(), - }; - } - - let lines = source_lines(source); - let mut clean_source = String::with_capacity(source.len()); - let mut ours_byte_ranges = Vec::with_capacity(conflicts.len()); - let mut cursor_line = 1usize; - - for conflict in conflicts { - let marker_start = conflict.marker_start_line.saturating_sub(1); - for line in lines - .iter() - .take(marker_start) - .skip(cursor_line.saturating_sub(1)) - { - clean_source.push_str(line); - } - let ours_start_byte = clean_source.len(); - let ours_start_index = conflict.ours_start_line.saturating_sub(1); - let ours_end_index = if conflict.ours_start_line <= conflict.ours_end_line { - conflict.ours_end_line - } else { - ours_start_index - }; - for line in lines.iter().take(ours_end_index).skip(ours_start_index) { - clean_source.push_str(line); - } - ours_byte_ranges.push((ours_start_byte, clean_source.len())); - cursor_line = conflict.marker_end_line + 1; - } - - for line in lines.iter().skip(cursor_line.saturating_sub(1)) { - clean_source.push_str(line); - } - - CleanResult { source: clean_source, conflicts: conflicts.to_vec(), ours_byte_ranges } -} - -pub fn reconstruct_markers( - clean_source: &str, - conflict_meta: &HashMap, -) -> String { - if conflict_meta.is_empty() { - return clean_source.to_owned(); - } - - let mut conflicts = conflict_meta.iter().collect::>(); - conflicts.sort_unstable_by_key(|(_, meta)| meta.ours_start_byte); - - let mut rendered = String::with_capacity(clean_source.len() + conflict_meta.len() * 64); - let mut cursor = 0usize; - for (_, meta) in conflicts { - if meta.ours_start_byte > clean_source.len() - || meta.ours_end_byte > clean_source.len() - || meta.ours_start_byte > meta.ours_end_byte - { - continue; - } - - rendered.push_str(&clean_source[cursor..meta.ours_start_byte]); - push_marker_line(&mut rendered, "<<<<<<<", Some(meta.ours_label.as_str())); - push_conflict_content(&mut rendered, &clean_source[meta.ours_start_byte..meta.ours_end_byte]); - if let Some(base_content) = meta.base_content.as_deref() { - push_marker_line(&mut rendered, "|||||||", meta.base_label.as_deref()); - push_conflict_content(&mut rendered, base_content); - } - push_marker_line(&mut rendered, "=======", None); - push_conflict_content(&mut rendered, meta.theirs_content.as_str()); - push_marker_line(&mut rendered, ">>>>>>>", Some(meta.theirs_label.as_str())); - cursor = meta.ours_end_byte; - } - rendered.push_str(&clean_source[cursor..]); - rendered -} - -pub fn inject_conflict_chunks( - tree: &mut ChunkTree, - source: &str, - clean_result: &CleanResult, -) -> HashMap { - let mut pending = Vec::with_capacity(clean_result.conflicts.len()); - for (conflict, (ours_start_byte, ours_end_byte)) in clean_result - .conflicts - .iter() - .zip(clean_result.ours_byte_ranges.iter().copied()) - { - pending.push(PendingConflict { - path: None, - parent_path: None, - ours_start_byte, - ours_end_byte, - ours_content: conflict.ours_content.clone(), - theirs_content: conflict.theirs_content.clone(), - base_content: conflict.base_content.clone(), - base_label: conflict.base_label.clone(), - ours_label: conflict.ours_label.clone(), - theirs_label: conflict.theirs_label.clone(), - }); - } - inject_pending_conflicts(tree, source, pending) -} - -pub fn reinject_conflict_chunks( - tree: &mut ChunkTree, - source: &str, - conflict_meta: &HashMap, -) -> HashMap { - let mut pending = conflict_meta - .iter() - .map(|(path, meta)| PendingConflict { - path: Some(path.clone()), - parent_path: conflict_parent_path(path), - ours_start_byte: meta.ours_start_byte, - ours_end_byte: meta.ours_end_byte, - ours_content: source - .get(meta.ours_start_byte..meta.ours_end_byte) - .unwrap_or_default() - .to_owned(), - theirs_content: meta.theirs_content.clone(), - base_content: meta.base_content.clone(), - base_label: meta.base_label.clone(), - ours_label: meta.ours_label.clone(), - theirs_label: meta.theirs_label.clone(), - }) - .collect::>(); - pending.sort_unstable_by_key(|conflict| conflict.ours_start_byte); - inject_pending_conflicts(tree, source, pending) -} - -fn inject_pending_conflicts( - tree: &mut ChunkTree, - source: &str, - pending: Vec, -) -> HashMap { - let mut conflict_meta = HashMap::new(); - let mut existing_paths = tree - .chunks - .iter() - .map(|chunk| chunk.path.clone()) - .collect::>(); - let line_starts = line_start_offsets(source); - let mut counters = HashMap::::new(); - - for pending_conflict in pending { - if pending_conflict.ours_start_byte > pending_conflict.ours_end_byte - || pending_conflict.ours_end_byte > source.len() - { - continue; - } - - let parent_path = match pending_conflict.parent_path.as_deref() { - Some(parent_path) => { - if tree.chunks.iter().any(|chunk| chunk.path == parent_path) { - parent_path.to_owned() - } else { - continue; - } - }, - None => find_innermost_parent_path( - tree, - pending_conflict.ours_start_byte, - pending_conflict.ours_end_byte, - ) - .unwrap_or_default(), - }; - - let conflict_path = pending_conflict.path.unwrap_or_else(|| { - next_conflict_path(parent_path.as_str(), &mut counters, &existing_paths) - }); - let ours_path = format!("{conflict_path}.ours"); - let theirs_path = format!("{conflict_path}.theirs"); - existing_paths.insert(conflict_path.clone()); - existing_paths.insert(ours_path.clone()); - existing_paths.insert(theirs_path.clone()); - - let (start_line, end_line, line_count) = real_line_stats( - &line_starts, - pending_conflict.ours_start_byte, - pending_conflict.ours_end_byte, - ); - let theirs_line_count = display_line_count(pending_conflict.theirs_content.as_str()) as u32; - let (indent, indent_char) = - detect_indent(source, pending_conflict.ours_start_byte.min(source.len())); - let conflict_identifier = conflict_path - .rsplit('.') - .next() - .and_then(|leaf| leaf.strip_prefix("conflict_")) - .map(ToOwned::to_owned); - let conflict_checksum = chunk_checksum( - format!( - "{}\0{}\0{}\0{}\0{}", - pending_conflict.ours_label, - pending_conflict.theirs_label, - pending_conflict.ours_content, - pending_conflict.theirs_content, - pending_conflict.base_content.as_deref().unwrap_or_default(), - ) - .as_bytes(), - ); - let theirs_start_line = start_line; - let theirs_end_line = if theirs_line_count == 0 { - theirs_start_line - } else { - theirs_start_line + theirs_line_count - 1 - }; - let conflict_end_line = end_line.max(theirs_end_line); - let conflict_line_count = if conflict_end_line >= start_line { - conflict_end_line - start_line + 1 - } else { - 0 - }; - - tree.chunks.push(ChunkNode { - path: conflict_path.clone(), - identifier: conflict_identifier, - kind: ChunkKind::Conflict, - leaf: false, - virtual_content: None, - parent_path: Some(parent_path.clone()), - children: vec![ours_path.clone(), theirs_path.clone()], - signature: None, - start_line, - end_line: conflict_end_line, - line_count: conflict_line_count, - start_byte: pending_conflict.ours_start_byte as u32, - end_byte: pending_conflict.ours_end_byte as u32, - checksum_start_byte: pending_conflict.ours_start_byte as u32, - prologue_end_byte: None, - epilogue_start_byte: None, - checksum: conflict_checksum, - error: false, - indent, - indent_char: indent_char.clone(), - group: false, - }); - tree.chunks.push(ChunkNode { - path: ours_path.clone(), - identifier: None, - kind: ChunkKind::Ours, - leaf: true, - virtual_content: (pending_conflict.ours_start_byte == pending_conflict.ours_end_byte) - .then(String::new), - parent_path: Some(conflict_path.clone()), - children: Vec::new(), - signature: None, - start_line, - end_line, - line_count, - start_byte: pending_conflict.ours_start_byte as u32, - end_byte: pending_conflict.ours_end_byte as u32, - checksum_start_byte: pending_conflict.ours_start_byte as u32, - prologue_end_byte: None, - epilogue_start_byte: None, - checksum: chunk_checksum( - source - .as_bytes() - .get(pending_conflict.ours_start_byte..pending_conflict.ours_end_byte) - .unwrap_or_default(), - ), - error: false, - indent, - indent_char: indent_char.clone(), - group: false, - }); - tree.chunks.push(ChunkNode { - path: theirs_path.clone(), - identifier: None, - kind: ChunkKind::Theirs, - leaf: true, - virtual_content: Some(pending_conflict.theirs_content.clone()), - parent_path: Some(conflict_path.clone()), - children: Vec::new(), - signature: None, - start_line: theirs_start_line, - end_line: theirs_end_line, - line_count: theirs_line_count, - start_byte: pending_conflict.ours_start_byte as u32, - end_byte: pending_conflict.ours_start_byte as u32, - checksum_start_byte: pending_conflict.ours_start_byte as u32, - prologue_end_byte: None, - epilogue_start_byte: None, - checksum: chunk_checksum(pending_conflict.theirs_content.as_bytes()), - error: false, - indent, - indent_char, - group: false, - }); - - insert_conflict_child( - tree, - parent_path.as_str(), - conflict_path.as_str(), - pending_conflict.ours_start_byte, - pending_conflict.ours_end_byte, - ); - - conflict_meta.insert(conflict_path, ConflictMeta { - theirs_content: pending_conflict.theirs_content, - ours_label: pending_conflict.ours_label, - theirs_label: pending_conflict.theirs_label, - base_content: pending_conflict.base_content, - base_label: pending_conflict.base_label, - ours_start_byte: pending_conflict.ours_start_byte, - ours_end_byte: pending_conflict.ours_end_byte, - }); - } - - conflict_meta -} - -fn source_lines(source: &str) -> Vec<&str> { - if source.is_empty() { - Vec::new() - } else { - source.split_inclusive('\n').collect() - } -} - -fn strip_line_ending(line: &str) -> &str { - line.trim_end_matches(['\n', '\r']) -} - -fn marker_label(line: &str, prefix: &str) -> Option { - let stripped = strip_line_ending(line); - let remainder = stripped.strip_prefix(prefix)?; - Some(remainder.trim_start().to_owned()) -} - -fn is_marker_line(line: &str, prefix: &str) -> bool { - strip_line_ending(line).starts_with(prefix) -} - -fn join_lines(lines: &[&str]) -> String { - let mut joined = String::new(); - for line in lines { - joined.push_str(line); - } - joined -} - -const fn content_line_range(start_index: usize, end_index: usize) -> (usize, usize) { - if start_index < end_index { - (start_index + 1, end_index) - } else { - (start_index + 1, start_index) - } -} - -fn push_marker_line(out: &mut String, marker: &str, label: Option<&str>) { - out.push_str(marker); - if let Some(label) = label - && !label.is_empty() - { - out.push(' '); - out.push_str(label); - } - out.push('\n'); -} - -fn push_conflict_content(out: &mut String, content: &str) { - out.push_str(content); - if !content.is_empty() && !content.ends_with('\n') { - out.push('\n'); - } -} - -fn find_innermost_parent_path(tree: &ChunkTree, start: usize, end: usize) -> Option { - tree - .chunks - .iter() - .filter(|chunk| { - (chunk.start_byte as usize) <= start - && end <= (chunk.end_byte as usize) - && (chunk.end_byte as usize).saturating_sub(chunk.start_byte as usize) - >= end.saturating_sub(start) - }) - .min_by_key(|chunk| { - ( - (chunk.end_byte as usize).saturating_sub(chunk.start_byte as usize), - chunk.path.split('.').count(), - ) - }) - .map(|chunk| chunk.path.clone()) -} - -fn next_conflict_path( - parent_path: &str, - counters: &mut HashMap, - existing_paths: &HashSet, -) -> String { - let key = parent_path.to_owned(); - let next = counters.entry(key).or_insert(1); - loop { - let leaf = format!("conflict_{next}"); - let path = if parent_path.is_empty() { - leaf - } else { - format!("{parent_path}.{leaf}") - }; - *next += 1; - if !existing_paths.contains(path.as_str()) { - return path; - } - } -} - -fn insert_conflict_child( - tree: &mut ChunkTree, - parent_path: &str, - conflict_path: &str, - ours_start_byte: usize, - ours_end_byte: usize, -) { - let Some(parent_index) = tree - .chunks - .iter() - .position(|chunk| chunk.path == parent_path) - else { - return; - }; - let current_children = tree.chunks[parent_index].children.clone(); - let mut updated_children = Vec::with_capacity(current_children.len() + 1); - let mut inserted = false; - - for child_path in current_children { - let Some(child) = tree.chunks.iter().find(|chunk| chunk.path == child_path) else { - continue; - }; - let child_start = child.start_byte as usize; - let child_end = child.end_byte as usize; - let overlaps = child_start < ours_end_byte && ours_start_byte < child_end; - if overlaps { - continue; - } - if !inserted && child_start > ours_start_byte { - updated_children.push(conflict_path.to_owned()); - inserted = true; - } - updated_children.push(child_path); - } - - if !inserted { - updated_children.push(conflict_path.to_owned()); - } - - tree.chunks[parent_index] - .children - .clone_from(&updated_children); - if parent_path.is_empty() { - tree.root_children = updated_children; - } -} - -fn byte_to_line(line_starts: &[usize], byte: usize) -> u32 { - if line_starts.is_empty() { - return 0; - } - line_starts.partition_point(|offset| *offset <= byte) as u32 -} - -fn real_line_stats(line_starts: &[usize], start: usize, end: usize) -> (u32, u32, u32) { - if start == end { - let line = byte_to_line(line_starts, start); - return (line, line, 0); - } - let start_line = byte_to_line(line_starts, start); - let end_line = byte_to_line(line_starts, end.saturating_sub(1)); - let line_count = if end_line >= start_line { - end_line - start_line + 1 - } else { - 0 - }; - (start_line, end_line, line_count) -} - -fn display_line_count(content: &str) -> usize { - if content.is_empty() { - 0 - } else if content.ends_with('\n') { - content.split_terminator('\n').count() - } else { - total_line_count(content) - } -} - -fn conflict_parent_path(path: &str) -> Option { - match path.rsplit_once('.') { - Some((parent, _)) => Some(parent.to_owned()), - None if !path.is_empty() => Some(String::new()), - None => None, - } -} - -#[cfg(test)] -mod tests { - use std::collections::HashMap; - - use super::*; - use crate::chunk::state::ConflictMeta; - - #[test] - fn detects_standard_and_diff3_conflicts() { - let source = "\ -one\n<<<<<<< HEAD\nours\n||||||| base\nbase\n=======\ntheirs\n>>>>>>> topic\ntwo\n<<<<<<< \ - HEAD\nx\n=======\ny\n>>>>>>> other\n"; - let conflicts = detect_conflicts(source); - assert_eq!(conflicts.len(), 2); - assert_eq!(conflicts[0].ours_content, "ours\n"); - assert_eq!(conflicts[0].theirs_content, "theirs\n"); - assert_eq!(conflicts[0].base_content.as_deref(), Some("base\n")); - assert_eq!(conflicts[0].base_label.as_deref(), Some("base")); - assert_eq!(conflicts[1].ours_content, "x\n"); - assert_eq!(conflicts[1].theirs_content, "y\n"); - assert!(conflicts[1].base_content.is_none()); - } - - #[test] - fn accept_ours_returns_clean_source_and_byte_ranges() { - let source = "\ -fn a() {\n<<<<<<< HEAD\n\treturn foo();\n=======\n\treturn bar();\n>>>>>>> topic\n}\n"; - let conflicts = detect_conflicts(source); - let clean = accept_ours(source, &conflicts); - assert_eq!(clean.source, "fn a() {\n\treturn foo();\n}\n"); - assert_eq!(clean.ours_byte_ranges, vec![(9, 24)]); - } - - #[test] - fn reconstructs_conflict_markers_from_clean_source() { - let clean_source = "fn a() {\n\treturn foo();\n}\n"; - let mut conflict_meta = HashMap::new(); - conflict_meta.insert("fn_a.conflict_1".to_owned(), ConflictMeta { - theirs_content: "\treturn bar();\n".to_owned(), - ours_label: "HEAD".to_owned(), - theirs_label: "topic".to_owned(), - base_content: Some("\treturn baz();\n".to_owned()), - base_label: Some("base".to_owned()), - ours_start_byte: 9, - ours_end_byte: 24, - }); - - let reconstructed = reconstruct_markers(clean_source, &conflict_meta); - assert_eq!( - reconstructed, - "fn a() {\n<<<<<<< HEAD\n\treturn foo();\n||||||| base\n\treturn \ - baz();\n=======\n\treturn bar();\n>>>>>>> topic\n}\n", - ); - } -} diff --git a/crates/pi-natives/src/chunk/defaults.rs b/crates/pi-natives/src/chunk/defaults.rs deleted file mode 100644 index 1b0397fd7..000000000 --- a/crates/pi-natives/src/chunk/defaults.rs +++ /dev/null @@ -1,85 +0,0 @@ -//! Default (shared) classification logic. -//! -//! These are minimal catch-all fallbacks for node kinds not handled by any -//! per-language classifier. The real classification lives in the `ast_*` -//! modules; these defaults only fire for truly unrecognized node kinds. - -use tree_sitter::Node; - -use super::{common::*, kind::ChunkKind}; - -pub fn classify_root_default<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> { - infer_named_candidate(node, source) -} - -pub fn classify_class_default<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> { - infer_named_candidate(node, source) -} - -pub fn classify_function_default<'tree>( - node: Node<'tree>, - source: &str, -) -> RawChunkCandidate<'tree> { - let kind_name = sanitize_node_kind(node.kind()); - let kind = ChunkKind::from_sanitized_kind(kind_name); - let candidate = auto_classify(node, kind, source); - if candidate.recurse.is_some() || candidate.identifier.is_some() { - candidate - } else { - group_candidate(node, kind, source) - } -} - -pub fn classify_var_decl<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> { - if let Some(candidate) = promote_assigned_expression(node, node, source) { - return candidate; - } - if let Some(name) = extract_single_declarator_name(node, source) { - return make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None); - } - group_candidate(node, ChunkKind::Declarations, source) -} - -pub fn promote_assigned_expression<'tree>( - range_node: Node<'tree>, - declaration_node: Node<'tree>, - source: &str, -) -> Option> { - let declarators: Vec> = named_children(declaration_node) - .into_iter() - .filter(|c| c.kind() == "variable_declarator") - .collect(); - if declarators.len() != 1 { - return None; - } - - let decl = declarators[0]; - let value = decl.child_by_field_name("value")?; - let name = extract_identifier(decl, source).unwrap_or_else(|| "anonymous".to_string()); - - match value.kind() { - "arrow_function" | "function_expression" | "function" => { - let recurse = recurse_body(value, ChunkContext::FunctionBody); - Some(make_kind_chunk_from( - range_node, - value, - ChunkKind::Function, - Some(name), - source, - recurse, - )) - }, - "class" | "class_expression" => { - let recurse = recurse_class(value); - Some(make_container_chunk_from( - range_node, - value, - ChunkKind::Class, - Some(name), - source, - recurse, - )) - }, - _ => None, - } -} diff --git a/crates/pi-natives/src/chunk/edit.rs b/crates/pi-natives/src/chunk/edit.rs deleted file mode 100644 index 06b2ca7d6..000000000 --- a/crates/pi-natives/src/chunk/edit.rs +++ /dev/null @@ -1,6223 +0,0 @@ -use std::{collections::HashMap, path::Path}; - -use crate::chunk::{ - indent::{ - dedent_python_style, denormalize_from_tabs, detect_file_indent_char, detect_file_indent_step, - indent_non_empty_lines, normalize_leading_whitespace_char, normalize_to_tabs, - reindent_inserted_block, strip_content_prefixes, - }, - kind::ChunkKind, - resolve::{ - ParsedSelector, chunk_region_range, resolve_chunk_selector, - resolve_chunk_selector_with_crc_filter, resolve_chunk_with_crc, sanitize_crc, - split_selector_crc_and_region, verify_any_ancestor_crc_match, - }, - state::{ChunkState, ChunkStateInner, ConflictMeta}, - types::{ - ChunkAnchorStyle, ChunkEditOp, ChunkFocusMode, ChunkNode, ChunkRegion, EditOperation, - EditParams, EditResult, FocusedPath, RenderParams, - }, -}; - -#[derive(Clone)] -struct ScheduledEditOperation { - operation: EditOperation, - original_index: usize, - requested_selector: Option, - supplied_checksum: Option, - initial_chunk: Option, - checksum_validated: bool, -} - -#[derive(Clone, Debug, Eq, Hash, PartialEq)] -struct BatchTargetKey { - path: String, - checksum: String, -} - -#[derive(Clone)] -enum CurrentBatchTarget { - Path(String), - Missing, -} - -#[derive(Clone)] -struct AppliedEditTarget { - before: ChunkNode, - full_chunk_removed: bool, -} - -type CurrentBatchTargets = HashMap; - -#[derive(Clone, Copy, PartialEq, Eq)] -enum InsertPosition { - Before, - After, - FirstChild, - LastChild, -} - -#[derive(Clone, Copy)] -struct InsertSpacing { - blank_line_before: bool, - blank_line_after: bool, -} - -#[derive(Clone)] -struct InsertionPoint { - offset: usize, - indent: String, -} - -#[derive(Clone)] -struct ResolvedEditTarget { - chunk: ChunkNode, - region: Option, -} - -enum RegionFallback { - Reject(String), - Warn(String), -} - -const NORMALIZED_TAB_REPLACEMENT: &str = " "; -const PRESERVED_TAB_REPLACEMENT: &str = "\t"; - -const fn operation_requires_checksum(op: ChunkEditOp) -> bool { - matches!(op, ChunkEditOp::Put | ChunkEditOp::Replace | ChunkEditOp::Delete) -} - -fn resolve_initial_edit_chunk<'a>( - state: &'a ChunkStateInner, - cleaned_selector: Option<&str>, - cleaned_crc: Option<&str>, - lenient_multi_crc: bool, - warnings: &mut Vec, -) -> Result<(&'a ChunkNode, Option), String> { - if lenient_multi_crc { - let chunk = resolve_chunk_selector(state, cleaned_selector, warnings)?; - return Ok((chunk, cleaned_crc.map(str::to_owned))); - } - - let mut selector_warnings = Vec::new(); - match resolve_chunk_selector_with_crc_filter( - state, - cleaned_selector, - cleaned_crc, - &mut selector_warnings, - ) { - Ok(chunk) => { - warnings.extend(selector_warnings); - Ok((chunk, cleaned_crc.map(str::to_owned))) - }, - Err(selector_error) => { - let Some(cleaned_crc) = cleaned_crc else { - warnings.extend(selector_warnings); - return Err(selector_error); - }; - if let Ok(resolved) = - resolve_chunk_with_crc(state, cleaned_selector, Some(cleaned_crc), warnings) - { - Ok((resolved.chunk, resolved.crc)) - } else { - warnings.extend(selector_warnings); - Err(selector_error) - } - }, - } -} - -fn schedule_edit_operation( - state: &ChunkStateInner, - operation: EditOperation, - original_index: usize, - default_selector: Option<&str>, - default_crc: Option<&str>, - warnings: &mut Vec, -) -> Result { - let normalized_operation = normalize_operation_literals(&operation); - let selector = normalized_operation.sel.as_deref().or(default_selector); - let crc = normalized_operation.crc.as_deref().or_else(|| { - if normalized_operation.sel.is_none() { - default_crc - } else { - None - } - }); - let ParsedSelector { - selector: cleaned_selector, - crc: cleaned_crc, - all_crcs: parsed_crcs, - has_trailing_crc, - .. - } = split_selector_crc_and_region(selector, crc, normalized_operation.region)?; - let lenient_multi_crc = parsed_crcs.len() >= 2 || (!parsed_crcs.is_empty() && !has_trailing_crc); - let (resolved_chunk, resolved_crc) = resolve_initial_edit_chunk( - state, - cleaned_selector.as_deref(), - cleaned_crc.as_deref(), - lenient_multi_crc, - warnings, - )?; - let requires_checksum = operation_requires_checksum(normalized_operation.op); - let checksum_validated = if lenient_multi_crc { - verify_any_ancestor_crc_match(state, resolved_chunk, &parsed_crcs)?; - !parsed_crcs.is_empty() - } else { - validate_batch_crc(resolved_chunk, resolved_crc.as_deref(), requires_checksum)?; - resolved_crc.is_some() - }; - - Ok(ScheduledEditOperation { - operation, - original_index, - requested_selector: cleaned_selector, - supplied_checksum: cleaned_crc, - initial_chunk: Some(resolved_chunk.clone()), - checksum_validated, - }) -} - -pub fn apply_edits(state: &ChunkState, params: &EditParams) -> Result { - let original_text = normalize_chunk_source(state.inner().source()); - let initial_notebook_ctx = state.inner().notebook.clone(); - let initial_conflict_meta = state.inner().conflict_meta.clone(); - let mut state = rebuild_chunk_state( - original_text.clone(), - state.inner().language().to_string(), - initial_notebook_ctx.clone(), - initial_conflict_meta.clone(), - )?; - let original_state = state.clone(); - let file_indent_step = detect_file_indent_step(&state.source, &state.tree) as usize; - let file_indent_char = detect_file_indent_char(&state.source, &state.tree); - let initial_parse_errors = state.tree.parse_errors; - let initial_chunk_paths: std::collections::HashSet = - state.tree.chunks.iter().map(|c| c.path.clone()).collect(); - let initial_chunks_by_path: std::collections::HashMap = state - .tree - .chunks - .iter() - .map(|chunk| (chunk.path.clone(), chunk.clone())) - .collect(); - let initial_chunk_checksums: std::collections::HashMap = state - .tree - .chunks - .iter() - .map(|chunk| (chunk.path.clone(), chunk.checksum.clone())) - .collect(); - let normalize_indent = params.normalize_indent.unwrap_or(true); - let mut touched_paths = Vec::new(); - let mut warnings = Vec::new(); - let mut last_scheduled: Option = None; - let initial_default_selector = params.default_selector.clone(); - let initial_default_crc = params.default_crc.clone(); - - let mut scheduled_ops = Vec::with_capacity(params.operations.len()); - for (original_index, operation) in params.operations.iter().cloned().enumerate() { - match schedule_edit_operation( - &state, - operation, - original_index, - initial_default_selector.as_deref(), - initial_default_crc.as_deref(), - &mut warnings, - ) { - Ok(scheduled) => scheduled_ops.push(scheduled), - Err(err) => { - let display_path = display_path_for_file(¶ms.file_path, ¶ms.cwd); - let context = render_error_context( - &state, - params.operations[original_index] - .sel - .as_deref() - .or(initial_default_selector.as_deref()), - &display_path, - params.anchor_style, - normalize_indent, - "Current content (file unchanged)", - ); - return Err(format!( - "Edit operation {}/{} failed during initial checksum validation: {}\nNo changes \ - were saved; the batch was rolled back. Fix the failing operation and retry the \ - entire batch.{context}", - original_index + 1, - params.operations.len(), - err, - )); - }, - } - } - - let execution_ops = scheduled_ops; - let current_default_selector = initial_default_selector.as_deref(); - let mut current_default_crc = initial_default_crc; - let mut current_batch_targets = CurrentBatchTargets::new(); - let total_ops = params.operations.len(); - - for scheduled in execution_ops { - last_scheduled = Some(scheduled.clone()); - let operation = normalize_operation_literals(&scheduled.operation); - let result = match operation.op { - ChunkEditOp::Put => apply_put( - &mut state, - &operation, - &scheduled, - current_default_selector, - current_default_crc.as_deref(), - file_indent_step, - file_indent_char, - normalize_indent, - &mut touched_paths, - ¤t_batch_targets, - &mut warnings, - ), - ChunkEditOp::Replace => apply_find_replace( - &mut state, - &operation, - &scheduled, - current_default_selector, - current_default_crc.as_deref(), - file_indent_step, - file_indent_char, - normalize_indent, - &mut touched_paths, - ¤t_batch_targets, - &mut warnings, - ), - ChunkEditOp::Delete => apply_delete( - &mut state, - &operation, - &scheduled, - current_default_selector, - current_default_crc.as_deref(), - &mut touched_paths, - ¤t_batch_targets, - &mut warnings, - ), - ChunkEditOp::Before | ChunkEditOp::After | ChunkEditOp::Prepend | ChunkEditOp::Append => { - apply_insert( - &mut state, - &operation, - &scheduled, - current_default_selector, - current_default_crc.as_deref(), - file_indent_step, - file_indent_char, - normalize_indent, - &mut touched_paths, - ¤t_batch_targets, - &mut warnings, - ) - }, - }; - - let applied_target = match result { - Ok(applied_target) => applied_target, - Err(err) => { - let display_path = display_path_for_file(¶ms.file_path, ¶ms.cwd); - let sel = operation.sel.as_deref().or(current_default_selector); - let context = render_error_context( - &original_state, - sel, - &display_path, - params.anchor_style, - normalize_indent, - "Current content (batch rolled back; file unchanged)", - ); - return Err(format!( - "Edit operation {}/{} failed ({}): {}\nNo changes were saved. Fix the failing \ - operation and retry the entire batch; the batch was rolled back.{context}", - scheduled.original_index + 1, - total_ops, - describe_scheduled_operation(&scheduled), - err, - )); - }, - }; - - state = rebuild_chunk_state( - state.source.clone(), - state.language.clone(), - state.notebook.clone(), - state.conflict_meta.clone(), - )?; - update_current_batch_target(&mut current_batch_targets, &state, &scheduled, &applied_target); - if operation.sel.is_none() { - current_default_crc = None; - } - } - - let parse_valid = state.tree.parse_errors <= initial_parse_errors; - if !parse_valid && initial_parse_errors == 0 { - // Produce per-error-location summaries. Prefer the ChunkKind::Error - // chunks (which carry signature snippets) if any exist; fall back to - // the raw tree-sitter error line positions stored in the tree. - let mut error_summaries = format_parse_error_summaries(&state); - if error_summaries.is_empty() { - for &line in &state.tree.parse_error_lines { - error_summaries.push(format!( - "L{line} parse error introduced while editing {}", - last_scheduled - .as_ref() - .and_then(|s| s - .initial_chunk - .as_ref() - .map(|c| c.path.as_str()) - .or(s.requested_selector.as_deref())) - .unwrap_or(""), - )); - } - if error_summaries.is_empty() - && let Some(scheduled) = last_scheduled.as_ref() - { - let chunk_label = scheduled - .initial_chunk - .as_ref() - .map(|c| c.path.as_str()) - .or(scheduled.requested_selector.as_deref()) - .unwrap_or(""); - error_summaries.push(format!("Parse error introduced while editing {chunk_label}")); - } - } - let details = if error_summaries.is_empty() { - String::new() - } else { - format!( - "\nParse errors:\n{}", - error_summaries - .into_iter() - .map(|summary| format!("- {summary}")) - .collect::>() - .join("\n") - ) - }; - let display_path = display_path_for_file(¶ms.file_path, ¶ms.cwd); - let sel = last_scheduled - .as_ref() - .and_then(|s| s.operation.sel.as_deref()) - .or(initial_default_selector.as_deref()); - let context = render_error_context( - &state, - sel, - &display_path, - params.anchor_style, - normalize_indent, - "Hypothetical post-edit content (edit rejected; file unchanged)", - ); - return Err(format!( - "Edit rejected: introduced {} parse error(s). The file was valid before the edit but is \ - not after. The edit was rejected and no changes were saved. Fix the content and \ - retry.{details}{context}", - state.tree.parse_errors, - )); - } - if !parse_valid { - warnings.push(format!( - "Edit introduced {} new parse error(s).", - state.tree.parse_errors.saturating_sub(initial_parse_errors) - )); - } - - let display_path = display_path_for_file(¶ms.file_path, ¶ms.cwd); - let changed_virtual = original_text != state.source; - - // For notebooks, translate the virtual source back to JSON so the - // caller sees the actual ipynb file content in `diff_before`/`diff_after`. - // `initial_notebook_ctx` is the context captured at the very start of - // this call; it holds the pre-edit cell metadata. We use it to stamp - // the original JSON for `diff_before` and to produce the new JSON from - // the mutated virtual source for `diff_after`. - let (diff_before, diff_after) = if let Some(initial_ctx) = initial_notebook_ctx.as_ref() { - let before_json = crate::chunk::ast_ipynb::notebook_to_json(&original_text, initial_ctx) - .map_err(|err| format!("Failed to reconstruct pre-edit notebook JSON: {err}"))?; - let after_json = crate::chunk::ast_ipynb::notebook_to_json(&state.source, initial_ctx) - .map_err(|err| format!("Failed to serialize edited notebook JSON: {err}"))?; - (before_json, after_json) - } else { - let diff_before = if initial_conflict_meta.is_empty() { - original_text - } else { - crate::chunk::conflict::reconstruct_markers(&original_text, &initial_conflict_meta) - }; - let diff_after = if state.conflict_meta.is_empty() { - state.source.clone() - } else { - crate::chunk::conflict::reconstruct_markers(&state.source, &state.conflict_meta) - }; - (diff_before, diff_after) - }; - let changed = diff_before != diff_after || changed_virtual; - if !state.conflict_meta.is_empty() { - let mut unresolved = state.conflict_meta.keys().cloned().collect::>(); - unresolved.sort(); - warnings.push(format!( - "NOTICE: This file still has unresolved conflicts: {}.", - unresolved.join(", ") - )); - } - // Newly-created chunks (e.g. inserted siblings that landed outside the anchor's - // parent subtree) are not reflected in `touched_paths` yet. Detect any chunk - // that did not exist in the pre-edit tree and include it so the scoped - // response tree actually shows the inserted content. - for chunk in &state.tree.chunks { - if !initial_chunk_paths.contains(&chunk.path) && !touched_paths.contains(&chunk.path) { - touched_paths.push(chunk.path.clone()); - } - } - - let response_text = if changed { - render_changed_hunks( - &state, - &display_path, - &diff_before, - &diff_after, - params.anchor_style, - &touched_paths, - &initial_chunk_checksums, - &initial_chunks_by_path, - normalize_indent, - ) - } else { - render_unchanged_response(&state, &display_path, params.anchor_style, normalize_indent) - }; - - Ok(EditResult { - state: ChunkState::from_inner(state), - diff_before, - diff_after, - response_text, - changed, - parse_valid, - touched_paths, - warnings, - }) -} - -fn unique_current_chunk_by_initial_checksum<'a>( - state: &'a ChunkStateInner, - scheduled: &ScheduledEditOperation, -) -> Option<&'a ChunkNode> { - if !scheduled.checksum_validated { - return None; - } - let initial_chunk = scheduled.initial_chunk.as_ref()?; - if initial_chunk.path.is_empty() { - return state - .chunk("") - .filter(|chunk| chunk.checksum == initial_chunk.checksum); - } - - let scoped_matches = match initial_chunk.parent_path.as_deref() { - Some(parent_path) => state.chunk(parent_path).map(|parent| { - parent - .children - .iter() - .filter_map(|child_path| state.chunk(child_path)) - .filter(|chunk| chunk.checksum == initial_chunk.checksum) - .collect::>() - }), - None => Some( - state - .tree - .root_children - .iter() - .filter_map(|child_path| state.chunk(child_path)) - .filter(|chunk| chunk.checksum == initial_chunk.checksum) - .collect::>(), - ), - }; - if let Some(matches) = scoped_matches - && matches.len() == 1 - { - return Some(matches[0]); - } - - let all_matches = state - .tree - .chunks - .iter() - .filter(|chunk| chunk.checksum == initial_chunk.checksum) - .collect::>(); - if all_matches.len() == 1 { - Some(all_matches[0]) - } else { - None - } -} - -fn batch_target_key(chunk: &ChunkNode) -> BatchTargetKey { - BatchTargetKey { path: chunk.path.clone(), checksum: chunk.checksum.clone() } -} - -fn format_batch_target_key(key: &BatchTargetKey) -> String { - if key.path.is_empty() { - format!("#{}", key.checksum) - } else { - format!("{}#{}", key.path, key.checksum) - } -} - -fn resolve_current_batch_chunk<'a>( - state: &'a ChunkStateInner, - cleaned_selector: Option<&str>, - scheduled: &ScheduledEditOperation, - current_batch_targets: &CurrentBatchTargets, - warnings: &mut Vec, -) -> Result<&'a ChunkNode, String> { - if let Some(initial_chunk) = scheduled.initial_chunk.as_ref() { - let key = batch_target_key(initial_chunk); - if let Some(current_target) = current_batch_targets.get(&key) { - return match current_target { - CurrentBatchTarget::Path(path) => state.chunk(path.as_str()).ok_or_else(|| { - format!( - "Batch target {} was remapped to \"{}\" by an earlier operation, but that chunk \ - no longer exists after later edits.", - format_batch_target_key(&key), - if path.is_empty() { - "" - } else { - path.as_str() - } - ) - }), - CurrentBatchTarget::Missing => Err(format!( - "Batch target {} was removed or could not be remapped after an earlier operation.", - format_batch_target_key(&key) - )), - }; - } - } - if let Some(chunk) = unique_current_chunk_by_initial_checksum(state, scheduled) { - return Ok(chunk); - } - let resolved = resolve_chunk_with_crc(state, cleaned_selector, None, warnings)?; - Ok(resolved.chunk) -} - -fn resolve_edit_target( - state: &ChunkStateInner, - operation: &EditOperation, - scheduled: &ScheduledEditOperation, - default_selector: Option<&str>, - default_crc: Option<&str>, - current_batch_targets: &CurrentBatchTargets, - warnings: &mut Vec, -) -> Result { - let selector = operation.sel.as_deref().or(default_selector); - let crc = operation.crc.as_deref().or_else(|| { - if operation.sel.is_none() { - default_crc - } else { - None - } - }); - let ParsedSelector { selector: cleaned_selector, region: parsed_region, .. } = - split_selector_crc_and_region(selector, crc, operation.region)?; - let mut region = operation.region.or(parsed_region); - let chunk = resolve_current_batch_chunk( - state, - cleaned_selector.as_deref(), - scheduled, - current_batch_targets, - warnings, - )? - .clone(); - let requested_region = region; - let python_leaf_control_flow = state.language == "python" - && chunk.leaf - && matches!( - chunk.kind, - ChunkKind::If - | ChunkKind::Loop - | ChunkKind::Try - | ChunkKind::Block - | ChunkKind::Match - | ChunkKind::Elif - | ChunkKind::Except - ); - if let Some(action) = unsupported_region_action(state, &chunk, operation.op, requested_region) { - match action { - RegionFallback::Reject(reason) => return Err(reason), - RegionFallback::Warn(reason) => warnings.push(reason), - } - region = None; - } else if python_leaf_control_flow { - region = None; - } - - Ok(ResolvedEditTarget { chunk, region }) -} - -const fn region_label(region: ChunkRegion) -> &'static str { - match region { - ChunkRegion::Head => "^", - ChunkRegion::Body => "~", - } -} - -fn unsupported_region_action( - state: &ChunkStateInner, - chunk: &ChunkNode, - op: ChunkEditOp, - requested_region: Option, -) -> Option { - let region = requested_region?; - let label = region_label(region); - let chunk_label = chunk_path_opt(chunk); - let warn_on_region_fallback = - matches!( - state.language.as_str(), - "markdown" - | "md" | "handlebars" - | "hbs" | "yaml" - | "yml" | "json" - | "jsonc" - | "toml" | "text" - | "txt" - ); - - let detail = if state.language == "python" - && chunk.leaf - && matches!( - chunk.kind, - ChunkKind::If - | ChunkKind::Loop - | ChunkKind::Try - | ChunkKind::Block - | ChunkKind::Match - | ChunkKind::Elif - | ChunkKind::Except - ) { - Some("Python compound-statement leaf chunks do not expose a safe body/head edit boundary") - } else if chunk.kind == ChunkKind::Section - && matches!(op, ChunkEditOp::Put) - && !chunk.children.is_empty() - { - Some( - "section writes do not expose a safe body/head edit boundary, and this section has child \ - chunks that will be replaced unless you include them", - ) - } else if chunk.kind == ChunkKind::Section && matches!(op, ChunkEditOp::Put) { - Some("section writes do not expose a safe body/head edit boundary") - } else if chunk.prologue_end_byte.is_none() || chunk.epilogue_start_byte.is_none() { - Some("this chunk has no body/head edit boundary") - } else { - None - }?; - - if warn_on_region_fallback { - return Some(RegionFallback::Warn(format!( - "Region suffix `{label}` on {chunk_label} fell back to whole-chunk editing because \ - {detail}. Include the complete chunk content, including headings, fences, list markers, \ - or table rows." - ))); - } - - Some(RegionFallback::Reject(format!( - "Region suffix `{label}` is not supported on {chunk_label}: {detail}. Use the unsuffixed \ - selector with complete replacement content, or edit a parent container's `~` body instead." - ))) -} - -fn find_current_batch_target_after_edit<'a>( - state: &'a ChunkStateInner, - before: &ChunkNode, -) -> Option<&'a ChunkNode> { - if let Some(chunk) = state.chunk(before.path.as_str()) { - return Some(chunk); - } - if before.path.is_empty() { - return state.chunk(""); - } - - let same_start_same_kind = state - .tree - .chunks - .iter() - .filter(|chunk| { - !chunk.path.is_empty() - && chunk.start_line == before.start_line - && chunk.kind == before.kind - }) - .collect::>(); - if same_start_same_kind.len() == 1 { - return Some(same_start_same_kind[0]); - } - - let same_start = state - .tree - .chunks - .iter() - .filter(|chunk| !chunk.path.is_empty() && chunk.start_line == before.start_line) - .collect::>(); - if same_start.len() == 1 { - return Some(same_start[0]); - } - - None -} - -fn update_current_batch_target( - current_batch_targets: &mut CurrentBatchTargets, - state: &ChunkStateInner, - scheduled: &ScheduledEditOperation, - applied_target: &AppliedEditTarget, -) { - let Some(initial_chunk) = scheduled.initial_chunk.as_ref() else { - return; - }; - let key = batch_target_key(initial_chunk); - let current_target = if applied_target.full_chunk_removed { - CurrentBatchTarget::Missing - } else { - find_current_batch_target_after_edit(state, &applied_target.before) - .map_or(CurrentBatchTarget::Missing, |chunk| CurrentBatchTarget::Path(chunk.path.clone())) - }; - current_batch_targets.insert(key, current_target); -} - -/// Re-indent replacement content to match the original matched source's -/// indentation. Detects the base indent of the first line in `original` and -/// applies it to `replacement`. -fn reindent_replacement( - original: &str, - replacement: &str, - file_indent_char: char, - file_indent_step: usize, -) -> String { - let orig_indent = original - .lines() - .find(|line| !line.trim().is_empty()) - .map_or("", |line| &line[..line.len() - line.trim_start().len()]); - let normalized = normalize_chunk_source(replacement) - .split('\n') - .map(|line| denormalize_from_tabs(line, file_indent_char, file_indent_step)) - .collect::>() - .join("\n"); - let dedented = dedent_python_style(&normalized); - indent_non_empty_lines(&dedented, orig_indent) -} - -fn expanded_match_start_for_multiline_replacement( - source: &str, - abs_start: usize, - replacement: &str, -) -> Option { - if !replacement.contains('\n') { - return None; - } - - let line_start = source[..abs_start].rfind('\n').map_or(0, |idx| idx + 1); - if line_start == abs_start { - return None; - } - - let existing_prefix = &source[line_start..abs_start]; - if existing_prefix.is_empty() - || !existing_prefix - .bytes() - .all(|byte| matches!(byte, b' ' | b'\t')) - { - return None; - } - - if replacement.starts_with(existing_prefix) { - Some(line_start) - } else { - None - } -} - -/// Try to find `needle` in `haystack` by normalizing leading whitespace on each -/// line. Returns `(byte_offset, byte_length)` of the match in `haystack`. -fn find_indent_normalized(haystack: &str, needle: &str) -> Option<(usize, usize)> { - let needle_trimmed: Vec<&str> = needle.lines().map(|l| l.trim_start()).collect(); - if needle_trimmed.is_empty() { - return None; - } - let haystack_lines: Vec<(usize, &str)> = haystack - .split('\n') - .scan(0usize, |offset, line| { - let start = *offset; - *offset += line.len() + 1; // +1 for the \n - Some((start, line)) - }) - .collect(); - - let mut matches = Vec::new(); - 'outer: for i in 0..haystack_lines.len() { - if i + needle_trimmed.len() > haystack_lines.len() { - break; - } - for (j, needle_line) in needle_trimmed.iter().enumerate() { - if haystack_lines[i + j].1.trim_start() != *needle_line { - continue 'outer; - } - } - let start = haystack_lines[i].0; - let last_idx = i + needle_trimmed.len() - 1; - let end = haystack_lines[last_idx].0 + haystack_lines[last_idx].1.len(); - matches.push((start, end - start)); - } - if matches.len() == 1 { - Some(matches[0]) - } else { - None // 0 or ambiguous - } -} - -fn find_replace_region_range( - state: &ChunkStateInner, - anchor: &ChunkNode, - region: Option, -) -> (usize, usize) { - match region { - Some(r) => chunk_region_range(anchor, r), - None if state.language == "rust" && anchor.kind == ChunkKind::Variant => { - let offsets = line_offsets(&state.source); - (anchor.start_byte as usize, line_end_offset(&offsets, anchor.end_line, &state.source)) - }, - None => (anchor.start_byte as usize, anchor.end_byte as usize), - } -} - -fn apply_find_replace( - state: &mut ChunkStateInner, - operation: &EditOperation, - scheduled: &ScheduledEditOperation, - default_selector: Option<&str>, - default_crc: Option<&str>, - file_indent_step: usize, - file_indent_char: char, - normalize_indent: bool, - touched_paths: &mut Vec, - current_batch_targets: &CurrentBatchTargets, - warnings: &mut Vec, -) -> Result { - let target = resolve_edit_target( - state, - operation, - scheduled, - default_selector, - default_crc, - current_batch_targets, - warnings, - )?; - let anchor = target.chunk; - let target_before = anchor.clone(); - - let (region_start, region_end) = find_replace_region_range(state, &anchor, target.region); - - let find = operation.find.as_deref().unwrap_or_default(); - if find.is_empty() { - return Err(format!( - "replace on {}: 'find' cannot be empty.", - describe_scheduled_operation(scheduled) - )); - } - - let chunk_source = &state.source[region_start..region_end]; - - // Try exact match first, then fall back to indent-normalized match. - let (rel_offset, match_len) = if let Some((off, _)) = { - let mut m = chunk_source.match_indices(find); - let first = m.next(); - if first.is_some() && m.next().is_some() { - let total = 2 + chunk_source.match_indices(find).skip(2).count(); - return Err(format!( - "replace on {}: 'find' is ambiguous ({} matches in chunk). Extend 'find' with \ - surrounding context so exactly one match remains.", - anchor.path, total - )); - } - first - } { - (off, find.len()) - } else if normalize_indent && let Some((off, len)) = find_indent_normalized(chunk_source, find) { - (off, len) - } else { - return Err(format!( - "replace on {}: 'find' text not found inside chunk. Re-read the file to confirm current \ - content.", - anchor.path - )); - }; - - let raw_replacement = operation.content.as_deref().unwrap_or_default(); - let mut abs_start = region_start + rel_offset; - let abs_end = abs_start + match_len; - if let Some(expanded_start) = - expanded_match_start_for_multiline_replacement(&state.source, abs_start, raw_replacement) - { - abs_start = expanded_start; - } - - // Re-indent replacement to match the matched source's indentation when - // indent normalization is active. - let matched_source = &state.source[abs_start..abs_end]; - let replacement = if normalize_indent { - reindent_replacement(matched_source, raw_replacement, file_indent_char, file_indent_step) - } else { - raw_replacement.to_string() - }; - - let mut new_source = - String::with_capacity(state.source.len() - matched_source.len() + replacement.len()); - new_source.push_str(&state.source[..abs_start]); - new_source.push_str(&replacement); - new_source.push_str(&state.source[abs_end..]); - replace_source_and_adjust_conflicts(state, new_source, warnings); - touched_paths.push(anchor.path); - Ok(AppliedEditTarget { before: target_before, full_chunk_removed: false }) -} - -fn apply_put( - state: &mut ChunkStateInner, - operation: &EditOperation, - scheduled: &ScheduledEditOperation, - default_selector: Option<&str>, - default_crc: Option<&str>, - file_indent_step: usize, - file_indent_char: char, - normalize_indent: bool, - touched_paths: &mut Vec, - current_batch_targets: &CurrentBatchTargets, - warnings: &mut Vec, -) -> Result { - let target = resolve_edit_target( - state, - operation, - scheduled, - default_selector, - default_crc, - current_batch_targets, - warnings, - )?; - let anchor = target.chunk; - let target_before = anchor.clone(); - if anchor.kind == ChunkKind::Theirs { - return Err( - "Virtual conflict branches cannot be replaced directly. Delete conflict.theirs to accept \ - ours, delete conflict.ours to accept theirs, or replace the parent conflict chunk for a \ - manual merge." - .to_owned(), - ); - } - validate_region_edit_safety(state, &anchor, operation.op, target.region)?; - - let requested_region = requested_region_for_operation(operation, default_selector, default_crc); - - let initial_target_indent = - target_indent_for_region(state, &anchor, target.region, file_indent_char, file_indent_step); - - let content = operation.content.as_deref().unwrap_or_default(); - let mut replacement = if should_preserve_put_content_verbatim(state, &anchor, target.region) { - normalize_chunk_source(content) - } else { - normalize_inserted_content( - content, - &initial_target_indent, - Some(file_indent_step), - file_indent_char, - normalize_indent, - ) - }; - - let effective_region = target.region; - let full_chunk_removed = effective_region.is_none() && replacement.is_empty(); - if should_preserve_head_for_fallback_body_replace( - state, - &anchor, - requested_region, - target.region, - &replacement, - ) && let Some(preserved_replacement) = build_head_preserved_full_replacement( - state, - &anchor, - content, - file_indent_step, - file_indent_char, - normalize_indent, - ) { - replacement = preserved_replacement; - warnings.push(format!( - "Auto-preserved {} head while applying fallback body edit.", - chunk_path_opt(&anchor) - )); - } - - let (mut effective_region_start, effective_region_end) = match effective_region { - None => (anchor.start_byte as usize, anchor.end_byte as usize), - Some(r) => chunk_region_range(&anchor, r), - }; - if matches!(effective_region, Some(ChunkRegion::Head)) { - effective_region_start = - line_start_offset(&line_offsets(&state.source), anchor.start_line, &state.source); - } - - if effective_region.is_none() { - if !replacement.is_empty() - && !replacement.ends_with('\n') - && anchor.end_line < state.tree.line_count - { - replacement.push('\n'); - } - // If the chunk's range included a trailing blank line (common in - // markdown lists/paragraphs), preserve it so the replacement doesn't - // collapse into the next structural element. - let offsets = line_offsets(&state.source); - let last_line_text = state - .source - .split('\n') - .nth(anchor.end_line.saturating_sub(1) as usize) - .unwrap_or(""); - if last_line_text.trim().is_empty() - && !replacement.is_empty() - && !replacement.ends_with("\n\n") - { - if !replacement.ends_with('\n') { - replacement.push('\n'); - } - replacement.push('\n'); - } - let range_start = line_start_offset(&offsets, anchor.start_line, &state.source); - let mut new_source = - replace_range_by_lines(&state.source, anchor.start_line, anchor.end_line, &replacement); - if replacement.is_empty() { - new_source = cleanup_blank_line_artifacts_at_offset(&new_source, range_start); - } - if anchor.kind == ChunkKind::Conflict { - state.conflict_meta.remove(anchor.path.as_str()); - } - replace_source_and_adjust_conflicts(state, new_source, warnings); - } else { - // Preserve the region's trailing newline boundary so the next line stays - // structurally separate after a head/body replacement. - if !replacement.is_empty() - && !replacement.ends_with('\n') - && state - .source - .as_bytes() - .get(effective_region_end.saturating_sub(1)) - == Some(&b'\n') - { - replacement.push('\n'); - } - let new_source = replace_byte_range( - &state.source, - effective_region_start, - effective_region_end, - &replacement, - ); - if anchor.kind == ChunkKind::Conflict { - state.conflict_meta.remove(anchor.path.as_str()); - } - replace_source_and_adjust_conflicts(state, new_source, warnings); - } - touched_paths.push(anchor.path); - Ok(AppliedEditTarget { before: target_before, full_chunk_removed }) -} - -fn requested_region_for_operation( - operation: &EditOperation, - default_selector: Option<&str>, - default_crc: Option<&str>, -) -> Option { - if operation.region.is_some() { - return operation.region; - } - let selector = operation.sel.as_deref().or(default_selector); - let crc = operation.crc.as_deref().or(default_crc); - split_selector_crc_and_region(selector, crc, None) - .ok() - .and_then(|parsed| parsed.region) -} - -fn collapse_whitespace(text: &str) -> String { - text.split_whitespace().collect::>().join(" ") -} - -fn should_preserve_existing_epilogue(epilogue: &str) -> bool { - let first_non_empty = epilogue.lines().find(|line| !line.trim().is_empty()); - let Some(first) = first_non_empty.map(str::trim_start) else { - return false; - }; - first.starts_with('}') - || first.starts_with(']') - || first.starts_with(')') - || first.starts_with("end") -} - -fn should_preserve_put_content_verbatim( - state: &ChunkStateInner, - anchor: &ChunkNode, - region: Option, -) -> bool { - state.language == "markdown" - && anchor.path.is_empty() - && matches!(region, None | Some(ChunkRegion::Body)) -} - -fn python_head_has_decorator(state: &ChunkStateInner, anchor: &ChunkNode) -> bool { - let (head_start, head_end) = chunk_region_range(anchor, ChunkRegion::Head); - state.source[head_start..head_end] - .lines() - .any(|line| line.trim_start().starts_with('@')) -} - -fn validate_region_edit_safety( - state: &ChunkStateInner, - anchor: &ChunkNode, - op: ChunkEditOp, - region: Option, -) -> Result<(), String> { - if state.language != "python" || region != Some(ChunkRegion::Head) { - return Ok(()); - } - if matches!(op, ChunkEditOp::Delete) { - return Err(format!( - "Deleting the Python head region of {} is unsafe because it can leave an indented body \ - attached to the previous block while still parsing. Delete the whole chunk, replace the \ - whole chunk, or use replace on a specific head line instead.", - chunk_path_opt(anchor) - )); - } - if matches!(op, ChunkEditOp::Put) - && matches!(anchor.kind, ChunkKind::Class | ChunkKind::Function) - && python_head_has_decorator(state, anchor) - { - return Err(format!( - "Head writes on decorated Python {} are unsafe because decorator/signature indentation \ - changes can move the existing body while still parsing. Replace the whole chunk or use \ - replace on the exact decorator/signature line instead.", - chunk_path_opt(anchor) - )); - } - Ok(()) -} - -fn should_preserve_head_for_fallback_body_replace( - state: &ChunkStateInner, - anchor: &ChunkNode, - requested_region: Option, - resolved_region: Option, - replacement: &str, -) -> bool { - if requested_region != Some(ChunkRegion::Body) || resolved_region.is_some() { - return false; - } - if replacement.trim().is_empty() { - return false; - } - if anchor.prologue_end_byte.is_none() || anchor.epilogue_start_byte.is_none() { - return false; - } - let (head_start, head_end) = chunk_region_range(anchor, ChunkRegion::Head); - let (body_start, body_end) = chunk_region_range(anchor, ChunkRegion::Body); - if head_end <= head_start || body_end <= body_start { - return false; - } - let head_text = state.source[head_start..head_end].trim(); - if head_text.is_empty() { - return false; - } - let head_collapsed = collapse_whitespace(head_text); - if head_collapsed.is_empty() { - return false; - } - let replacement_collapsed = collapse_whitespace(replacement); - if replacement_collapsed.is_empty() { - return false; - } - !replacement_collapsed.contains(&head_collapsed) -} - -fn build_head_preserved_full_replacement( - state: &ChunkStateInner, - anchor: &ChunkNode, - content: &str, - file_indent_step: usize, - file_indent_char: char, - normalize_indent: bool, -) -> Option { - if anchor.prologue_end_byte.is_none() || anchor.epilogue_start_byte.is_none() { - return None; - } - let (_head_start, _head_end) = chunk_region_range(anchor, ChunkRegion::Head); - let (_body_start, body_end) = chunk_region_range(anchor, ChunkRegion::Body); - let chunk_start = anchor.start_byte as usize; - let chunk_end = anchor.end_byte as usize; - if chunk_end <= chunk_start { - return None; - } - let chunk_text = &state.source[chunk_start..chunk_end]; - let first_line_end = chunk_text - .find('\n') - .map_or(chunk_end, |idx| chunk_start + idx + 1); - let head = &state.source[chunk_start..first_line_end]; - if head.trim().is_empty() { - return None; - } - let inferred_body_end = body_end.min(chunk_end).max(first_line_end); - let inferred_body_indent = state.source[first_line_end..inferred_body_end] - .lines() - .find_map(|line| { - if line.trim().is_empty() { - None - } else { - Some( - line - .chars() - .take_while(|ch| *ch == ' ' || *ch == '\t') - .collect::(), - ) - } - }) - .unwrap_or_default(); - let normalized_body = normalize_inserted_content( - content, - "", - Some(file_indent_step), - file_indent_char, - normalize_indent, - ); - let effective_body_indent = if inferred_body_indent.is_empty() { - compute_insert_indent(state, anchor, true, file_indent_char, file_indent_step) - } else { - inferred_body_indent - }; - let mut body = if effective_body_indent.is_empty() { - normalized_body - } else { - indent_non_empty_lines(&normalized_body, &effective_body_indent) - }; - let epilogue_start = body_end.max(first_line_end).min(chunk_end); - let raw_epilogue = &state.source[epilogue_start..chunk_end]; - let epilogue = if should_preserve_existing_epilogue(raw_epilogue) { - raw_epilogue - } else { - "" - }; - if !body.is_empty() && !body.ends_with('\n') && !epilogue.is_empty() { - body.push('\n'); - } - Some(format!("{head}{body}{epilogue}")) -} - -fn apply_delete( - state: &mut ChunkStateInner, - operation: &EditOperation, - scheduled: &ScheduledEditOperation, - default_selector: Option<&str>, - default_crc: Option<&str>, - touched_paths: &mut Vec, - current_batch_targets: &CurrentBatchTargets, - warnings: &mut Vec, -) -> Result { - let target = resolve_edit_target( - state, - operation, - scheduled, - default_selector, - default_crc, - current_batch_targets, - warnings, - )?; - let anchor = target.chunk; - let target_before = anchor.clone(); - validate_region_edit_safety(state, &anchor, operation.op, target.region)?; - if target.region.is_none() { - match anchor.kind { - ChunkKind::Ours => { - let Some(conflict_path) = anchor.parent_path.as_deref() else { - return Err("Conflict branch is missing its parent conflict chunk".to_owned()); - }; - let Some(conflict_meta) = state.conflict_meta.remove(conflict_path) else { - return Err(format!("Conflict metadata missing for {conflict_path}")); - }; - let new_source = replace_byte_range( - &state.source, - conflict_meta.ours_start_byte, - conflict_meta.ours_end_byte, - conflict_meta.theirs_content.as_str(), - ); - replace_source_and_adjust_conflicts(state, new_source, warnings); - touched_paths.push(conflict_path.to_owned()); - return Ok(AppliedEditTarget { - before: target_before, - full_chunk_removed: true, - }); - }, - ChunkKind::Theirs => { - let Some(conflict_path) = anchor.parent_path.as_deref() else { - return Err("Conflict branch is missing its parent conflict chunk".to_owned()); - }; - state.conflict_meta.remove(conflict_path); - touched_paths.push(conflict_path.to_owned()); - return Ok(AppliedEditTarget { - before: target_before, - full_chunk_removed: true, - }); - }, - ChunkKind::Conflict => { - state.conflict_meta.remove(anchor.path.as_str()); - }, - _ => {}, - } - } - if anchor.kind == ChunkKind::Theirs { - return Err( - "Virtual conflict branches only support delete. Delete conflict.theirs to accept ours, \ - delete conflict.ours to accept theirs, or replace the parent conflict chunk for a \ - manual merge." - .to_owned(), - ); - } - - if let Some(r) = target.region { - let (range_start, range_end) = chunk_region_range(&anchor, r); - replace_source_and_adjust_conflicts( - state, - replace_byte_range(&state.source, range_start, range_end, ""), - warnings, - ); - } else { - let offsets = line_offsets(&state.source); - let range_start = line_start_offset(&offsets, anchor.start_line, &state.source); - let new_source = cleanup_blank_line_artifacts_at_offset( - &replace_range_by_lines(&state.source, anchor.start_line, anchor.end_line, ""), - range_start, - ); - replace_source_and_adjust_conflicts(state, new_source, warnings); - } - touched_paths.push(anchor.path); - Ok(AppliedEditTarget { - before: target_before, - full_chunk_removed: target.region.is_none(), - }) -} - -fn apply_insert( - state: &mut ChunkStateInner, - operation: &EditOperation, - scheduled: &ScheduledEditOperation, - default_selector: Option<&str>, - default_crc: Option<&str>, - file_indent_step: usize, - file_indent_char: char, - normalize_indent: bool, - touched_paths: &mut Vec, - current_batch_targets: &CurrentBatchTargets, - warnings: &mut Vec, -) -> Result { - let target = resolve_edit_target( - state, - operation, - scheduled, - default_selector, - default_crc, - current_batch_targets, - warnings, - )?; - let anchor = target.chunk; - let target_before = anchor.clone(); - if anchor.kind == ChunkKind::Theirs { - return Err( - "Virtual conflict branches cannot be edited in place. Delete conflict.theirs to accept \ - ours, delete conflict.ours to accept theirs, or replace the parent conflict chunk for a \ - manual merge." - .to_owned(), - ); - } - let (insertion, pos) = resolve_insertion_point( - state, - &anchor, - target.region, - operation.op, - operation.content.as_deref(), - file_indent_char, - file_indent_step, - )?; - warn_on_container_boundary_insert(&anchor, target.region, operation.op, pos, warnings); - let suppress_markdown_table_row_append = matches!(pos, InsertPosition::After) - && markdown_table_row_append_insertion_point(state, &anchor, operation.content.as_deref()) - .is_some_and(|point| point.offset == insertion.offset); - let suppress_chunk_adjacency = suppress_markdown_table_row_append - || matches!(operation.op, ChunkEditOp::Prepend | ChunkEditOp::Append) - && !(matches!(operation.op, ChunkEditOp::Append) - && pos == InsertPosition::After - && owned_container_end_line(state, &anchor) > anchor.end_line); - let spacing = compute_insert_spacing(state, &anchor, pos, suppress_chunk_adjacency); - let content = operation.content.as_deref().unwrap_or_default(); - let mut replacement = normalize_inserted_content( - content, - &insertion.indent, - Some(file_indent_step), - file_indent_char, - normalize_indent, - ); - replacement = - normalize_insertion_boundary_content(state, insertion.offset, &replacement, spacing); - - if pos == InsertPosition::FirstChild { - let body = replacement.trim_matches('\n'); - let comment_only = - !body.is_empty() && body.lines().all(|line| is_comment_only_line(line.trim())); - if comment_only - && anchor.path.is_empty() - && anchor.children.iter().any(|child| child == "preamble") - { - return Err( - "Comment-only ~.prepend on root is not allowed when the file has a preamble chunk. \ - Use replace on the preamble chunk instead." - .to_owned(), - ); - } - if comment_only && !anchor.children.is_empty() { - warnings.push( - "Comment-only ~.prepend can merge into the following chunk's first line; it is not a \ - separate named chunk." - .to_owned(), - ); - } - } - - replace_source_and_adjust_conflicts( - state, - insert_at_offset(&state.source, insertion.offset, &replacement), - warnings, - ); - touched_paths.push(anchor.path); - Ok(AppliedEditTarget { before: target_before, full_chunk_removed: false }) -} - -fn warn_on_container_boundary_insert( - anchor: &ChunkNode, - region: Option, - op: ChunkEditOp, - pos: InsertPosition, - warnings: &mut Vec, -) { - if region.is_some() || !is_container_like_chunk(anchor) { - return; - } - match (op, pos) { - (ChunkEditOp::Append, InsertPosition::After) => warnings.push(format!( - "append on container {} without `~` inserts after the chunk, not inside its body. Use \ - {}~ with append to insert inside the container.", - chunk_path_opt(anchor), - chunk_path_opt(anchor), - )), - (ChunkEditOp::Prepend, InsertPosition::Before) => warnings.push(format!( - "prepend on container {} without `~` inserts before the chunk, not inside its body. Use \ - {}~ with prepend to insert inside the container.", - chunk_path_opt(anchor), - chunk_path_opt(anchor), - )), - _ => {}, - } -} - -fn normalize_operation_literals(operation: &EditOperation) -> EditOperation { - let mut operation = operation.clone(); - if matches!(operation.sel.as_deref(), Some("null" | "undefined")) { - operation.sel = None; - } - if matches!(operation.crc.as_deref(), Some("null" | "undefined")) { - operation.crc = None; - } - operation -} - -fn normalize_chunk_source(text: &str) -> String { - text - .strip_prefix('\u{feff}') - .unwrap_or(text) - .replace("\r\n", "\n") - .replace('\r', "\n") -} - -fn rebuild_chunk_state( - source: String, - language: String, - notebook: Option, - conflict_meta: HashMap, -) -> Result { - let mut tree = if let Some(ctx) = ¬ebook { - crate::chunk::ast_ipynb::build_notebook_tree_from_virtual( - source.as_str(), - ctx.kernel_language.as_str(), - )? - } else { - crate::chunk::build_chunk_tree(source.as_str(), language.as_str()) - .map_err(|err| err.to_string())? - }; - let rebuilt_conflicts = if conflict_meta.is_empty() { - HashMap::new() - } else { - crate::chunk::conflict::reinject_conflict_chunks(&mut tree, source.as_str(), &conflict_meta) - }; - let mut inner = ChunkStateInner::new(source, language, tree); - inner.notebook = notebook; - inner.conflict_meta = rebuilt_conflicts; - Ok(inner) -} - -fn validate_batch_crc(chunk: &ChunkNode, crc: Option<&str>, required: bool) -> Result<(), String> { - if !required && crc.is_none() { - return Ok(()); - } - validate_crc(chunk, crc) -} - -fn validate_crc(chunk: &ChunkNode, crc: Option<&str>) -> Result<(), String> { - let cleaned = sanitize_crc(crc).ok_or_else(|| { - let selector = if chunk.path.is_empty() { - format!("#{}", chunk.checksum) - } else { - format!("{}#{}", chunk.path, chunk.checksum) - }; - format!( - "Checksum required for {}. Re-read the chunk to get the current checksum, then include \ - it in the selector. Hint: use target \"{}\" for container replacement, or append \ - another region such as ~.", - chunk_path_opt(chunk), - selector - ) - })?; - if chunk.checksum != cleaned { - return Err(format!( - "Checksum mismatch for {}: expected \"{}\", got \"{}\". The chunk content has changed \ - since you last read it. Use the fresh checksum from the context below to retry.", - chunk_path_opt(chunk), - chunk.checksum, - cleaned - )); - } - Ok(()) -} - -const fn chunk_path_opt(chunk: &ChunkNode) -> &str { - if chunk.path.is_empty() { - "root" - } else { - chunk.path.as_str() - } -} - -fn describe_scheduled_operation(scheduled: &ScheduledEditOperation) -> String { - let op = scheduled.operation.op.as_str(); - match (scheduled.requested_selector.as_deref(), scheduled.supplied_checksum.as_deref()) { - (Some(selector), Some(checksum)) => format!("{op} on \"{selector}#{checksum}\""), - (Some(selector), None) => format!("{op} on \"{selector}\""), - (None, Some(checksum)) => format!("{op} on \"#{checksum}\""), - (None, None) => op.to_owned(), - } -} - -/// Return `true` when `line` (already trimmed) is empty or a line comment in -/// any of the languages this tool edits. -/// -/// Distinguishes shell/Python `# comment` (hash followed by whitespace, `!`, -/// or end-of-line) from TS/JS `#private` field declarations and Rust `#[attr]` -/// attributes, which all start with `#` but are not comments. -fn is_comment_only_line(line: &str) -> bool { - if line.is_empty() { - return true; - } - // C-family single-line and block comments. `//` covers `///` doc comments. - if line.starts_with("//") || line.starts_with("/*") { - return true; - } - // Hash-family comments (shell, Python, YAML, TOML, Nix, make, ...). - // A bare `#`, shebang `#!`, or `#` followed by whitespace is a comment. - // `#[attr]` (Rust), `#![attr]` (Rust inner), and `#foo` (TS private field) - // are **not** comments. - if let Some(rest) = line.strip_prefix('#') { - return rest.is_empty() - || rest.starts_with('!') && !rest.starts_with("![") - || rest.starts_with(' ') - || rest.starts_with('\t'); - } - false -} - -fn replace_byte_range(source: &str, start: usize, end: usize, replacement: &str) -> String { - let mut new_source = String::with_capacity( - source - .len() - .saturating_sub(end.saturating_sub(start)) - .saturating_add(replacement.len()), - ); - new_source.push_str(&source[..start]); - new_source.push_str(replacement); - new_source.push_str(&source[end..]); - new_source -} - -fn changed_span(before: &str, after: &str) -> (usize, usize, usize) { - let before_bytes = before.as_bytes(); - let after_bytes = after.as_bytes(); - let mut prefix = 0usize; - let max_prefix = before_bytes.len().min(after_bytes.len()); - while prefix < max_prefix && before_bytes[prefix] == after_bytes[prefix] { - prefix += 1; - } - - let mut before_suffix = before_bytes.len(); - let mut after_suffix = after_bytes.len(); - while before_suffix > prefix && after_suffix > prefix { - if before_bytes[before_suffix - 1] != after_bytes[after_suffix - 1] { - break; - } - before_suffix -= 1; - after_suffix -= 1; - } - - (prefix, before_suffix, after_suffix) -} - -const fn adjust_offset(offset: usize, delta: isize) -> usize { - if delta >= 0 { - offset.saturating_add(delta as usize) - } else { - offset.saturating_sub((-delta) as usize) - } -} - -fn update_conflict_meta_after_source_change( - conflict_meta: &mut HashMap, - before: &str, - after: &str, - warnings: &mut Vec, -) { - let (change_start, before_end, after_end) = changed_span(before, after); - if change_start == before_end && change_start == after_end { - return; - } - - let delta = (after_end.saturating_sub(change_start) as isize) - - (before_end.saturating_sub(change_start) as isize); - let mut removed = Vec::new(); - - for (path, meta) in conflict_meta.iter_mut() { - if before_end <= meta.ours_start_byte { - meta.ours_start_byte = adjust_offset(meta.ours_start_byte, delta); - meta.ours_end_byte = adjust_offset(meta.ours_end_byte, delta); - continue; - } - if change_start >= meta.ours_end_byte { - continue; - } - if change_start >= meta.ours_start_byte && before_end <= meta.ours_end_byte { - meta.ours_end_byte = adjust_offset(meta.ours_end_byte, delta); - continue; - } - - removed.push(path.clone()); - } - - for path in removed { - conflict_meta.remove(path.as_str()); - warnings.push(format!( - "Conflict {path} no longer maps cleanly after a surrounding edit, so it was marked as \ - resolved." - )); - } -} - -fn replace_source_and_adjust_conflicts( - state: &mut ChunkStateInner, - new_source: String, - warnings: &mut Vec, -) { - let before = state.source.clone(); - update_conflict_meta_after_source_change( - &mut state.conflict_meta, - before.as_str(), - new_source.as_str(), - warnings, - ); - state.source = new_source; -} - -fn target_indent_for_region( - state: &ChunkStateInner, - anchor: &ChunkNode, - region: Option, - file_indent_char: char, - file_indent_step: usize, -) -> String { - match region { - None | Some(ChunkRegion::Head) => anchor.indent_char.repeat(anchor.indent as usize), - Some(ChunkRegion::Body) => { - compute_insert_indent(state, anchor, true, file_indent_char, file_indent_step) - }, - } -} - -fn normalize_inserted_content( - content: &str, - target_indent: &str, - file_indent_step: Option, - file_indent_char: char, - normalize_indent: bool, -) -> String { - let mut normalized = normalize_chunk_source(content); - normalized = strip_content_prefixes(&normalized); - if !normalize_indent { - let dedented = dedent_python_style(&normalized); - return indent_non_empty_lines(&dedented, target_indent); - } - normalized = normalized - .split('\n') - .map(|line| denormalize_from_tabs(line, file_indent_char, file_indent_step.unwrap_or(1))) - .collect::>() - .join("\n"); - if target_indent.is_empty() { - // Even at indent level 0, normalize the content's indent character - // to match the file's convention (e.g. LLM sends spaces for a tab file). - normalized = - normalize_leading_whitespace_char(&normalized, file_indent_char, file_indent_step); - } else { - normalized = reindent_inserted_block(&normalized, target_indent, file_indent_step); - } - normalized -} - -fn line_offsets(text: &str) -> Vec { - let mut offsets = vec![0usize]; - for (index, ch) in text.char_indices() { - if ch == '\n' { - offsets.push(index + 1); - } - } - offsets -} - -fn line_start_offset(offsets: &[usize], line: u32, text: &str) -> usize { - if line <= 1 { - 0 - } else { - offsets - .get((line - 1) as usize) - .copied() - .unwrap_or(text.len()) - } -} - -fn line_end_offset(offsets: &[usize], line: u32, text: &str) -> usize { - offsets.get(line as usize).copied().unwrap_or(text.len()) -} - -fn replace_range_by_lines(text: &str, start_line: u32, end_line: u32, replacement: &str) -> String { - let offsets = line_offsets(text); - let start_offset = line_start_offset(&offsets, start_line, text); - let end_offset = line_end_offset(&offsets, end_line, text); - format!("{}{}{}", &text[..start_offset], replacement, &text[end_offset..]) -} - -fn insert_at_offset(text: &str, offset: usize, content: &str) -> String { - format!("{}{}{}", &text[..offset], content, &text[offset..]) -} - -fn cleanup_blank_line_artifacts_at_offset(text: &str, offset: usize) -> String { - let mut run_start = offset.min(text.len()); - while run_start > 0 && text.as_bytes()[run_start - 1] == b'\n' { - run_start -= 1; - } - - let mut run_end = offset.min(text.len()); - while run_end < text.len() && text.as_bytes()[run_end] == b'\n' { - run_end += 1; - } - - let newline_run = &text[run_start..run_end]; - if !newline_run.contains("\n\n") { - return text.to_owned(); - } - - let after_run = &text[run_end..]; - let before_run = &text[..run_start]; - let after_starts_with_close = after_run - .trim_start_matches([' ', '\t']) - .chars() - .next() - .is_some_and(|ch| matches!(ch, '}' | ']' | ')')); - - if after_starts_with_close { - if newline_run.contains("\n\n\n") { - return format!("{}{}{}", before_run, collapse_newline_runs(newline_run, 2), after_run); - } - // In a deletion context, blank lines before closing delimiters are - // artifacts of the removed chunk, not intentional formatting. - return format!("{before_run}\n{after_run}"); - } - // After deleting a first-child chunk, collapse the blank line between an - // opening delimiter and the next sibling content. - let before_ends_with_open = before_run - .trim_end() - .chars() - .last() - .is_some_and(|ch| matches!(ch, '{' | '[' | '(' | ':')); - if before_ends_with_open && newline_run.contains("\n\n") && !newline_run.contains("\n\n\n") { - return format!("{before_run}\n{after_run}"); - } - if !newline_run.contains("\n\n\n") { - return text.to_owned(); - } - format!("{}{}{}", before_run, collapse_newline_runs(newline_run, 2), after_run) -} - -fn collapse_newline_runs(run: &str, max_newlines: usize) -> String { - let mut out = String::with_capacity(run.len()); - let mut newline_count = 0usize; - for ch in run.chars() { - if ch == '\n' { - newline_count += 1; - if newline_count <= max_newlines { - out.push(ch); - } - } else { - newline_count = 0; - out.push(ch); - } - } - out -} - -fn chunk_slice(text: &str, chunk: &ChunkNode) -> String { - if chunk.line_count == 0 { - return String::new(); - } - text - .split('\n') - .skip(chunk.start_line.saturating_sub(1) as usize) - .take((chunk.end_line - chunk.start_line + 1) as usize) - .collect::>() - .join("\n") -} - -const fn is_container_like_chunk(chunk: &ChunkNode) -> bool { - let traits = chunk.kind.traits(); - !chunk.leaf - || traits.container - || traits.has_addressable_members - || traits.always_preserve_children -} - -fn go_receiver_belongs_to_type(source: &str, chunk: &ChunkNode, type_name: &str) -> bool { - let header = source[chunk.start_byte as usize..chunk.end_byte as usize] - .lines() - .next() - .unwrap_or_default() - .trim_start(); - header.starts_with("func ") - && (header.contains(&format!(" {type_name})")) || header.contains(&format!("*{type_name})"))) -} - -fn owned_container_end_line(state: &ChunkStateInner, anchor: &ChunkNode) -> u32 { - if state.language != "go" || anchor.kind != ChunkKind::Type { - return anchor.end_line; - } - - let type_name = anchor - .identifier - .as_deref() - .unwrap_or_else(|| anchor.kind.prefix()); - let mut owned_end_line = anchor.end_line; - let mut top_level_chunks = state - .tree - .chunks - .iter() - .filter(|chunk| chunk.parent_path.as_deref() == Some("")) - .collect::>(); - top_level_chunks.sort_by_key(|chunk| chunk.start_line); - - let Some(start_index) = top_level_chunks - .iter() - .position(|chunk| chunk.path == anchor.path) - else { - return anchor.end_line; - }; - - for chunk in top_level_chunks.into_iter().skip(start_index + 1) { - if chunk.start_line < owned_end_line { - continue; - } - if chunk.kind == ChunkKind::Function - && go_receiver_belongs_to_type(&state.source, chunk, type_name) - { - owned_end_line = chunk.end_line; - continue; - } - break; - } - - owned_end_line -} - -fn before_chunk_insertion_point(state: &ChunkStateInner, anchor: &ChunkNode) -> InsertionPoint { - if anchor.path.is_empty() { - return InsertionPoint { offset: 0, indent: String::new() }; - } - let offsets = line_offsets(&state.source); - InsertionPoint { - offset: line_start_offset(&offsets, anchor.start_line, &state.source), - indent: anchor.indent_char.repeat(anchor.indent as usize), - } -} - -fn after_chunk_insertion_point(state: &ChunkStateInner, anchor: &ChunkNode) -> InsertionPoint { - if anchor.path.is_empty() { - return InsertionPoint { offset: state.source.len(), indent: String::new() }; - } - let offsets = line_offsets(&state.source); - let end_line = owned_container_end_line(state, anchor); - InsertionPoint { - offset: line_end_offset(&offsets, end_line, &state.source), - indent: anchor.indent_char.repeat(anchor.indent as usize), - } -} - -fn line_looks_markdown_table_row(line: &str) -> bool { - let trimmed = line.trim(); - trimmed.starts_with('|') && trimmed.ends_with('|') && trimmed.matches('|').count() >= 2 -} - -fn content_looks_like_markdown_table_rows(content: &str) -> bool { - let mut saw_row = false; - for line in content.lines() { - if line.trim().is_empty() { - continue; - } - if !line_looks_markdown_table_row(line) { - return false; - } - saw_row = true; - } - saw_row -} - -fn markdown_table_row_append_insertion_point( - state: &ChunkStateInner, - anchor: &ChunkNode, - content: Option<&str>, -) -> Option { - if state.language != "markdown" || !content.is_some_and(content_looks_like_markdown_table_rows) { - return None; - } - - let mut last_table_line = None; - let mut saw_table_line = false; - for (index, line) in state.source.split('\n').enumerate() { - let line_number = index as u32 + 1; - if line_number < anchor.start_line || line_number > anchor.end_line { - continue; - } - if line.trim().is_empty() { - continue; - } - if !line_looks_markdown_table_row(line) { - return None; - } - saw_table_line = true; - last_table_line = Some(line_number); - } - - if !saw_table_line { - return None; - } - - let offsets = line_offsets(&state.source); - let line = last_table_line?; - Some(InsertionPoint { - offset: line_end_offset(&offsets, line, &state.source), - indent: anchor.indent_char.repeat(anchor.indent as usize), - }) -} - -fn body_insertion_point( - state: &ChunkStateInner, - anchor: &ChunkNode, - at_end: bool, - file_indent_char: char, - file_indent_step: usize, -) -> InsertionPoint { - let offsets = line_offsets(&state.source); - let indent = compute_insert_indent(state, anchor, true, file_indent_char, file_indent_step); - if at_end { - if let Some(last_child_path) = anchor.children.last() - && let Some(last_child) = state - .tree - .chunks - .iter() - .find(|chunk| &chunk.path == last_child_path) - { - let child_indent = if last_child.indent_char.is_empty() { - indent - } else { - last_child.indent_char.repeat(last_child.indent as usize) - }; - return InsertionPoint { - offset: line_end_offset(&offsets, last_child.end_line, &state.source), - indent: child_indent, - }; - } - let (_, body_end) = chunk_region_range(anchor, ChunkRegion::Body); - return InsertionPoint { offset: body_end, indent }; - } - - if let Some(first_child_path) = anchor.children.first() - && let Some(first_child) = state - .tree - .chunks - .iter() - .find(|chunk| &chunk.path == first_child_path) - { - return InsertionPoint { - offset: line_start_offset(&offsets, first_child.start_line, &state.source), - indent, - }; - } - let (body_start, _) = chunk_region_range(anchor, ChunkRegion::Body); - InsertionPoint { offset: body_start, indent } -} - -fn resolve_insertion_point( - state: &ChunkStateInner, - anchor: &ChunkNode, - region: Option, - op: ChunkEditOp, - file_content: Option<&str>, - file_indent_char: char, - file_indent_step: usize, -) -> Result<(InsertionPoint, InsertPosition), String> { - match (region, op) { - // Before chunk boundary - (None, ChunkEditOp::Before | ChunkEditOp::Prepend) => { - Ok((before_chunk_insertion_point(state, anchor), InsertPosition::Before)) - }, - // After chunk boundary - (None, ChunkEditOp::After | ChunkEditOp::Append) => Ok(( - markdown_table_row_append_insertion_point(state, anchor, file_content) - .unwrap_or_else(|| after_chunk_insertion_point(state, anchor)), - InsertPosition::After, - )), - // Inner first-child position - (Some(ChunkRegion::Body), ChunkEditOp::Before | ChunkEditOp::Prepend) - | (Some(ChunkRegion::Head), ChunkEditOp::After | ChunkEditOp::Append) => Ok(( - body_insertion_point(state, anchor, false, file_indent_char, file_indent_step), - InsertPosition::FirstChild, - )), - // Inner last-child position - (Some(ChunkRegion::Body), ChunkEditOp::After | ChunkEditOp::Append) - | (Some(ChunkRegion::Head), ChunkEditOp::Before | ChunkEditOp::Prepend) => Ok(( - body_insertion_point(state, anchor, true, file_indent_char, file_indent_step), - InsertPosition::LastChild, - )), - (_, ChunkEditOp::Put | ChunkEditOp::Replace | ChunkEditOp::Delete) => { - Err("Internal error: insertion point requested for non-insert op".to_owned()) - }, - } -} - -fn indent_prefix_for_level( - anchor: &ChunkNode, - file_indent_char: char, - file_indent_step: usize, - extra_levels: usize, -) -> String { - let step = file_indent_step.max(1); - let indent_char = if matches!(file_indent_char, ' ' | '\t') { - file_indent_char - } else { - anchor.indent_char.chars().next().unwrap_or(' ') - }; - if indent_char == '\t' { - return "\t".repeat(anchor.indent as usize + extra_levels); - } - let indent_levels = (anchor.indent as usize / step).saturating_add(extra_levels); - " ".repeat(step * indent_levels) -} - -fn compute_insert_indent( - state: &ChunkStateInner, - anchor: &ChunkNode, - inside: bool, - file_indent_char: char, - file_indent_step: usize, -) -> String { - if !inside || anchor.path.is_empty() { - return String::new(); - } - if let Some(first_child_path) = anchor.children.first() - && let Some(first_child) = state - .tree - .chunks - .iter() - .find(|chunk| &chunk.path == first_child_path) - { - let indent_char = if first_child.indent_char.is_empty() { - if anchor.indent_char.is_empty() { - "\t" - } else { - anchor.indent_char.as_str() - } - } else { - first_child.indent_char.as_str() - }; - return indent_char.repeat(first_child.indent as usize); - } - - // Scan only the ~ region (between prologue and epilogue), not the full - // chunk. This avoids picking up the closing delimiter's indent for - // empty/sparse bodies. - let (body_start, body_end) = chunk_region_range(anchor, ChunkRegion::Body); - if body_start < body_end && body_end <= state.source.len() { - for line in state.source[body_start..body_end].split('\n') { - if line.trim().is_empty() { - continue; - } - let prefix_len = line.len() - line.trim_start_matches([' ', '\t']).len(); - if prefix_len > 0 { - return line[..prefix_len].to_owned(); - } - break; - } - } - - let indent_char = if anchor.indent_char.is_empty() { - file_indent_char.to_string() - } else { - anchor.indent_char.clone() - }; - if indent_char == "\t" { - "\t".repeat(anchor.indent as usize + 1) - } else { - indent_prefix_for_level(anchor, file_indent_char, file_indent_step, 1) - } -} - -fn sibling_index(state: &ChunkStateInner, anchor: &ChunkNode) -> Option<(usize, usize)> { - let parent_path = anchor.parent_path.as_deref().unwrap_or(""); - let parent = state - .tree - .chunks - .iter() - .find(|chunk| chunk.path == parent_path)?; - let index = parent - .children - .iter() - .position(|child| child == &anchor.path)?; - Some((index, parent.children.len())) -} - -fn has_sibling_before(state: &ChunkStateInner, anchor: &ChunkNode) -> bool { - sibling_index(state, anchor).is_some_and(|(index, _)| index > 0) -} - -fn has_sibling_after(state: &ChunkStateInner, anchor: &ChunkNode) -> bool { - sibling_index(state, anchor).is_some_and(|(index, total)| index + 1 < total) -} - -fn container_has_interior_content(state: &ChunkStateInner, anchor: &ChunkNode) -> bool { - if !is_container_like_chunk(anchor) { - return false; - } - chunk_slice(&state.source, anchor) - .split('\n') - .skip(1) - .collect::>() - .into_iter() - .rev() - .skip(1) - .any(|line| !line.trim().is_empty()) -} - -fn visible_child_chunks<'a>( - state: &'a ChunkStateInner, - anchor: &'a ChunkNode, -) -> Vec<&'a ChunkNode> { - anchor - .children - .iter() - .filter_map(|child_path| { - state - .tree - .chunks - .iter() - .find(|chunk| chunk.path == *child_path) - }) - .filter(|child| child.kind != ChunkKind::Chunk) - .collect() -} - -fn sibling_gap_has_blank_line( - state: &ChunkStateInner, - left: &ChunkNode, - right: &ChunkNode, -) -> bool { - right.start_line > owned_container_end_line(state, left) + 1 -} - -/// Returns true if a container's children should be separated by blank lines. -/// Root-level children (functions, classes) and containers with non-leaf -/// children (methods) want blank line spacing. Containers whose children are -/// all packed declarations (struct fields, enum variants) are tightly packed. -fn children_want_blank_line_spacing(state: &ChunkStateInner, anchor: &ChunkNode) -> bool { - if anchor.path.is_empty() { - let root_children = visible_child_chunks(state, anchor); - if root_children.len() >= 2 { - return root_children - .windows(2) - .any(|pair| sibling_gap_has_blank_line(state, pair[0], pair[1])); - } - return true; - } - if anchor.children.is_empty() { - return true; - } - - let visible_children = visible_child_chunks(state, anchor); - if visible_children.is_empty() { - return true; - } - if visible_children.len() >= 2 { - return visible_children - .windows(2) - .any(|pair| sibling_gap_has_blank_line(state, pair[0], pair[1])); - } - - let all_packed = visible_children - .iter() - .all(|child| child.kind.traits().packed); - !all_packed -} - -/// Returns true if sibling insertions around `anchor` should have blank line -/// spacing. Checks whether the anchor's parent container uses spaced or packed -/// layout. -fn is_spaced_sibling(state: &ChunkStateInner, anchor: &ChunkNode) -> bool { - let parent_path = anchor.parent_path.as_deref().unwrap_or(""); - if let Some(parent) = state.tree.chunks.iter().find(|c| c.path == parent_path) { - children_want_blank_line_spacing(state, parent) - } else { - true // Default to spaced if parent not found - } -} - -fn compute_insert_spacing( - state: &ChunkStateInner, - anchor: &ChunkNode, - pos: InsertPosition, - is_prepend_or_append: bool, -) -> InsertSpacing { - let has_interior_content = container_has_interior_content(state, anchor); - let markdown_block_spacing = state.language == "markdown"; - match pos { - InsertPosition::FirstChild => { - let spaced = markdown_block_spacing || children_want_blank_line_spacing(state, anchor); - InsertSpacing { - blank_line_before: false, - blank_line_after: spaced && (!anchor.children.is_empty() || has_interior_content), - } - }, - InsertPosition::LastChild => { - let spaced = markdown_block_spacing || children_want_blank_line_spacing(state, anchor); - InsertSpacing { - blank_line_before: spaced && (!anchor.children.is_empty() || has_interior_content), - blank_line_after: markdown_block_spacing && has_sibling_after(state, anchor), - } - }, - InsertPosition::Before => { - let spaced = markdown_block_spacing || is_spaced_sibling(state, anchor); - InsertSpacing { - blank_line_before: has_sibling_before(state, anchor) && spaced, - // When the op is `prepend` (container.prepend), omit the trailing - // blank line so the content stays adjacent to the chunk and gets - // absorbed as leading trivia on tree rebuild. - blank_line_after: !is_prepend_or_append && spaced, - } - }, - InsertPosition::After => { - let spaced = markdown_block_spacing || is_spaced_sibling(state, anchor); - InsertSpacing { - // When the op is `append` (container.append), omit the leading - // blank line so the content stays adjacent to the chunk. - blank_line_before: !is_prepend_or_append && spaced, - blank_line_after: has_sibling_after(state, anchor) && spaced, - } - }, - } -} - -fn count_trailing_newlines_before_offset(text: &str, offset: usize) -> usize { - let mut count = 0usize; - let bytes = text.as_bytes(); - let mut index = offset; - while index > 0 && bytes[index - 1] == b'\n' { - count += 1; - index -= 1; - } - count -} - -fn count_leading_newlines_after_offset(text: &str, offset: usize) -> usize { - let mut count = 0usize; - let bytes = text.as_bytes(); - let mut index = offset; - while index < bytes.len() && bytes[index] == b'\n' { - count += 1; - index += 1; - } - count -} - -fn normalize_insertion_boundary_content( - state: &ChunkStateInner, - offset: usize, - content: &str, - spacing: InsertSpacing, -) -> String { - let trimmed = content.trim_matches('\n'); - if trimmed.is_empty() { - return content.to_owned(); - } - - let prev_char = if offset > 0 { - state - .source - .as_bytes() - .get(offset - 1) - .copied() - .map(char::from) - } else { - None - }; - let next_char = state.source.as_bytes().get(offset).copied().map(char::from); - let prefix_newlines = if spacing.blank_line_before { - 2usize.saturating_sub(count_trailing_newlines_before_offset(&state.source, offset)) - } else { - usize::from(prev_char.is_some() && prev_char != Some('\n')) - }; - let leading_newlines_after = count_leading_newlines_after_offset(&state.source, offset); - let suffix_newlines = if spacing.blank_line_after { - 2usize.saturating_sub(leading_newlines_after) - } else if leading_newlines_after == 1 && content.ends_with('\n') { - // Preserve an existing blank-line separator when appending line-oriented - // content immediately before the separator's single newline. - 1 - } else { - usize::from(next_char.is_some() && next_char != Some('\n')) - }; - - format!("{}{}{}", "\n".repeat(prefix_newlines), trimmed, "\n".repeat(suffix_newlines)) -} - -fn line_column_at_offset(text: &str, offset: usize) -> (usize, usize) { - let offsets = line_offsets(text); - let mut low = 0usize; - let mut high = offsets.len().saturating_sub(1); - while low <= high { - let mid = usize::midpoint(low, high); - let start = offsets[mid]; - let next = offsets.get(mid + 1).copied().unwrap_or(text.len() + 1); - if offset < start { - if mid == 0 { - break; - } - high = mid - 1; - continue; - } - if offset >= next { - low = mid + 1; - continue; - } - return (mid + 1, offset - start + 1); - } - (offsets.len(), 1) -} - -fn format_parse_error_summaries(state: &ChunkStateInner) -> Vec { - state - .tree - .chunks - .iter() - .filter(|chunk| chunk.error) - .take(3) - .map(|chunk| { - let (line, column) = line_column_at_offset(&state.source, chunk.start_byte as usize); - match chunk - .signature - .as_deref() - .map(str::trim) - .filter(|value| !value.is_empty()) - { - Some(snippet) => format!("L{line}:C{column} unexpected syntax near {snippet:?}"), - None => format!("L{line}:C{column} unexpected syntax"), - } - }) - .collect() -} - -fn display_path_for_file(file_path: &str, cwd: &str) -> String { - let file = Path::new(file_path); - let cwd = Path::new(cwd); - match file.strip_prefix(cwd) { - Ok(relative) => relative.to_string_lossy().replace('\\', "/"), - Err(_) => file.to_string_lossy().replace('\\', "/"), - } -} - -/// A parsed unified diff hunk. -struct DiffHunk { - header: String, - lines: Vec, - old_start: u32, - old_len: u32, - new_start: u32, -} - -/// Normalize the content part of a diff hunk line (after the +/-/space prefix) -/// so that its indentation matches the chunk tree display format. -fn render_hunk_line( - line: &str, - normalize_indent: bool, - indent_char: char, - indent_step: usize, -) -> String { - if !normalize_indent { - return line.to_owned(); - } - if line.is_empty() { - return line.to_owned(); - } - let first = line.as_bytes()[0]; - if matches!(first, b'+' | b'-' | b' ') { - let prefix = &line[..1]; - let content = &line[1..]; - let normalized = normalize_to_tabs(content, indent_char, indent_step); - format!("{prefix}{normalized}") - } else { - line.to_owned() - } -} - -/// Generate unified diff hunks between two texts using the `similar` crate. -fn generate_diff_hunks(before: &str, after: &str, context: usize) -> Vec { - use similar::{ChangeTag, TextDiff}; - - let diff = TextDiff::from_lines(before, after); - let mut hunks = Vec::new(); - - for group in diff.grouped_ops(context) { - let mut hunk_lines = Vec::new(); - - let first = &group[0]; - let last = &group[group.len() - 1]; - let old_start = first.old_range().start + 1; - let old_len = last.old_range().end - first.old_range().start; - let new_start = first.new_range().start + 1; - let new_len = last.new_range().end - first.new_range().start; - - let header = format!("@@ -{old_start},{old_len} +{new_start},{new_len} @@"); - - for op in &group { - for change in diff.iter_changes(op) { - let line = change.value().trim_end_matches('\n'); - match change.tag() { - ChangeTag::Equal => hunk_lines.push(format!(" {line}")), - ChangeTag::Delete => hunk_lines.push(format!("-{line}")), - ChangeTag::Insert => hunk_lines.push(format!("+{line}")), - } - } - } - - hunks.push(DiffHunk { - header, - lines: hunk_lines, - old_start: old_start as u32, - old_len: old_len as u32, - new_start: new_start as u32, - }); - } - - hunks -} - -fn deleted_chunk_anchor_label(chunk: &ChunkNode, style: ChunkAnchorStyle) -> String { - match style { - ChunkAnchorStyle::Full | ChunkAnchorStyle::FullOmit => chunk.path.clone(), - ChunkAnchorStyle::Kind - | ChunkAnchorStyle::KindOmit - | ChunkAnchorStyle::Bare - | ChunkAnchorStyle::None => chunk.kind.path_segment(chunk.identifier.as_deref()), - } -} - -fn deleted_chunk_anchor_indent( - chunk: &ChunkNode, - normalize_indent: bool, - file_indent_char: char, - file_indent_step: usize, - tab_replacement: &str, -) -> String { - let indent_char = if chunk.indent_char.is_empty() { - file_indent_char.to_string() - } else { - chunk.indent_char.clone() - }; - let raw_indent = indent_char.repeat(chunk.indent as usize); - if normalize_indent { - normalize_to_tabs(&raw_indent, file_indent_char, file_indent_step) - } else { - raw_indent.replace('\t', tab_replacement) - } -} - -fn deleted_hunk_owner<'a>( - hunk: &DiffHunk, - before_chunks: &'a HashMap, - current_lookup: &HashMap<&str, &ChunkNode>, - touched_paths: &[String], -) -> Option<&'a ChunkNode> { - if hunk.old_len == 0 { - return None; - } - let old_end = hunk - .old_start - .saturating_add(hunk.old_len.saturating_sub(1)); - touched_paths - .iter() - .filter(|path| !current_lookup.contains_key(path.as_str())) - .filter_map(|path| before_chunks.get(path)) - .filter(|chunk| chunk.start_line <= old_end && hunk.old_start <= chunk.end_line) - .min_by_key(|chunk| chunk.line_count) -} - -/// Render the response text for a changed file, combining the current chunked -/// tree view with inline diff hunks placed inside the owning chunk blocks. -fn render_changed_hunks( - state: &ChunkStateInner, - display_path: &str, - before: &str, - after: &str, - anchor_style: Option, - touched_paths: &[String], - before_checksums: &std::collections::HashMap, - before_chunks: &HashMap, - normalize_indent: bool, -) -> String { - use std::collections::{HashMap, HashSet}; - - let show_leaf_preview = state.language == "tlaplus"; - let focused_paths = compute_focus(state.tree(), touched_paths); - let hunks = generate_diff_hunks(before, after, 0); - - let tree = state.tree(); - let tab_replacement = if normalize_indent { - NORMALIZED_TAB_REPLACEMENT - } else { - PRESERVED_TAB_REPLACEMENT - }; - let file_indent_char = detect_file_indent_char(state.source(), tree); - let file_indent_step = detect_file_indent_step(state.source(), tree) as usize; - let lookup: HashMap<&str, &ChunkNode> = - tree.chunks.iter().map(|c| (c.path.as_str(), c)).collect(); - let render_indent = normalize_indent.then_some((file_indent_char, file_indent_step)); - let mut inline_hunks: HashMap> = HashMap::new(); - let mut changed_anchor_paths = HashSet::new(); - let style = anchor_style.unwrap_or_default(); - - for hunk in &hunks { - if let Some(deleted_chunk) = deleted_hunk_owner(hunk, before_chunks, &lookup, touched_paths) { - let owner_path = deleted_chunk - .parent_path - .as_deref() - .unwrap_or("") - .to_owned(); - let anchor_indent = deleted_chunk_anchor_indent( - deleted_chunk, - normalize_indent, - file_indent_char, - file_indent_step, - tab_replacement, - ); - let diff_indent = if normalize_indent { - format!("{anchor_indent}\t") - } else { - format!("{anchor_indent}{tab_replacement}") - }; - let anchor_label = deleted_chunk_anchor_label(deleted_chunk, style); - let mut lines = Vec::with_capacity(hunk.lines.len() + 2); - lines.push(crate::chunk::render::InlineHunkLine { - text: style.render( - &anchor_indent, - anchor_label.as_str(), - deleted_chunk.checksum.as_str(), - ), - marker: Some('-'), - }); - lines.push(crate::chunk::render::InlineHunkLine { - text: format!("{diff_indent}{}", hunk.header), - marker: None, - }); - for line in &hunk.lines { - let normalized = - render_hunk_line(line, normalize_indent, file_indent_char, file_indent_step); - lines.push(crate::chunk::render::InlineHunkLine { - text: format!("{diff_indent}{normalized}"), - marker: None, - }); - } - inline_hunks - .entry(owner_path) - .or_default() - .push(crate::chunk::render::InlineHunk { lines }); - continue; - } - - let owner_path = - crate::chunk::render::find_hunk_owner_chunk(tree, &lookup, hunk.new_start).unwrap_or(""); - let indent = if owner_path.is_empty() { - String::new() - } else { - crate::chunk::render::hunk_indent_for_chunk( - &lookup, - owner_path, - state.source(), - tab_replacement, - render_indent, - ) - }; - let mut lines = Vec::with_capacity(hunk.lines.len() + 1); - lines.push(crate::chunk::render::InlineHunkLine { - text: format!("{indent}{}", hunk.header), - marker: None, - }); - for line in &hunk.lines { - let normalized = - render_hunk_line(line, normalize_indent, file_indent_char, file_indent_step); - lines.push(crate::chunk::render::InlineHunkLine { - text: format!("{indent}{normalized}"), - marker: None, - }); - } - inline_hunks - .entry(owner_path.to_owned()) - .or_default() - .push(crate::chunk::render::InlineHunk { lines }); - } - - for path in touched_paths { - let mut current = Some(path.as_str()); - while let Some(chunk_path) = current { - if chunk_path.is_empty() { - break; - } - let Some(chunk) = lookup.get(chunk_path) else { - current = chunk_path.rfind('.').map(|dot| &chunk_path[..dot]); - continue; - }; - if before_checksums - .get(&chunk.path) - .is_none_or(|previous| previous != &chunk.checksum) - { - changed_anchor_paths.insert(chunk.path.clone()); - } - current = chunk.parent_path.as_deref(); - } - } - - crate::chunk::render::render_state_with_hunks( - state, - &RenderParams { - chunk_path: Some(String::new()), - title: display_path.to_owned(), - language_tag: Some(state.language.clone()), - visible_range: None, - render_children_only: true, - omit_checksum: false, - anchor_style, - show_leaf_preview, - tab_replacement: Some(tab_replacement.to_owned()), - normalize_indent: Some(normalize_indent), - focused_paths, - }, - inline_hunks, - changed_anchor_paths, - ) -} - -/// Build a focus list that includes touched chunks as Expanded and all -/// ancestors as Container. -/// Falls back to no focus (full render) when more than 20 chunks were touched. -fn compute_focus( - tree: &crate::chunk::types::ChunkTree, - touched: &[String], -) -> Option> { - use std::collections::HashMap; - - if touched.is_empty() || touched.len() > 20 { - return None; - } - - let lookup: HashMap<&str, &ChunkNode> = - tree.chunks.iter().map(|c| (c.path.as_str(), c)).collect(); - let mut focus: HashMap = HashMap::new(); - - for path in touched { - let Some(chunk) = lookup.get(path.as_str()) else { - // Deleted chunk: derive parent from the path string and mark it - // Expanded so the diff hunk (owned by the parent) still renders. - if let Some(dot) = path.rfind('.') { - let parent_path = &path[..dot]; - focus - .entry(parent_path.to_string()) - .and_modify(|m| { - if *m == ChunkFocusMode::Container { - *m = ChunkFocusMode::Expanded; - } - }) - .or_insert(ChunkFocusMode::Expanded); - // Walk ancestors of the parent upward. - let mut current = lookup - .get(parent_path) - .and_then(|p| p.parent_path.as_deref()); - while let Some(anc) = current { - focus - .entry(anc.to_string()) - .or_insert(ChunkFocusMode::Container); - current = lookup.get(anc).and_then(|p| p.parent_path.as_deref()); - } - } - continue; - }; - focus.insert(path.clone(), ChunkFocusMode::Expanded); - - // Ancestors -> Container (don't downgrade Expanded). - let mut current = chunk.parent_path.as_deref(); - while let Some(parent_path) = current { - focus - .entry(parent_path.to_string()) - .or_insert(ChunkFocusMode::Container); - current = lookup - .get(parent_path) - .and_then(|p| p.parent_path.as_deref()); - } - } - - // Root chunk must always be Container so the walk starts. - focus - .entry(String::new()) - .or_insert(ChunkFocusMode::Container); - - Some( - focus - .into_iter() - .map(|(path, mode)| FocusedPath { path, mode }) - .collect(), - ) -} - -/// Render a focused chunk view to append to error messages. Resolves the -/// selector ignoring CRC so the agent sees fresh anchors without a re-read. -fn render_error_context( - state: &ChunkStateInner, - selector: Option<&str>, - display_path: &str, - anchor_style: Option, - normalize_indent: bool, - label: &str, -) -> String { - // When an edit introduces a parse error we render the full chunk tree - // (no focus) so the agent can see the failure location and surrounding - // context in a single response, without a follow-up read. For non-parse - // failures (a single operation rejected before parse validation) the - // error is local to the targeted chunk, so we keep a narrow focus. - let has_parse_errors = !state.tree().parse_error_lines.is_empty(); - let focused_paths = if has_parse_errors { - None - } else { - let Ok(ParsedSelector { selector: clean_path, .. }) = - split_selector_crc_and_region(selector, None, None) - else { - return String::new(); - }; - let mut ignored = Vec::new(); - let Ok(chunk) = resolve_chunk_selector(state, clean_path.as_deref(), &mut ignored) else { - return String::new(); - }; - compute_focus(state.tree(), std::slice::from_ref(&chunk.path)) - }; - let tab_replacement = if normalize_indent { - NORMALIZED_TAB_REPLACEMENT - } else { - PRESERVED_TAB_REPLACEMENT - }; - let rendered = crate::chunk::render::render_state(state, &RenderParams { - chunk_path: Some(String::new()), - title: display_path.to_owned(), - language_tag: Some(state.language.clone()), - visible_range: None, - render_children_only: true, - omit_checksum: false, - anchor_style, - show_leaf_preview: true, - tab_replacement: Some(tab_replacement.to_owned()), - normalize_indent: Some(normalize_indent), - focused_paths, - }); - format!("\n\n{label}:\n{rendered}") -} - -fn render_unchanged_response( - state: &ChunkStateInner, - display_path: &str, - anchor_style: Option, - normalize_indent: bool, -) -> String { - let tab_replacement = if normalize_indent { - NORMALIZED_TAB_REPLACEMENT - } else { - PRESERVED_TAB_REPLACEMENT - }; - crate::chunk::render::render_state(state, &RenderParams { - chunk_path: Some(String::new()), - title: display_path.to_owned(), - language_tag: Some(state.language.clone()), - visible_range: None, - render_children_only: true, - omit_checksum: false, - anchor_style, - focused_paths: None, - show_leaf_preview: true, - tab_replacement: Some(tab_replacement.to_owned()), - normalize_indent: Some(normalize_indent), - }) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::chunk::build_chunk_tree; - - fn state_for(source: &str, language: &str) -> ChunkState { - let tree = build_chunk_tree(source, language).expect("tree should build"); - ChunkState::from_inner(ChunkStateInner::new(source.to_owned(), language.to_owned(), tree)) - } - - fn parsed_state_for(source: &str, language: &str) -> ChunkState { - ChunkState::parse(source.to_owned(), language.to_owned()).expect("state should parse") - } - - fn apply_single_edit( - state: &ChunkState, - file_path: &str, - operation: EditOperation, - ) -> EditResult { - apply_edits(state, &EditParams { - operations: vec![operation], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: file_path.to_owned(), - normalize_indent: None, - }) - .expect("edit should apply") - } - - fn edit_params(operations: Vec) -> EditParams { - EditParams { - operations, - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.ts".to_owned(), - normalize_indent: None, - } - } - - #[test] - fn root_level_replace_preserves_space_indentation() { - let source = "fn main() {\n println!(\"old\");\n}\n"; - let state = state_for(source, "rust"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some("fn_mai".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("fn main() {\n println!(\"new\");\n}".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.rs".to_owned(), - normalize_indent: None, - }) - .expect("edit should apply"); - - assert!( - result.diff_after.contains("println!(\"new\");"), - "expected updated body text, got {:?}", - result.diff_after - ); - assert!( - !result.diff_after.contains("\n\tprintln!(\"new\");\n"), - "expected no tab-indented body, got {:?}", - result.diff_after - ); - } - - #[test] - fn diff_hunks_use_normalized_indentation() { - // Source uses 4-space indentation with nested children, so the tree - // can detect indent_char=' ' and indent_step=4. The diff hunk lines - // in the response should use tab-normalized indentation (matching the - // read tool's output) instead of the raw file indentation. - let source = "class Foo {\n value: number = 0;\n\n increment(): void {\n \ - this.value += 1;\n }\n\n decrement(): void {\n this.value -= \ - 1;\n }\n}\n"; - let state = state_for(source, "typescript"); - let chunk = state.inner().chunk("cls_Foo.fn_inc").expect("fn_inc"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some("cls_Foo.fn_inc".to_owned()), - crc: Some(chunk.checksum.clone()), - region: Some(ChunkRegion::Body), - content: Some("this.value += 2;\n".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.ts".to_owned(), - normalize_indent: None, - }) - .expect("edit should apply"); - - // The response text should contain diff hunks with tab-normalized - // indentation, not the raw 4-space (two levels = 8-space) indentation. - assert!( - result.response_text.contains("-\t\tthis.value += 1;"), - "diff hunk removed line should use tab-normalized indent. Response:\n{}", - result.response_text - ); - assert!( - result.response_text.contains("+\t\tthis.value += 2;"), - "diff hunk added line should use tab-normalized indent. Response:\n{}", - result.response_text - ); - // Should NOT contain the raw 8-space-indented diff lines. - assert!( - !result.response_text.contains("- this.value += 1;"), - "should not have raw space-indented diff lines. Response:\n{}", - result.response_text - ); - } - - #[test] - fn whole_chunk_replace_does_not_duplicate_attributes() { - let source = "#[napi]\nfn close() {\n old();\n}\n"; - let state = state_for(source, "rust"); - let chunk = state.inner().chunk("fn_clo").expect("fn_clo"); - // The chunk range should include the #[napi] attribute. - assert_eq!(chunk.start_line, 1, "chunk should start at the attribute line"); - - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Put, - sel: Some("fn_clo".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("/// doc\n#[napi]\nfn close() {\n new();\n}".to_owned()), - find: None, - }); - - let occurrences = result.diff_after.matches("#[napi]").count(); - assert_eq!( - occurrences, 1, - "expected exactly one #[napi] attribute, got {occurrences}. Full text:\n{}", - result.diff_after - ); - assert!(result.diff_after.contains("new()"), "replacement body should be present"); - } - - #[test] - fn edit_auto_resolves_unique_chunk_paths() { - let source = "class Worker {\n\trun(): void {\n\t\tconsole.log(this.name);\n\t}\n}\n"; - let state = state_for(source, "typescript"); - let chunk = state - .inner() - .chunk("cls_Wor.fn_run") - .expect("cls_Wor.fn_run should exist"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some("run".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("run(): void {\n\tconsole.log(\"resolved\");\n}".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.ts".to_owned(), - normalize_indent: None, - }) - .expect("edit should resolve a unique fuzzy selector"); - - assert!( - result.diff_after.contains("console.log(\"resolved\");"), - "expected updated body text, got {:?}", - result.diff_after - ); - assert!( - result - .warnings - .iter() - .any(|warning| warning - .contains("Auto-resolved chunk selector \"run\" to \"cls_Wor.fn_run#")), - "expected auto-resolution warning, got {:?}", - result.warnings - ); - } - - #[test] - fn edit_batch_uses_initial_checksum_identity_after_same_file_renumbering() { - let source = "\ -const transportAlpha = 1; - -const transportBeta = 2; - -const transportGamma = 3; -"; - let state = state_for(source, "typescript"); - let first = state.inner().chunk("var_tra_1").expect("var_tra_1").clone(); - let second = state.inner().chunk("var_tra_2").expect("var_tra_2").clone(); - - let result = apply_edits( - &state, - &edit_params(vec![ - EditOperation { - op: ChunkEditOp::Delete, - sel: Some("var_tra_1".to_owned()), - crc: Some(first.checksum), - region: None, - content: None, - find: None, - }, - EditOperation { - op: ChunkEditOp::Put, - sel: Some("var_tra_2".to_owned()), - crc: Some(second.checksum), - region: None, - content: Some("const transportBeta = 20;".to_owned()), - find: None, - }, - ]), - ) - .expect("same-file batch should keep the original logical target"); - - assert!( - !result.diff_after.contains("transportAlpha"), - "first sibling should be deleted:\n{}", - result.diff_after - ); - assert!( - result.diff_after.contains("const transportBeta = 20;"), - "renumbered second sibling should be updated:\n{}", - result.diff_after - ); - assert!( - result.diff_after.contains("const transportGamma = 3;"), - "third sibling should not receive the second operation:\n{}", - result.diff_after - ); - assert!( - result - .warnings - .iter() - .all(|warning| !warning.contains("Auto-resolved stale selector")), - "same-batch renumbering should not emit stale-selector warnings: {:?}", - result.warnings - ); - } - - #[test] - fn edit_batch_rejects_bad_initial_checksum_before_applying_operations() { - let source = "\ -const transportAlpha = 1; - -const transportBeta = 2; -"; - let state = state_for(source, "typescript"); - let first = state.inner().chunk("var_tra_1").expect("var_tra_1").clone(); - - let err = apply_edits( - &state, - &edit_params(vec![ - EditOperation { - op: ChunkEditOp::Delete, - sel: Some("var_tra_1".to_owned()), - crc: Some(first.checksum), - region: None, - content: None, - find: None, - }, - EditOperation { - op: ChunkEditOp::Put, - sel: Some("var_tra_2".to_owned()), - crc: Some("0000".to_owned()), - region: None, - content: Some("const transportBeta = 20;".to_owned()), - find: None, - }, - ]), - ) - .err() - .expect("bad initial checksum should reject the batch"); - - assert!( - err.contains("Edit operation 2/2 failed during initial checksum validation"), - "error should identify initial validation failure: {err}" - ); - assert!( - err.contains("Checksum mismatch for var_tra_2"), - "error should report the stale checksum target: {err}" - ); - assert_eq!(state.source(), source, "original state should remain unchanged"); - } - - #[test] - fn edit_batch_failure_after_recompute_is_atomic_for_original_state() { - let source = "\ -const transportAlpha = 1; - -const transportBeta = 2; -"; - let state = state_for(source, "typescript"); - let first = state.inner().chunk("var_tra_1").expect("var_tra_1").clone(); - let second = state.inner().chunk("var_tra_2").expect("var_tra_2").clone(); - - let err = apply_edits( - &state, - &edit_params(vec![ - EditOperation { - op: ChunkEditOp::Delete, - sel: Some("var_tra_1".to_owned()), - crc: Some(first.checksum), - region: None, - content: None, - find: None, - }, - EditOperation { - op: ChunkEditOp::Replace, - sel: Some("var_tra_2".to_owned()), - crc: Some(second.checksum), - region: None, - content: Some("const transportBeta = 20;".to_owned()), - find: Some("not in current chunk".to_owned()), - }, - ]), - ) - .err() - .expect("second operation should fail after recompute"); - - assert!( - err.contains("Edit operation 2/2 failed"), - "error should identify the failing operation: {err}" - ); - assert!( - err.contains("'find' text not found inside chunk"), - "error should preserve the real operation failure: {err}" - ); - assert_eq!(state.source(), source, "original state should remain unchanged"); - } - - #[test] - fn edit_batch_failure_context_uses_original_snapshot_after_partial_apply() { - let source = "\ -const alphaValue = 1; - -const betaValue = 2; -"; - let state = state_for(source, "typescript"); - let first = state - .inner() - .tree() - .root_children - .first() - .and_then(|path| state.inner().chunk(path)) - .expect("first top-level chunk") - .clone(); - let first_path = first.path.clone(); - let first_checksum = first.checksum; - - let err = apply_edits( - &state, - &edit_params(vec![ - EditOperation { - op: ChunkEditOp::Put, - sel: Some(first_path.clone()), - crc: Some(first_checksum.clone()), - region: None, - content: Some("const alphaValue = 10;".to_owned()), - find: None, - }, - EditOperation { - op: ChunkEditOp::Replace, - sel: Some(first_path), - crc: Some(first_checksum), - region: None, - content: Some("replacement".to_owned()), - find: Some("not in current file".to_owned()), - }, - ]), - ) - .err() - .expect("second operation should fail after the first mutates in-memory state"); - - assert!( - err.contains("No changes were saved"), - "error should preserve the transactional message: {err}" - ); - assert!( - err.contains("const alphaValue = 1;"), - "current content should show the original saved snapshot: {err}" - ); - assert!( - !err.contains("const alphaValue = 10;"), - "current content must not show partial in-memory edits: {err}" - ); - } - - #[test] - fn edit_batch_maps_reused_checksum_after_target_changes() { - let source = "\ -function target(): void { -\tconsole.log(\"old\"); -} - -function helper(): void { -\tconsole.log(\"helper\"); -} -"; - let state = state_for(source, "typescript"); - let target = state.inner().chunk("fn_tar").expect("fn_tar").clone(); - - let result = apply_edits( - &state, - &edit_params(vec![ - EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("#{}", target.checksum)), - crc: None, - region: None, - content: Some("function target(): void {\n\tconsole.log(\"middle\");\n}".to_owned()), - find: None, - }, - EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("#{}", target.checksum)), - crc: None, - region: None, - content: Some("function target(): void {\n\tconsole.log(\"final\");\n}".to_owned()), - find: None, - }, - ]), - ) - .expect("same-batch reused checksum should follow the changed target"); - - assert!( - result.diff_after.contains("console.log(\"final\")"), - "second edit should apply to the original target:\n{}", - result.diff_after - ); - assert!( - !result.diff_after.contains("console.log(\"middle\")"), - "second edit should replace the first edit on the same target:\n{}", - result.diff_after - ); - assert!( - result.diff_after.contains("function helper(): void"), - "reused checksum must not fall back to root replacement:\n{}", - result.diff_after - ); - } - - #[test] - fn edit_accepts_multi_crc_inline_segments() { - let source = "class Worker {\n\trun(): void {\n\t\tconsole.log(this.name);\n\t}\n}\n"; - let state = state_for(source, "typescript"); - let class_chunk = state.inner().chunk("cls_Wor").expect("cls_Wor").clone(); - let method_chunk = state - .inner() - .chunk("cls_Wor.fn_run") - .expect("fn_run") - .clone(); - - // Both CRCs present and valid – selector carries `#CRC` on every segment. - let sel = format!( - "cls_Wor#{parent}.fn_run#{leaf}", - parent = class_chunk.checksum, - leaf = method_chunk.checksum, - ); - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some(sel), - crc: None, - region: None, - content: Some("run(): void {\n\tconsole.log(\"multi-crc\");\n}".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.ts".to_owned(), - normalize_indent: None, - }) - .expect("multi-crc selector with all fresh CRCs should edit successfully"); - assert!(result.diff_after.contains("console.log(\"multi-crc\");"), "{}", result.diff_after); - } - - #[test] - fn edit_accepts_multi_crc_selector_when_only_ancestor_crc_matches() { - let source = "class Worker {\n\trun(): void {\n\t\tconsole.log(this.name);\n\t}\n}\n"; - let state = state_for(source, "typescript"); - let class_chunk = state.inner().chunk("cls_Wor").expect("cls_Wor").clone(); - - // Stale leaf CRC but fresh ancestor CRC. Under the new lenient rule, - // at least one provided CRC matches the ancestor chain, so the edit - // proceeds. - let sel = format!("cls_Wor#{parent}.fn_run#eaea", parent = class_chunk.checksum); - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some(sel), - crc: None, - region: None, - content: Some("run(): void {\n\tconsole.log(\"any-match\");\n}".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.ts".to_owned(), - normalize_indent: None, - }) - .expect("edit should accept a multi-crc path when any crc matches an ancestor"); - assert!(result.diff_after.contains("console.log(\"any-match\");"), "{}", result.diff_after); - } - - #[test] - fn edit_rejects_multi_crc_selector_when_all_crcs_stale() { - let source = "class Worker {\n\trun(): void {\n\t\tconsole.log(this.name);\n\t}\n}\n"; - let state = state_for(source, "typescript"); - - let err = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some("cls_Wor#eaea.fn_run#nene".to_owned()), - crc: None, - region: None, - content: Some("run(): void {\n\tconsole.log(\"should-not-apply\");\n}".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.ts".to_owned(), - normalize_indent: None, - }) - .err() - .expect("all-stale multi-crc selector should fail"); - assert!(err.contains("None of the provided checksums"), "{err}"); - } - - #[test] - fn edit_auto_resolves_prefixed_function_names() { - let source = "function fuzzyMatch(): void {\n\tconsole.log(\"old\");\n}\n"; - let state = state_for(source, "typescript"); - let chunk = state.inner().chunk("fn_fuz").expect("fn_fuz should exist"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some("fuzzyM".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some( - "function fuzzyMatch(): void {\n\tconsole.log(\"resolved\");\n}".to_owned(), - ), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "box.ts".to_owned(), - normalize_indent: None, - }) - .expect("edit should resolve a prefixed bare selector"); - - assert!(result.diff_after.contains("console.log(\"resolved\");"), "{}", result.diff_after); - assert!(result.warnings.iter().any(|warning| { - warning.contains("Auto-resolved chunk selector \"fuzzyM\" to \"fn_fuz#") - })); - } - - #[test] - fn edit_accepts_file_prefixed_checksum_targets() { - let source = "function main(): void {\n\tconsole.log(\"old\");\n}\n"; - let state = state_for(source, "typescript"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai should exist"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some("box.ts".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("function main(): void {\n\tconsole.log(\"normalized\");\n}".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "box.ts".to_owned(), - normalize_indent: None, - }) - .expect("edit should resolve file-prefixed checksum target"); - - assert!(result.diff_after.contains("console.log(\"normalized\");"), "{}", result.diff_after); - } - - #[test] - fn line_number_selector_auto_resolves_to_containing_chunk() { - let source = "function main(): void {\n\tconsole.log(\"old\");\n}\n"; - let state = state_for(source, "typescript"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - // L2 falls inside fn_main — should auto-resolve and apply the edit. - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("L2#{}", chunk.checksum)), - crc: None, - region: None, - content: Some("function main(): void {\n\tconsole.log(\"new\");\n}".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "box.ts".to_owned(), - normalize_indent: None, - }); - - let result = result.expect("line-number selector should auto-resolve"); - assert!( - result - .warnings - .iter() - .any(|w| w.contains("Auto-resolved line target")), - "should warn about auto-resolution: {:?}", - result.warnings - ); - assert!( - result.diff_after.contains("\"new\""), - "edit should have applied: {}", - result.diff_after - ); - } - - #[test] - fn line_number_outside_any_chunk_returns_error() { - let source = "function main(): void {\n\tconsole.log(\"old\");\n}\n"; - let state = state_for(source, "typescript"); - - // L999 is way beyond the file — should fail. - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some("L999".to_owned()), - crc: None, - region: None, - content: Some("// hello".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "box.ts".to_owned(), - normalize_indent: None, - }); - - assert!(result.is_err(), "line outside any chunk should fail"); - let err = result.err().unwrap(); - assert!(err.contains("does not fall inside any chunk"), "{err}"); - } - - #[test] - fn markdown_section_replace_preserves_next_sibling_heading() { - let source = "# Top\n\n## Building\n\nOld content.\n\n## Code Style\n\n- style one\n"; - let state = state_for(source, "markdown"); - let chunk = state.inner().chunk("sct_Top.sct_Bui").expect("sct_Bui"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some("sct_Top.sct_Bui".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("## Building\n\nNew content.\n".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.md".to_owned(), - normalize_indent: None, - }) - .expect("replace should succeed"); - - assert!( - result.diff_after.contains("## Code Style"), - "next sibling heading must survive section replace, got:\n{}", - result.diff_after, - ); - assert!( - result.diff_after.contains("New content."), - "replacement content must be present, got:\n{}", - result.diff_after, - ); - } - - #[test] - fn find_replace_single_match() { - let source = "fn main() {\n println!(\"hello\");\n println!(\"world\");\n}\n"; - let state = state_for(source, "rust"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Replace, - sel: Some("fn_mai".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("warn!(\"hello\")".to_owned()), - find: Some("println!(\"hello\")".to_owned()), - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.rs".to_owned(), - normalize_indent: None, - }) - .expect("edit should apply"); - - assert!( - result.diff_after.contains("warn!(\"hello\")"), - "expected replacement, got {:?}", - result.diff_after - ); - assert!( - result.diff_after.contains("println!(\"world\")"), - "non-matched line must survive, got {:?}", - result.diff_after - ); - } - - #[test] - fn find_replace_not_found() { - let source = "fn main() {\n println!(\"hello\");\n}\n"; - let state = state_for(source, "rust"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Replace, - sel: Some("fn_mai".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("replacement".to_owned()), - find: Some("nonexistent text".to_owned()), - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.rs".to_owned(), - normalize_indent: None, - }); - - assert!(result.is_err(), "expected error for not-found find text"); - assert!( - result - .err() - .expect("err") - .contains("not found inside chunk"), - "error should mention not found" - ); - } - - #[test] - fn find_replace_ambiguous() { - let source = "fn main() {\n let a = 1;\n let b = 1;\n let c = 1;\n}\n"; - let state = state_for(source, "rust"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Replace, - sel: Some("fn_mai".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("2".to_owned()), - find: Some("= 1".to_owned()), - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.rs".to_owned(), - normalize_indent: None, - }); - - assert!(result.is_err(), "expected error for ambiguous find text"); - let err = result.err().expect("err"); - assert!(err.contains("ambiguous"), "error should mention ambiguous: {err}"); - assert!(err.contains("3 matches"), "error should report count: {err}"); - } - - #[test] - fn find_replace_empty_find_rejected() { - let source = "fn main() {\n println!(\"hello\");\n}\n"; - let state = state_for(source, "rust"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Replace, - sel: Some("fn_mai".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("replacement".to_owned()), - find: Some(String::new()), - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.rs".to_owned(), - normalize_indent: None, - }); - - assert!(result.is_err(), "expected error for empty find text"); - assert!(result.err().expect("err").contains("cannot be empty"), "error should mention empty"); - } - - #[test] - fn find_replace_respects_chunk_bounds() { - // 'hello' appears in fn_greet but NOT in fn_main. Searching fn_main should - // fail. - let source = "fn greet() {\n println!(\"hello\");\n}\n\nfn main() {\n greet();\n}\n"; - let state = state_for(source, "rust"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Replace, - sel: Some("fn_mai".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("goodbye".to_owned()), - find: Some("hello".to_owned()), - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.rs".to_owned(), - normalize_indent: None, - }); - - assert!(result.is_err(), "find outside target chunk should fail"); - assert!( - result - .err() - .expect("err") - .contains("not found inside chunk") - ); - } - - #[test] - fn find_replace_preserves_multiline_replacement_indentation() { - let source = "enum Status {\n\tRunning,\n\tStopped,\n}\n"; - let state = state_for(source, "rust"); - let chunk = state.inner().chunk("en_Sta").expect("en_Sta"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Replace, - sel: Some("en_Sta".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("\tPaused,\n\tStopped,\n\tFailed,".to_owned()), - find: Some("Stopped,".to_owned()), - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.rs".to_owned(), - normalize_indent: None, - }) - .expect("edit should apply"); - - assert_eq!( - result.diff_after, - "enum Status {\n\tRunning,\n\tPaused,\n\tStopped,\n\tFailed,\n}\n" - ); - } - - #[test] - fn python_multiline_replace_dedents_user_base_indent_before_reindenting() { - let source = concat!( - "class Server:\n", - "\tdef outer(self):\n", - "\t\tdef inner():\n", - "\t\t\tresult = compute()\n", - "\t\t\treturn result\n", - "\t\treturn inner()\n", - ); - let state = state_for(source, "python"); - let chunk = state - .inner() - .chunk("cls_Ser.fn_out") - .expect("outer function"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Replace, - sel: Some("cls_Ser.fn_out".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - find: Some("\t\t\t\tresult = compute()\n\t\t\t\treturn result".to_owned()), - content: Some( - "\t\t\t\tvalue = compute()\n\t\t\t\tif value:\n\t\t\t\t\treturn value".to_owned(), - ), - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.py".to_owned(), - normalize_indent: None, - }) - .expect("replace should apply"); - - assert!( - result - .diff_after - .contains("\t\t\tvalue = compute()\n\t\t\tif value:\n\t\t\t\treturn value"), - "replacement should be reindented to the matched source base indent: {:?}", - result.diff_after - ); - assert!( - !result.diff_after.contains("\t\t\t\tvalue = compute()"), - "replacement must not keep the caller-supplied base indent and compound it: {:?}", - result.diff_after - ); - } - - #[test] - fn rust_enum_variant_replace_accepts_trailing_comma_boundary() { - let source = "enum LogLevel {\n Info,\n Warn,\n}\n"; - let state = state_for(source, "rust"); - let variant = state - .inner() - .chunk("en_Log.vr_War") - .expect("Warn variant should exist"); - - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Replace, - sel: Some(format!("{}#{}", variant.path, variant.checksum)), - crc: None, - region: None, - content: Some("Warning,".to_owned()), - find: Some("Warn,".to_owned()), - }); - - assert!( - result.diff_after.contains(" Warning,\n"), - "replace should see the same trailing comma boundary as write:\n{}", - result.diff_after - ); - } - - #[test] - fn focus_emits_only_changed_chain() { - let source = "const a = 1;\n\nconst b = 2;\n\nconst c = 3;\n\nconst d = 4;\n\nconst e = 5;\n"; - let state = state_for(source, "typescript"); - let chunk = state.inner().chunk("var_c").expect("var_c"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some("var_c".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("const c = 33;".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: Some(ChunkAnchorStyle::Full), - cwd: ".".to_owned(), - file_path: "test.ts".to_owned(), - normalize_indent: None, - }) - .expect("edit should apply"); - - // The focused edit view should show only the changed chunk chain, not - // sibling context blocks. - let response = &result.response_text; - assert!(response.contains("var_c"), "touched chunk should appear: {response}"); - assert!( - response.contains("*@var_c#"), - "changed chunk should be marked in the gutter: {response}" - ); - assert!( - !response.contains("var_b"), - "prev sibling should not appear in the focused edit view: {response}" - ); - assert!( - !response.contains("var_d"), - "next sibling should not appear in the focused edit view: {response}" - ); - assert!( - !response.contains("const a"), - "distant chunk var_a body should not appear: {response}" - ); - assert!( - !response.contains("const e"), - "distant chunk var_e body should not appear: {response}" - ); - } - - #[test] - fn error_context_expands_around_parse_failure_in_other_chunk() { - // Regression: when an edit introduces a parse error in a chunk other - // than the edit target, the rejected post-edit preview must - // expand around the error location, not only around the targeted chunk. - // Previously the focus only covered the targeted chunk, so the parse - // error could land inside a truncated region and force the agent to - // do a follow-up read to diagnose the failure. - // - // Construct a file with five top-level functions. Edit fn_a's body with - // unbalanced-brace content so the parser bleeds the error into fn_c's - // territory. After the fix, the error message must include identifying - // text from the chunks flagged with parse errors, even though only - // fn_a was the edit target. - let source = concat!( - "fn alpha() {\n let x = 1;\n}\n\n", - "fn bravo() {\n let y = 2;\n}\n\n", - "fn charlie() {\n let z = 3;\n}\n\n", - "fn delta() {\n let w = 4;\n}\n\n", - "fn echo() {\n let v = 5;\n}\n", - ); - let state = parsed_state_for(source, "rust"); - let alpha = state.inner().chunk("fn_alp").expect("fn_alp"); - - let Err(err) = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some("fn_alp".to_owned()), - crc: Some(alpha.checksum.clone()), - region: Some(ChunkRegion::Body), - // Intentionally broken: dangling `{` consumes subsequent - // top-level functions until tree-sitter gives up. - content: Some("let broken = { { {\n".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: Some(ChunkAnchorStyle::Full), - cwd: ".".to_owned(), - file_path: "test.rs".to_owned(), - normalize_indent: None, - }) else { - panic!("edit should be rejected due to parse error"); - }; - - assert!( - err.contains("Edit rejected: introduced"), - "should be a parse-error rejection: {err}", - ); - assert!( - err.contains("Hypothetical post-edit content (edit rejected; file unchanged):"), - "should include rejected post-edit context: {err}" - ); - // The edit target must always be visible. - assert!(err.contains("fn_alp"), "target chunk should be in focus: {err}"); - // The fix: at least one of the downstream chunks where the parse error - // lands should also appear in the focused rejected preview. Without - // the fix, focus only covers fn_alpha and these downstream anchors are - // skipped, so the agent cannot see the failure location. - let downstream_visible = ["fn_bra", "fn_cha", "fn_del", "fn_ech"] - .iter() - .any(|name| err.contains(name)); - assert!( - downstream_visible, - "error-context focus should include at least one downstream chunk where the parse \ - failure lands, got: {err}", - ); - } - - #[test] - fn append_on_root_stmts_group_inserts_after_grouped_statements() { - // This fixture exposes the root-level `stmts` group. Appending to that - // group should place content after the grouped top-level statements. - // Use plain expression statements (no trailing callback) so they stay - // as groupable stmts rather than being promoted to named expr chunks. - let source = "import { foo } from \"bar\";\n\nconsole.log(\"a\");\nconsole.log(\"b\");\n"; - let state = parsed_state_for(source, "typescript"); - let stmts = state.inner().chunk("st").expect("st chunk should exist"); - assert!(stmts.group, "stmts chunk should be marked as group"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Append, - sel: Some(stmts.path.clone()), - crc: None, - region: None, - content: Some("\nconsole.log(\"c\");".to_owned()), - find: None, - }); - - assert!( - result.diff_after.contains("log(\"c\""), - "appended content should appear in output, got: {}", - result.diff_after - ); - assert!( - result - .diff_after - .contains("log(\"b\");\nconsole.log(\"c\");"), - "appended statement should land after the grouped top-level statements, got: {}", - result.diff_after - ); - } - - #[test] - fn replace_body_preserves_typescript_closing_brace_indentation() { - let source = "function main() {\n work();\n}\n"; - let state = state_for(source, "typescript"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("fn_mai#{}~", chunk.checksum)), - crc: None, - region: None, - content: Some("\treturn next();\n".to_owned()), - find: None, - }); - - assert_eq!(result.diff_after, "function main() {\n return next();\n}\n"); - } - - #[test] - fn replace_body_preserves_rust_closing_brace_indentation() { - let source = "fn main() {\n println!(\"old\");\n}\n"; - let state = state_for(source, "rust"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("fn_mai#{}~", chunk.checksum)), - crc: None, - region: None, - content: Some("\tprintln!(\"new\");\n".to_owned()), - find: None, - }); - - assert_eq!(result.diff_after, "fn main() {\n println!(\"new\");\n}\n"); - } - - #[test] - fn replace_body_preserves_go_closing_brace_indentation() { - let source = "func main() {\n work()\n}\n"; - let state = state_for(source, "go"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - let result = apply_single_edit(&state, "test.go", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("fn_mai#{}~", chunk.checksum)), - crc: None, - region: None, - content: Some("\treturn\n".to_owned()), - find: None, - }); - - assert_eq!(result.diff_after, "func main() {\n return\n}\n"); - } - - #[test] - fn three_space_body_replace_denormalizes_tabs_back_to_file_style() { - let source = "def run():\n return 1\n"; - let state = state_for(source, "python"); - let chunk = state.inner().chunk("fn_run").expect("fn_run"); - - let result = apply_single_edit(&state, "test.py", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("fn_run#{}~", chunk.checksum)), - crc: None, - region: None, - content: Some("\treturn 2\n".to_owned()), - find: None, - }); - - assert_eq!(result.diff_after, "def run():\n return 2\n"); - } - - #[test] - fn after_targets_chunk_directly_for_top_level_sibling_insertion() { - let source = "function alpha(): void {\n\twork();\n}\n"; - let state = state_for(source, "typescript"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::After, - sel: Some("fn_alp".to_owned()), - crc: None, - region: None, - content: Some("function beta(): void {\n\twork();\n}\n".to_owned()), - find: None, - }); - - assert!(result.diff_after.contains("function alpha(): void"), "{}", result.diff_after); - assert!(result.diff_after.contains("function beta(): void"), "{}", result.diff_after); - assert!( - result - .diff_after - .find("function alpha(): void") - .expect("alpha") - < result - .diff_after - .find("function beta(): void") - .expect("beta") - ); - } - - #[test] - - fn go_body_and_container_append_are_not_interchangeable() { - let source = "package main\n\ntype Server struct {\n Addr string\n}\n\nfunc (s *Server) \ - Start() {\n work()\n}\n"; - - let body_state = state_for(source, "go"); - let body_result = apply_single_edit(&body_state, "test.go", EditOperation { - op: ChunkEditOp::Append, - sel: Some("ty_Ser~".to_owned()), - crc: None, - region: None, - content: Some("\tPort int\n".to_owned()), - find: None, - }); - assert!( - body_result - .diff_after - .contains("Addr string\n Port int\n}"), - "{}", - body_result.diff_after - ); - assert!( - !body_result.diff_after.contains("func (s *Server) Port"), - "{}", - body_result.diff_after - ); - - let container_state = state_for(source, "go"); - let container_result = apply_single_edit(&container_state, "test.go", EditOperation { - op: ChunkEditOp::Append, - sel: Some("ty_Ser".to_owned()), - crc: None, - region: None, - content: Some("func (s *Server) Stop() {\n\twork()\n}\n".to_owned()), - find: None, - }); - assert!( - container_result - .diff_after - .contains("func (s *Server) Start()"), - "{}", - container_result.diff_after - ); - assert!( - container_result - .diff_after - .contains("func (s *Server) Stop()"), - "{}", - container_result.diff_after - ); - assert!( - container_result - .diff_after - .find("func (s *Server) Stop()") - .expect("stop") - < container_result - .diff_after - .find("func (s *Server) Start()") - .expect("start") - ); - } - - #[test] - fn go_type_container_append_after_receiver_methods_preserves_sibling_spacing() { - let source = "package main\n\ntype Server struct {}\n\nfunc (s *Server) Start() {}\nfunc (s \ - *Server) Stop() {}\n"; - let state = state_for(source, "go"); - - let result = apply_single_edit(&state, "test.go", EditOperation { - op: ChunkEditOp::Append, - sel: Some("ty_Ser".to_owned()), - crc: None, - region: None, - content: Some("func (s *Server) Restart() {}".to_owned()), - find: None, - }); - - assert!( - result - .diff_after - .contains("type Server struct {}\nfunc (s *Server) Restart() {}"), - "{}", - result.diff_after - ); - } - - #[test] - fn packed_toml_table_after_inserts_stay_tightly_packed() { - let source = "[dependencies]\nanyhow.workspace = true\nbytes.workspace = \ - true\nserde.workspace = true\nsolar-interface.workspace = \ - true\nparking_lot.workspace = true\nsolar-sema.workspace = \ - true\ntokio.workspace = true\ntracing.workspace = true\n"; - let state = state_for(source, "toml"); - - let result = apply_single_edit(&state, "Cargo.toml", EditOperation { - op: ChunkEditOp::After, - sel: Some("tbl_dep.key_par".to_owned()), - crc: None, - region: None, - content: Some("rayon.workspace = true\n".to_owned()), - find: None, - }); - - assert!( - result.diff_after.contains( - "parking_lot.workspace = true\nrayon.workspace = true\nsolar-sema.workspace = true" - ), - "{}", - result.diff_after - ); - assert!( - !result - .diff_after - .contains("parking_lot.workspace = true\n\nrayon.workspace = true"), - "{}", - result.diff_after - ); - assert!( - !result - .diff_after - .contains("rayon.workspace = true\n\nsolar-sema.workspace = true"), - "{}", - result.diff_after - ); - } - - #[test] - fn packed_top_level_typescript_variables_stay_tightly_packed() { - let source = "const a = 1;\nconst b = 2;\nconst c = 3;\n"; - let state = state_for(source, "typescript"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::After, - sel: Some("var_b".to_owned()), - crc: None, - region: None, - content: Some("const bb = 22;\n".to_owned()), - find: None, - }); - - assert!( - result - .diff_after - .contains("const b = 2;\nconst bb = 22;\nconst c = 3;"), - "{}", - result.diff_after - ); - assert!( - !result.diff_after.contains("const b = 2;\n\nconst bb = 22;"), - "{}", - result.diff_after - ); - assert!( - !result.diff_after.contains("const bb = 22;\n\nconst c = 3;"), - "{}", - result.diff_after - ); - } - - #[test] - fn crc_mismatch_error_includes_fresh_chunk_context() { - let source = "class Foo {\n bar() {\n return 1;\n }\n}\n"; - let state = state_for(source, "typescript"); - - let result = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some("cls_Foo.fn_bar#eaea".to_owned()), - crc: None, - region: None, - content: Some("baz() { return 2; }".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: Some(ChunkAnchorStyle::Full), - cwd: ".".to_owned(), - file_path: "test.ts".to_owned(), - normalize_indent: None, - }); - let err = result.err().expect("should fail with stale CRC"); - - assert!( - err.contains("Current content (file unchanged):"), - "error should include current content: {err}" - ); - assert!(err.contains("fn_bar"), "error should show the chunk with fresh anchor: {err}"); - assert!(err.contains("cls_Foo"), "error should show ancestor context: {err}"); - } - - #[test] - fn prologue_replace_preserves_newline_before_body() { - let source = "/// Old doc.\nfn main() {\n work();\n}\n"; - let state = state_for(source, "rust"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("fn_mai#{}^", chunk.checksum)), - crc: None, - region: None, - content: Some("/// New doc.\nfn main() {".to_owned()), - find: None, - }); - - // The body should NOT be joined onto the prologue line. - assert!( - !result.diff_after.contains("{ work"), - "prologue replace should not join body onto same line: {}", - result.diff_after - ); - assert_eq!(result.diff_after, "/// New doc.\nfn main() {\n work();\n}\n",); - } - - #[test] - fn markdown_table_pipes_preserved_in_replace() { - let new_table = "| Header A | Header B |\n| --- | --- |\n| cell A | cell B |\n"; - - // Simulate what normalize_inserted_content does to table content. - let result = super::normalize_inserted_content(new_table, "", None, ' ', true); - - assert!(result.contains("| Header A"), "table pipes should not be stripped: {result}"); - } - - #[test] - fn container_prepend_creates_addressable_chunk() { - // Prepending without a @region inserts before the chunk. After tree rebuild, - // the inserted content should be addressable (either absorbed as trivia - // or as a new preamble/chunk), not orphaned. - let source = "const a = 1;\n\nstruct Config {\n host: String,\n}\n"; - let state = state_for(source, "rust"); - - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Prepend, - sel: Some("stc_Con".to_owned()), - crc: None, - region: None, - content: Some("// Config documentation\n".to_owned()), - find: None, - }); - - // The comment should exist in the output. - assert!( - result.diff_after.contains("// Config documentation"), - "prepended content should be in the file: {}", - result.diff_after - ); - - // Re-parse and check that every non-empty line is covered by some chunk. - let new_state = state_for(&result.diff_after, "rust"); - let tree = new_state.inner().tree(); - let lines: Vec<&str> = result.diff_after.split('\n').collect(); - for (i, line) in lines.iter().enumerate() { - if line.trim().is_empty() { - continue; - } - let line_num = (i + 1) as u32; - let covered = tree - .chunks - .iter() - .any(|c| !c.path.is_empty() && c.start_line <= line_num && c.end_line >= line_num); - assert!( - covered, - "line {} ({:?}) should be covered by a chunk, but isn't. Chunks: {:?}", - line_num, - line, - tree - .chunks - .iter() - .filter(|c| !c.path.is_empty()) - .map(|c| format!("{}:L{}-L{}", c.path, c.start_line, c.end_line)) - .collect::>() - ); - } - } - - #[test] - fn leaf_chunk_supports_body_region_read() { - // A small method (under LEAF_THRESHOLD) should still have region boundaries - // if it has a body delimiter. - let source = "class Foo {\n bar() {\n return 1;\n }\n}\n"; - let state = state_for(source, "typescript"); - let fn_bar = state.inner().chunk("cls_Foo.fn_bar").expect("fn_bar"); - assert!(fn_bar.leaf, "fn_bar should be a leaf chunk"); - assert!(fn_bar.prologue_end_byte.is_some(), "leaf fn_bar should have prologue_end_byte set"); - assert!( - fn_bar.epilogue_start_byte.is_some(), - "leaf fn_bar should have epilogue_start_byte set" - ); - } - - #[test] - fn nested_body_replace_preserves_correct_indentation_4space() { - // 4-space file: method body at 2 levels of indent. - let source = "class Server {\n start() {\n work();\n }\n}\n"; - let state = state_for(source, "typescript"); - let chunk = state.inner().chunk("cls_Ser.fn_sta").expect("fn_sta"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("cls_Ser.fn_sta#{}~", chunk.checksum)), - crc: None, - region: None, - content: Some("\treturn 42;\n".to_owned()), - find: None, - }); - - assert_eq!( - result.diff_after, "class Server {\n start() {\n return 42;\n }\n}\n", - "4-space: nested body replace should produce correct 2-level indent" - ); - } - - #[test] - fn nested_body_replace_preserves_correct_indentation_2space() { - // 2-space file: method body at 2 levels of indent. - let source = "class Server {\n start() {\n work();\n }\n}\n"; - let state = state_for(source, "typescript"); - let chunk = state.inner().chunk("cls_Ser.fn_sta").expect("fn_sta"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("cls_Ser.fn_sta#{}~", chunk.checksum)), - crc: None, - region: None, - content: Some("\treturn 42;\n".to_owned()), - find: None, - }); - - assert_eq!( - result.diff_after, "class Server {\n start() {\n return 42;\n }\n}\n", - "2-space: nested body replace should produce correct 2-level indent" - ); - } - - #[test] - fn nested_body_replace_with_excess_tabs_corrected() { - // Agent accidentally includes base padding (2 tabs instead of 1). - // Correction mechanism should strip common indent and produce correct output. - let source = "class Server {\n start() {\n work();\n }\n}\n"; - let state = state_for(source, "typescript"); - let chunk = state.inner().chunk("cls_Ser.fn_sta").expect("fn_sta"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("cls_Ser.fn_sta#{}~", chunk.checksum)), - crc: None, - region: None, - content: Some("\t\tif (x) {\n\t\t\ty();\n\t\t}\n".to_owned()), - find: None, - }); - - assert_eq!( - result.diff_after, - "class Server {\n start() {\n if (x) {\n y();\n }\n }\n}\n", - "2-space: excess tabs should be corrected via dedent" - ); - } - - #[test] - fn body_append_inserts_inside_class() { - // Appending to ~ of a class should insert inside the body, - // not after the closing brace. - let source = "class Foo {\n bar() {\n return 1;\n }\n}\n"; - let state = state_for(source, "typescript"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Append, - sel: Some("cls_Foo~".to_owned()), - crc: None, - region: None, - content: Some("baz() {\n\treturn 2;\n}\n".to_owned()), - find: None, - }); - - assert!( - result.diff_after.contains("baz()"), - "appended method should appear: {}", - result.diff_after - ); - // baz should appear BEFORE the final closing brace - let baz_pos = result.diff_after.find("baz()").unwrap(); - let last_brace = result.diff_after.rfind('}').unwrap(); - assert!( - baz_pos < last_brace, - "baz() at {baz_pos} should be before last '}}' at {last_brace}: {}", - result.diff_after - ); - } - - #[test] - fn container_root_append_warns_that_insert_lands_outside_container() { - let source = "class Foo {\n bar() {\n return 1;\n }\n}\n"; - let state = state_for(source, "typescript"); - let class_chunk = state.inner().chunk("cls_Foo").expect("cls_Foo"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Append, - sel: Some(format!("cls_Foo#{}", class_chunk.checksum)), - crc: None, - region: None, - content: Some("\nfunction outside() {}\n".to_owned()), - find: None, - }); - - assert!( - result - .warnings - .iter() - .any(|warning| warning.contains("without `~` inserts after the chunk")), - "container-root append should warn about outside insertion: {:?}", - result.warnings - ); - assert!( - result.diff_after.ends_with("}\nfunction outside() {}"), - "append without ~ should still preserve existing outside semantics:\n{}", - result.diff_after - ); - } - - #[test] - fn body_prepend_inserts_after_opening_brace() { - // Prepending to ~ of an enum should insert after the opening brace, - // not before doc comments. - let source = "/** My enum. */\nenum Color {\n Red,\n Green,\n Blue,\n}\n"; - let state = state_for(source, "typescript"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Prepend, - sel: Some("en_Col~".to_owned()), - crc: None, - region: None, - content: Some("White,\n".to_owned()), - find: None, - }); - - assert!( - result.diff_after.contains("White"), - "prepended variant should appear: {}", - result.diff_after - ); - // White should appear AFTER the opening brace, before Red - let white_pos = result.diff_after.find("White").unwrap(); - let red_pos = result.diff_after.find("Red").unwrap(); - let doc_pos = result.diff_after.find("/** My enum.").unwrap(); - assert!(white_pos > doc_pos, "White should be after doc comment: {}", result.diff_after); - assert!(white_pos < red_pos, "White should be before Red: {}", result.diff_after); - } - - #[test] - fn successful_edit_response_contains_fresh_chunk_id() { - let source = "const count = 1;\n"; - let state = state_for(source, "typescript"); - let chunk_path = state - .inner() - .tree - .root_children - .first() - .expect("root child should exist") - .clone(); - let chunk = state.inner().chunk(&chunk_path).expect("root child chunk"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Put, - sel: Some(chunk_path.clone()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("const count = 2;".to_owned()), - find: None, - }); - - let fresh = result - .state - .inner() - .chunk(&chunk_path) - .expect("edited chunk should still exist"); - assert_ne!(fresh.checksum, chunk.checksum, "checksum should change after edit"); - assert!( - result - .response_text - .contains(format!("{chunk_path}#{}", fresh.checksum).as_str()), - "edit response should include the fresh chunk ID. Response:\n{}", - result.response_text - ); - } - - #[test] - fn rust_enum_body_write_preserves_following_impl_block() { - let source = concat!( - "struct Server;\n", - "\n", - "enum LogLevel {\n", - " Info,\n", - " Warn,\n", - "}\n", - "\n", - "impl Server {\n", - " fn start(&self) {\n", - " println!(\"start\");\n", - " }\n", - "}\n", - ); - let state = state_for(source, "rust"); - let enum_chunk = state.inner().chunk("en_Log").expect("en_Log"); - - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("en_Log#{}~", enum_chunk.checksum)), - crc: None, - region: None, - content: Some("Debug,\nInfo,\nWarn,\nError,\n".to_owned()), - find: None, - }); - - assert!( - result.diff_after.contains(" Error,\n"), - "enum body write should add the new variant:\n{}", - result.diff_after - ); - assert!( - result.diff_after.contains("impl Server"), - "enum body write must not delete the following impl block:\n{}", - result.diff_after - ); - assert!( - result.diff_after.contains("fn start(&self)"), - "enum body write must preserve methods inside the following impl block:\n{}", - result.diff_after - ); - } - - #[test] - fn markdown_list_replace_preserves_trailing_blank_line() { - let source = "# Title\n\n- item 1\n- item 2\n\n## Next\n"; - let state = state_for(source, "markdown"); - let list = state - .inner() - .tree - .chunks - .iter() - .find(|c| { - c.path - .rsplit('.') - .next() - .is_some_and(|leaf| leaf.starts_with("list")) - }) - .expect("list chunk"); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("{}#{}", list.path, list.checksum)), - crc: None, - region: None, - content: Some("- new 1\n- new 2\n".to_owned()), - find: None, - }); - - // The blank line between the list and ## Next must be preserved. - assert!( - result.diff_after.contains("- new 2\n\n## Next"), - "blank line between list and heading should be preserved: {:?}", - result.diff_after - ); - } - - #[test] - fn markdown_after_preserves_blank_line_before_next_section() { - let source = "# Title\n\n## Alpha\n\nalpha body\n\n## Beta\n\nbeta body\n"; - let state = state_for(source, "markdown"); - let section = state - .inner() - .chunk("sct_Tit.sct_Alp") - .expect("alpha section"); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::After, - sel: Some(format!("{}#{}", section.path, section.checksum)), - crc: None, - region: None, - content: Some("## Inserted\n\ninserted body\n".to_owned()), - find: None, - }); - - assert!( - result - .diff_after - .contains("## Inserted\n\ninserted body\n\n## Beta"), - "blank line between inserted section and next heading should be preserved: {:?}", - result.diff_after - ); - } - - #[test] - fn markdown_body_append_preserves_blank_line_before_next_section() { - let source = "# Title\n\n## Alpha\n\nalpha body\n\n## Beta\n\nbeta body\n"; - let state = state_for(source, "markdown"); - let section = state - .inner() - .chunk("sct_Tit.sct_Alp") - .expect("alpha section"); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::Append, - sel: Some(format!("{}#{}~", section.path, section.checksum)), - crc: None, - region: None, - content: Some("\nextra paragraph\n".to_owned()), - find: None, - }); - - assert!( - result - .diff_after - .contains("alpha body\n\n extra paragraph\n\n## Beta"), - "blank line between appended body content and next heading should be preserved: {:?}", - result.diff_after - ); - } - - #[test] - fn markdown_body_write_region_fallback_warns_before_whole_chunk_replace() { - let source = "# Title\n\n## Alpha\n\nalpha body\n\n## Beta\n\nbeta body\n"; - let state = state_for(source, "markdown"); - let section = state - .inner() - .chunk("sct_Tit.sct_Alp") - .expect("alpha section"); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("{}#{}~", section.path, section.checksum)), - crc: None, - region: None, - content: Some("## Alpha\n\nnew body\n".to_owned()), - find: None, - }); - - assert!( - result - .diff_after - .contains("## Alpha\n\nnew body\n\n## Beta"), - "whole-section replacement should preserve the next heading separator: {:?}", - result.diff_after - ); - assert!( - result - .warnings - .iter() - .any(|warning| warning.contains("fell back to whole-chunk editing")), - "markdown region fallback should warn: {:?}", - result.warnings - ); - } - - #[test] - fn markdown_section_region_fallback_warns_when_children_would_be_replaced() { - let source = "# Title\n\n## Embedded\n\n```python\ndef greet():\n return \"hi\"\n```\n"; - let state = state_for(source, "markdown"); - let section = state - .inner() - .chunk("sct_Tit.sct_Emb") - .expect("embedded section"); - assert!( - !section.children.is_empty(), - "fixture section should have child chunks: {:?}", - section.children - ); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("{}#{}^", section.path, section.checksum)), - crc: None, - region: None, - content: Some("## Embedded\n\nreplacement\n".to_owned()), - find: None, - }); - - assert!( - result - .warnings - .iter() - .any(|warning| warning.contains("has child chunks that will be replaced")), - "section fallback should warn when child chunks are being replaced: {:?}", - result.warnings - ); - assert!( - !result.diff_after.contains("def greet"), - "fallback whole-section replacement should still reflect current semantics:\n{}", - result.diff_after - ); - } - - #[test] - fn markdown_table_body_append_keeps_row_continuity() { - let source = "## Section\n\n| A |\n| --- |\n| one |\n\n## Next\n"; - let state = state_for(source, "markdown"); - let table = state - .inner() - .tree - .chunks - .iter() - .find(|chunk| chunk.start_line == 3 && chunk.end_line == 5) - .expect("table chunk"); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::Append, - sel: Some(format!("{}#{}~", table.path, table.checksum)), - crc: None, - region: None, - content: Some("| two |\n".to_owned()), - find: None, - }); - - assert!( - result.diff_after.contains("| one |\n| two |\n\n## Next"), - "appended table row should stay contiguous with the table and preserve next-section \ - spacing: {:?}", - result.diff_after - ); - assert!( - result - .warnings - .iter() - .any(|warning| warning.contains("fell back to whole-chunk editing")), - "table body-region append fallback should warn: {:?}", - result.warnings - ); - } - - #[test] - fn markdown_table_append_without_body_selector_keeps_row_continuity() { - let source = "## Section\n\n| A |\n| --- |\n| one |\n\n## Next\n"; - let state = state_for(source, "markdown"); - let table = state - .inner() - .tree - .chunks - .iter() - .find(|chunk| chunk.start_line == 3 && chunk.end_line == 5) - .expect("table chunk"); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::After, - sel: Some(format!("{}#{}", table.path, table.checksum)), - crc: None, - region: None, - content: Some("| two |\n".to_owned()), - find: None, - }); - - assert!( - result.diff_after.contains("| one |\n| two |\n\n## Next"), - "table-row append should land before the trailing blank-line separator: {:?}", - result.diff_after - ); - } - - #[test] - fn markdown_table_row_chunk_delete_removes_only_that_row() { - let source = "## Section\n\n| A | B |\n| --- | --- |\n| one | 1 |\n| two | 2 |\n\n## Next\n"; - let state = state_for(source, "markdown"); - let table = state - .inner() - .tree - .chunks - .iter() - .find(|chunk| chunk.start_line == 3 && chunk.end_line == 6) - .expect("table chunk"); - let row_path = table.children[2].clone(); - let row = state.inner().chunk(&row_path).expect("row chunk"); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::Delete, - sel: Some(format!("{}#{}", row.path, row.checksum)), - crc: None, - region: None, - content: None, - find: None, - }); - - assert!( - !result.diff_after.contains("| one | 1 |"), - "target row should be deleted: {:?}", - result.diff_after - ); - assert!( - result.diff_after.contains("| two | 2 |\n\n## Next"), - "other rows and section spacing should survive: {:?}", - result.diff_after - ); - } - - #[test] - fn markdown_fenced_python_body_write_preserves_code_indent() { - let source = "```python\ndef outer():\n if cond:\n return 1\n```\n"; - let state = state_for(source, "markdown"); - let function = state - .inner() - .tree - .chunks - .iter() - .find(|chunk| chunk.kind == ChunkKind::Function) - .expect("embedded python function chunk"); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("{}#{}~", function.path, function.checksum)), - crc: None, - region: None, - content: Some("if cond:\n\treturn 2\nreturn 3\n".to_owned()), - find: None, - }); - - assert!( - result - .diff_after - .contains("def outer():\n if cond:\n return 2\n return 3\n```"), - "embedded fenced Python body should keep 4-space code indentation: {:?}", - result.diff_after - ); - } - - #[test] - fn markdown_root_body_write_preserves_fenced_code_indentation_verbatim() { - let source = - "Intro\n\n# Title\n\n```python\ndef outer():\n return 1\n```\n\n## Notes\n\ntext\n"; - let state = state_for(source, "markdown"); - let root = state.inner().chunk("").expect("root chunk should exist"); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("#{}~", root.checksum)), - crc: None, - region: None, - content: Some( - "Intro changed\n\n# Title\n\n```python\ndef outer():\n return 1\n```\n\n## \ - Notes\n\ntext\n" - .to_owned(), - ), - find: None, - }); - - assert!( - result - .diff_after - .contains("def outer():\n return 1\n```"), - "root markdown body write should not add extra indentation inside fences: {:?}", - result.diff_after - ); - assert!( - !result.diff_after.contains("def outer():\n return 1"), - "fenced code indentation should not be inflated: {:?}", - result.diff_after - ); - } - - #[test] - fn rust_trait_members_are_addressable() { - let source = "trait Handler {\n fn handle(&self, req: &str) -> String;\n fn \ - name(&self) -> &str;\n}\n"; - let state = state_for(source, "rust"); - let tree = state.inner().tree(); - - let trait_chunk = tree - .chunks - .iter() - .find(|c| c.path == "tr_Han") - .expect("tr_Han should exist"); - - // Trait members should be listed as children even when they're - // single-line signatures (not collapsed as trivial). - assert!( - !trait_chunk.children.is_empty(), - "tr_Han should have children, got leaf. Chunks: {:?}", - tree.chunks.iter().map(|c| &c.path).collect::>() - ); - } - - #[test] - fn python_body_append_preserves_indentation() { - let source = "class Server:\n def __init__(self):\n self.x = 1\n\n def \ - start(self):\n pass\n"; - let state = state_for(source, "python"); - - let result = apply_single_edit(&state, "test.py", EditOperation { - op: ChunkEditOp::Append, - sel: Some("cls_Ser~".to_owned()), - crc: None, - region: None, - content: Some("def stop(self):\n\tpass\n".to_owned()), - find: None, - }); - - // The appended method should be at 4-space indent (class member level), - // with its body at 8-space indent. - assert!( - result - .diff_after - .contains(" def stop(self):\n pass"), - "appended method should have correct Python indentation: {}", - result.diff_after - ); - } - - #[test] - fn python_decorated_class_head_write_is_rejected() { - let source = "@dataclass\nclass Server:\n host: str\n port: int\n"; - let state = state_for(source, "python"); - let class_chunk = state.inner().chunk("cls_Ser").expect("cls_Ser"); - - let err = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("cls_Ser#{}^", class_chunk.checksum)), - crc: None, - region: None, - content: Some("@dataclass\nclass Server:\n".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.py".to_owned(), - normalize_indent: None, - }) - .err() - .expect("decorated Python class head write should be rejected"); - - assert!( - err.contains("Head writes on decorated Python cls_Ser are unsafe"), - "error should explain decorated Python head safety: {err}" - ); - } - - #[test] - fn python_head_delete_is_rejected_to_avoid_orphaned_body() { - let source = - "class Server:\n @property\n def address(self):\n return self.host\n"; - let state = state_for(source, "python"); - let function = state - .inner() - .chunk("cls_Ser.fn_add") - .expect("property function"); - - let err = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Delete, - sel: Some(format!("{}#{}^", function.path, function.checksum)), - crc: None, - region: None, - content: None, - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.py".to_owned(), - normalize_indent: None, - }) - .err() - .expect("Python head delete should be rejected"); - - assert!( - err.contains("Deleting the Python head region of cls_Ser.fn_add is unsafe"), - "error should explain orphaned-body risk: {err}" - ); - } - - #[test] - fn body_region_on_leaf_without_delimiters_is_rejected() { - let source = "enum LogLevel {\n Debug,\n Info,\n Warn,\n Fatal,\n}\n"; - let state = state_for(source, "rust"); - let chunk = state - .inner() - .chunk("en_Log.vr_Inf") - .expect("vr_Inf should exist"); - assert!(chunk.prologue_end_byte.is_none(), "leaf variant should not have prologue_end_byte"); - - for region_suffix in ["~", "^"] { - let sel = format!("en_Log.vr_Inf#{}{}", chunk.checksum, region_suffix); - let err = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some(sel), - crc: None, - region: None, - content: Some("Error,".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.rs".to_owned(), - normalize_indent: None, - }) - .err() - .expect("leaf region should be rejected"); - - assert!( - err.contains("Region suffix"), - "{region_suffix} should return a clear region error, got: {err}" - ); - assert!( - err.contains("unsuffixed selector"), - "{region_suffix} error should mention the safe workaround, got: {err}" - ); - } - } - #[test] - fn rust_impl_method_head_replace_no_body_duplication() { - let source = concat!( - "struct Server { -", - " running: bool, -", - "} -", - " -", - "impl Server { -", - " /// Starts the server. -", - " pub fn start(&mut self) { -", - " self.running = true; -", - " println!(\"started\"); -", - " } -", - "} -", - ); - let state = state_for(source, "rust"); - let chunk = state - .inner() - .chunk("ipl_Ser.fn_sta") - .expect("ipl_Ser.fn_sta should exist"); - assert!( - chunk.prologue_end_byte.is_some(), - "fn_sta should have prologue_end_byte, got: start_byte={}, end_byte={}, \ - prologue_end_byte={:?}, epilogue_start_byte={:?}", - chunk.start_byte, - chunk.end_byte, - chunk.prologue_end_byte, - chunk.epilogue_start_byte, - ); - - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("ipl_Ser.fn_sta#{}^", chunk.checksum)), - crc: None, - region: None, - content: Some( - " /// Initializes and starts the server.\n pub fn start(&mut self) {".to_owned(), - ), - find: None, - }); - - let body_count = result.diff_after.matches("self.running = true;").count(); - assert_eq!( - body_count, 1, - "body should appear exactly once after ^ replace, got {} occurrences in: -{}", - body_count, result.diff_after - ); - assert!( - result - .diff_after - .contains("/// Initializes and starts the server."), - "new doc comment should be in output: -{}", - result.diff_after - ); - assert!( - !result.diff_after.contains("/// Starts the server."), - "old doc comment should be removed: -{}", - result.diff_after - ); - } - - #[test] - fn typescript_class_method_head_replace_no_body_duplication() { - let source = concat!( - "class Server { -", - " /** Starts the server. */ -", - " start() { -", - " this.running = true; -", - " console.log(\"started\"); -", - " } -", - "} -", - ); - let state = state_for(source, "typescript"); - let chunk = state - .inner() - .chunk("cls_Ser.fn_sta") - .expect("cls_Ser.fn_sta should exist"); - assert!(chunk.prologue_end_byte.is_some(), "fn_sta should have prologue_end_byte"); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("cls_Ser.fn_sta#{}^", chunk.checksum)), - crc: None, - region: None, - content: Some(" /** Initializes the server. */\n start() {".to_owned()), - find: None, - }); - - let body_count = result.diff_after.matches("this.running = true;").count(); - assert_eq!( - body_count, 1, - "body should appear exactly once after ^ replace, got {} occurrences in: -{}", - body_count, result.diff_after - ); - assert!( - result.diff_after.contains("/** Initializes the server. */"), - "new doc comment should be in output: -{}", - result.diff_after - ); - } - - #[test] - fn python_body_replace_does_not_corrupt_surrounding_code() { - let source = - "import os\n\ndef main():\n x = 1\n print(x)\n\ndef helper():\n return 42\n"; - let state = state_for(source, "python"); - let chunk = state.inner().chunk("fn_mai").expect("fn_mai"); - - let result = apply_single_edit(&state, "test.py", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("fn_mai#{}~", chunk.checksum)), - crc: None, - region: None, - content: Some("y = 2\nprint(y)\n".to_owned()), - find: None, - }); - - assert!( - result.diff_after.contains("import os"), - "imports should survive body replace: {}", - result.diff_after - ); - assert!( - result.diff_after.contains("def main"), - "function head should survive body replace: {}", - result.diff_after - ); - assert!( - result.diff_after.contains("y = 2"), - "replacement body should appear: {}", - result.diff_after - ); - assert!( - result.diff_after.contains("def helper"), - "sibling function should survive body replace: {}", - result.diff_after - ); - assert!( - result.diff_after.contains("return 42"), - "sibling function body should survive: {}", - result.diff_after - ); - // Imports should remain at column 0, not indented - assert!( - result.diff_after.starts_with("import os"), - "import should be at column 0: {:?}", - &result.diff_after[..40.min(result.diff_after.len())] - ); - } - - #[test] - fn python_head_replace_does_not_orphan_body() { - let source = "class Server:\n def start(self) -> None:\n self.running = True\n"; - let state = state_for(source, "python"); - let chunk = state.inner().chunk("cls_Ser.fn_sta").expect("fn_sta"); - - let result = apply_single_edit(&state, "test.py", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("cls_Ser.fn_sta#{}^", chunk.checksum)), - crc: None, - region: None, - content: Some("def begin(self) -> None:\n".to_owned()), - find: None, - }); - - assert!( - result.diff_after.contains("def begin"), - "replaced head should appear: {}", - result.diff_after - ); - assert!( - result.diff_after.contains("self.running = True"), - "body should survive head replace: {}", - result.diff_after - ); - } - - #[test] - fn python_body_prepend_has_correct_indentation() { - let source = "def main():\n x = 1\n print(x)\n"; - let state = state_for(source, "python"); - - let result = apply_single_edit(&state, "test.py", EditOperation { - op: ChunkEditOp::Prepend, - sel: Some("fn_mai~".to_owned()), - crc: None, - region: None, - content: Some("y = 0\n".to_owned()), - find: None, - }); - - assert!( - result.diff_after.contains("y = 0"), - "prepended content should appear: {}", - result.diff_after - ); - assert!( - result.diff_after.contains("x = 1"), - "existing body should survive: {}", - result.diff_after - ); - // The prepended content should be at the body indent level - assert!( - result.diff_after.contains(" y = 0"), - "prepended content should be at body indent: {}", - result.diff_after - ); - } - - #[test] - fn python_class_body_replace_preserves_structure() { - let source = - "class Server:\n def start(self):\n pass\n\n def stop(self):\n pass\n"; - let state = state_for(source, "python"); - let chunk = state.inner().chunk("cls_Ser").expect("cls_Ser"); - - let result = apply_single_edit(&state, "test.py", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("cls_Ser#{}~", chunk.checksum)), - crc: None, - region: None, - content: Some("def run(self):\n\tpass\n".to_owned()), - find: None, - }); - - assert!( - result.diff_after.contains("class Server:"), - "class header should survive body replace: {}", - result.diff_after - ); - assert!( - result.diff_after.contains("def run(self)"), - "replaced body should appear: {}", - result.diff_after - ); - assert!( - !result.diff_after.contains("def start"), - "old body should be replaced: {}", - result.diff_after - ); - } - - #[test] - fn whole_chunk_replace_includes_leading_trivia_in_range() { - // Whole-chunk replace covers the full range including absorbed leading - // trivia (comments, attributes). If the replacement omits the trivia, - // it gets dropped — the read output shows the trivia as part of the - // chunk so the LLM knows to include it. - let source = "#[cfg(test)]\nmod tests {\n\tuse super::*;\n\n\t#[test]\n\tfn my_test() \ - {\n\t\told();\n\t}\n}\n"; - let state = state_for(source, "rust"); - let chunk = state - .inner() - .chunk("mod_tes.fn_my") - .expect("mod_tes.fn_my should exist"); - - // Verify the chunk absorbs the #[test] attribute as leading trivia. - assert!( - chunk.start_byte < chunk.checksum_start_byte, - "chunk should have absorbed leading trivia (start_byte {} < checksum_start_byte {})", - chunk.start_byte, - chunk.checksum_start_byte - ); - - // Replace the function WITHOUT including #[test] in the content. - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Put, - sel: Some("mod_tes.fn_my".to_owned()), - crc: Some(chunk.checksum.clone()), - region: None, - content: Some("fn my_test() {\n\tnew();\n}".to_owned()), - find: None, - }); - - // #[test] is dropped because the replacement didn't include it. - assert!( - !result.diff_after.contains("#[test]"), - "#[test] should be dropped when omitted from replacement. Full text:\n{}", - result.diff_after - ); - - // Verify the read output shows #[test] as part of the chunk's content - // so the LLM can see it needs to be included. - let read_output = crate::chunk::render::render_state(state.inner(), &RenderParams { - chunk_path: Some(String::new()), - title: "test.rs".to_owned(), - language_tag: Some("rust".to_owned()), - visible_range: None, - render_children_only: true, - omit_checksum: true, - anchor_style: Some(ChunkAnchorStyle::Full), - show_leaf_preview: true, - tab_replacement: Some(" ".to_owned()), - normalize_indent: Some(true), - focused_paths: None, - }); - println!("=== READ OUTPUT ===\n{read_output}\n=== END ==="); - assert!( - read_output.contains("#[test]"), - "read output must show #[test] as part of the chunk. Output:\n{read_output}" - ); - } - - #[test] - fn whole_chunk_replace_shows_diff_hunks_after_attribute_restoration() { - // Bug 2: After a first edit drops #[test] (bug 1), a follow-up edit that - // adds it back should show diff hunks in the response text. - // Uses a module with multiple functions and a batch of two replacements - // to match the real-world scenario. - let source = "\ -#[cfg(test)]\nmod tests {\n\tuse super::*;\n\n\tfn test_alpha() {\n\t\told_alpha();\n\t}\n\n\tfn \ - test_middle() {\n\t\tmiddle();\n\t}\n\n\tfn test_beta() \ - {\n\t\told_beta();\n\t}\n}\n"; - let state = state_for(source, "rust"); - let chunk_a = state - .inner() - .chunk("mod_tes.fn_tes_1") - .expect("fn_tes_1 should exist"); - let chunk_b = state - .inner() - .chunk("mod_tes.fn_tes_3") - .expect("fn_tes_3 should exist"); - - // Batch replace: add #[test] to both functions. - let result = apply_edits(&state, &EditParams { - operations: vec![ - EditOperation { - op: ChunkEditOp::Put, - sel: Some("mod_tes.fn_tes_1".to_owned()), - crc: Some(chunk_a.checksum.clone()), - region: None, - content: Some("#[test]\nfn test_alpha() {\n\tnew_alpha();\n}".to_owned()), - find: None, - }, - EditOperation { - op: ChunkEditOp::Put, - sel: Some("mod_tes.fn_tes_3".to_owned()), - crc: Some(chunk_b.checksum.clone()), - region: None, - content: Some("#[test]\nfn test_beta() {\n\tnew_beta();\n}".to_owned()), - find: None, - }, - ], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.rs".to_owned(), - normalize_indent: None, - }) - .expect("edit should apply"); - - assert!(result.changed, "edit should be detected as a change"); - assert!( - result.diff_after.contains("#[test]"), - "#[test] should be in the result. Full text:\n{}", - result.diff_after - ); - // The response text should contain diff hunks (@@) showing the changes. - assert!( - result.response_text.contains("@@"), - "response should include diff hunks showing the changes. Response:\n{}", - result.response_text - ); - } - - #[test] - fn diff_hunks_shown_for_non_leaf_function_replacement() { - // Bug 2 (realistic): When replacing functions that have children - // (sub-chunks like stmts, let bindings), the diff hunks should still - // appear in the response. Mirrors the real-world scenario where only - // #[test] is added and the function body stays identical. - let source = - "\ -#[cfg(test)]\nmod tests {\n\tuse super::*;\n\n\tfn test_alpha() {\n\t\tlet mut config = \ - base_config();\n\t\tconfig.enabled = Some(false);\n\t\tconfig.max_items = \ - Some(10);\n\n\t\tlet Err(error) = build_options(&config) else {\n\t\t\tpanic!(\"should \ - fail\");\n\t\t};\n\t\tassert_error_contains(&error, \"cannot be \ - combined\");\n\t}\n\n\tfn test_middle() {\n\t\tmiddle();\n\t}\n\n\tfn test_beta() \ - {\n\t\tlet mut config = base_config();\n\t\tconfig.enabled = \ - Some(true);\n\t\tconfig.max_size = Some(0);\n\n\t\tlet Err(error) = \ - build_options(&config) else {\n\t\t\tpanic!(\"must be \ - positive\");\n\t\t};\n\t\tassert_error_contains(&error, \"must be positive\");\n\t}\n}\n"; - let state = state_for(source, "rust"); - - // Verify the functions have children (sub-chunks). - let chunk_a = state - .inner() - .chunk("mod_tes.fn_tes_1") - .expect("fn_tes_1 should exist"); - assert!( - !chunk_a.children.is_empty(), - "fn_tes should have children (sub-chunks), got: {:?}", - chunk_a.children - ); - let chunk_b = state - .inner() - .chunk("mod_tes.fn_tes_3") - .expect("fn_tes_3 should exist"); - assert!( - !chunk_b.children.is_empty(), - "fn_tes should have children (sub-chunks), got: {:?}", - chunk_b.children - ); - - // Replace both functions: only adding #[test], body is identical. - let result = apply_edits(&state, &EditParams { - operations: vec![ - EditOperation { - op: ChunkEditOp::Put, - sel: Some("mod_tes.fn_tes_1".to_owned()), - crc: Some(chunk_a.checksum.clone()), - region: None, - content: Some( - "#[test]\nfn test_alpha() {\n\tlet mut config = \ - base_config();\n\tconfig.enabled = Some(false);\n\tconfig.max_items = \ - Some(10);\n\n\tlet Err(error) = build_options(&config) else \ - {\n\t\tpanic!(\"should fail\");\n\t};\n\tassert_error_contains(&error, \ - \"cannot be combined\");\n}" - .to_owned(), - ), - find: None, - }, - EditOperation { - op: ChunkEditOp::Put, - sel: Some("mod_tes.fn_tes_3".to_owned()), - crc: Some(chunk_b.checksum.clone()), - region: None, - content: Some( - "#[test]\nfn test_beta() {\n\tlet mut config = base_config();\n\tconfig.enabled \ - = Some(true);\n\tconfig.max_size = Some(0);\n\n\tlet Err(error) = \ - build_options(&config) else {\n\t\tpanic!(\"must be \ - positive\");\n\t};\n\tassert_error_contains(&error, \"must be positive\");\n}" - .to_owned(), - ), - find: None, - }, - ], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.rs".to_owned(), - normalize_indent: None, - }) - .expect("edit should apply"); - - assert!(result.changed, "edit should be detected as a change"); - assert!(result.diff_before != result.diff_after, "diff_before and diff_after should differ"); - // Count actual diff hunks. - let hunks = super::generate_diff_hunks(&result.diff_before, &result.diff_after, 0); - assert!( - !hunks.is_empty(), - "generate_diff_hunks should produce non-empty hunks.\ndiff_before:\n{}\ndiff_after:\n{}", - result.diff_before, - result.diff_after, - ); - // The response text should contain diff hunks (@@) showing the changes. - assert!( - result.response_text.contains("@@"), - "response should include diff hunks showing the changes.\nhunks: {}\nResponse:\n{}", - hunks.len(), - result.response_text, - ); - } - - #[test] - fn conflicted_reads_render_conflict_children_and_both_sides() { - let source = "\ -function foo() {\n<<<<<<< HEAD\n\treturn bar();\n=======\n\treturn baz();\n>>>>>>> topic\n}\n"; - let state = parsed_state_for(source, "typescript"); - assert!(state.has_conflicts()); - assert_eq!(state.conflict_count(), 1); - - let conflict = state - .inner() - .chunks() - .find(|chunk| chunk.kind == ChunkKind::Conflict) - .expect("conflict chunk should exist") - .clone(); - let rendered = state - .render_read(crate::chunk::types::ReadRenderParams { - read_path: String::new(), - display_path: "test.ts".to_owned(), - language_tag: Some("ts".to_owned()), - omit_checksum: false, - anchor_style: Some(ChunkAnchorStyle::Full), - absolute_line_range: None, - tab_replacement: Some(" ".to_owned()), - normalize_indent: Some(true), - }) - .expect("render should succeed"); - - assert!(rendered.text.contains(conflict.path.as_str())); - assert!( - rendered - .text - .contains(format!("{}.ours", conflict.path).as_str()) - ); - assert!( - rendered - .text - .contains(format!("{}.theirs", conflict.path).as_str()) - ); - assert!(rendered.text.contains("return bar();")); - assert!(rendered.text.contains("return baz();")); - } - - #[test] - fn delete_ours_accepts_theirs() { - let source = "\ -function foo() {\n<<<<<<< HEAD\n\treturn bar();\n=======\n\treturn baz();\n>>>>>>> topic\n}\n"; - let state = parsed_state_for(source, "typescript"); - let ours = state - .inner() - .chunks() - .find(|chunk| chunk.kind == ChunkKind::Ours) - .expect("ours chunk should exist") - .clone(); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Delete, - sel: Some(ours.path.clone()), - crc: Some(ours.checksum), - region: None, - content: None, - find: None, - }); - - assert!(!result.state.has_conflicts()); - assert!(result.diff_before.contains("<<<<<<< HEAD")); - assert!(result.diff_after.contains("return baz();")); - assert!(!result.diff_after.contains("<<<<<<<")); - } - - #[test] - fn delete_theirs_accepts_ours() { - let source = "\ -function foo() {\n<<<<<<< HEAD\n\treturn bar();\n=======\n\treturn baz();\n>>>>>>> topic\n}\n"; - let state = parsed_state_for(source, "typescript"); - let theirs = state - .inner() - .chunks() - .find(|chunk| chunk.kind == ChunkKind::Theirs) - .expect("theirs chunk should exist") - .clone(); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Delete, - sel: Some(theirs.path.clone()), - crc: Some(theirs.checksum), - region: None, - content: None, - find: None, - }); - - assert!(!result.state.has_conflicts()); - assert!(result.diff_after.contains("return bar();")); - assert!(!result.diff_after.contains("<<<<<<<")); - } - - #[test] - fn replace_conflict_manually_merges_and_clears_metadata() { - let source = "\ -function foo() {\n<<<<<<< HEAD\n\treturn bar();\n=======\n\treturn baz();\n>>>>>>> topic\n}\n"; - let state = parsed_state_for(source, "typescript"); - let conflict = state - .inner() - .chunks() - .find(|chunk| chunk.kind == ChunkKind::Conflict) - .expect("conflict chunk should exist") - .clone(); - - let result = apply_single_edit(&state, "test.ts", EditOperation { - op: ChunkEditOp::Put, - sel: Some(conflict.path.clone()), - crc: Some(conflict.checksum), - region: None, - content: Some("\treturn qux();\n".to_owned()), - find: None, - }); - - assert!(!result.state.has_conflicts()); - assert!(result.diff_after.contains("return qux();")); - assert!(!result.diff_after.contains("<<<<<<<")); - } - - #[test] - fn unresolved_conflicts_survive_rebuilds_within_a_batch() { - let source = "\ -function foo() {\n<<<<<<< HEAD\n\treturn bar();\n=======\n\treturn baz();\n>>>>>>> topic\n}\n"; - let state = parsed_state_for(source, "typescript"); - let conflict = state - .inner() - .chunks() - .find(|chunk| chunk.kind == ChunkKind::Conflict) - .expect("conflict chunk should exist") - .clone(); - let ours = state - .inner() - .chunk(format!("{}.ours", conflict.path).as_str()) - .expect("ours child should exist") - .clone(); - let theirs = state - .inner() - .chunk(format!("{}.theirs", conflict.path).as_str()) - .expect("theirs child should exist") - .clone(); - - let result = apply_edits(&state, &EditParams { - operations: vec![ - EditOperation { - op: ChunkEditOp::Put, - sel: Some(ours.path.clone()), - crc: Some(ours.checksum), - region: None, - content: Some("\treturn bar(1);\n".to_owned()), - find: None, - }, - EditOperation { - op: ChunkEditOp::Delete, - sel: Some(theirs.path.clone()), - crc: Some(theirs.checksum), - region: None, - content: None, - find: None, - }, - ], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.ts".to_owned(), - normalize_indent: None, - }) - .expect("batch edit should apply"); - - assert!(!result.state.has_conflicts()); - assert!(result.diff_after.contains("return bar(1);")); - assert!(!result.diff_after.contains("<<<<<<<")); - } - - #[test] - fn exported_decorated_class_is_addressable() { - let source = concat!( - "function sealed(target: any) {}\n", - "\n", - "@sealed\n", - "export class Server {\n", - " start(): void {\n", - " console.log(\"starting\");\n", - " }\n", - " stop(): void {\n", - " console.log(\"stopping\");\n", - " }\n", - "}\n", - "\n", - "function formatLog(msg: string): string {\n", - " return `[LOG] ${msg}`;\n", - "}\n", - ); - let state = state_for(source, "typescript"); - let tree = state.inner().tree(); - - let class_chunk = tree - .chunks - .iter() - .find(|c| c.path == "cls_Ser") - .unwrap_or_else(|| { - panic!( - "cls_Ser should be in the chunk tree. Available chunks: {:?}", - tree.chunks.iter().map(|c| &c.path).collect::>() - ) - }); - assert!(!class_chunk.children.is_empty(), "cls_Ser should have child methods"); - - let start = state.inner().chunk("cls_Ser.fn_sta"); - assert!(start.is_some(), "cls_Ser.fn_sta should exist"); - let stop = state.inner().chunk("cls_Ser.fn_sto"); - assert!(stop.is_some(), "cls_Ser.fn_sto should exist"); - } - - #[test] - fn head_replace_on_nested_rust_fn_uniform_indent() { - let source = concat!( - "pub struct Server {\n", - "\thost: String,\n", - "\tport: u16,\n", - "}\n", - "\n", - "impl Server {\n", - "\tpub fn address(&self) -> String {\n", - "\t\tformat!(\"{}:{}\", self.host, self.port)\n", - "\t}\n", - "}\n", - ); - let state = state_for(source, "rust"); - let chunk = state - .inner() - .chunk("ipl_Ser.fn_add") - .expect("ipl_Ser.fn_add should exist"); - - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("ipl_Ser.fn_add#{}^", chunk.checksum)), - crc: None, - region: None, - content: Some( - "/// Returns the server address.\n#[must_use]\npub fn address(&self) -> String {\n" - .to_owned(), - ), - find: None, - }); - - assert!( - result - .diff_after - .contains("\t/// Returns the server address."), - "doc comment should be at 1-tab indent, got:\n{}", - result.diff_after - ); - assert!( - result.diff_after.contains("\t#[must_use]"), - "attribute should be at 1-tab indent, got:\n{}", - result.diff_after - ); - assert!( - result.diff_after.contains("\tpub fn address"), - "signature should be at 1-tab indent, got:\n{}", - result.diff_after - ); - assert!( - !result.diff_after.contains("\t\t///"), - "doc comment must NOT be double-indented, got:\n{}", - result.diff_after - ); - } - - #[test] - fn markdown_append_chunk_preserves_trailing_blank_line() { - // Two sibling sections separated by a blank line. Appending to the - // paragraph chunk (leaf) inside the first section must preserve the - // blank-line gap before the next heading. - let source = "# Title\n\nSome text.\n\n## Next Section\n\nMore text.\n"; - let state = state_for(source, "markdown"); - - // The paragraph "Some text." is sect_Title.chunk_2. - let para_chunk = state - .inner() - .chunk("sct_Tit.ch_2") - .expect("paragraph chunk should exist"); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::Append, - sel: Some(para_chunk.path.clone()), - crc: None, - region: None, - content: Some("Appended line.\n".to_owned()), - find: None, - }); - - // The blank line before ## Next Section should be preserved - assert!( - result - .diff_after - .contains("Appended line.\n\n## Next Section"), - "blank line before next section must be preserved after append: {:?}", - result.diff_after - ); - } - - #[test] - fn markdown_after_chunk_preserves_blank_line_separator() { - // 'after' on a table chunk followed by a blank-line separator and a - // heading. The blank line must survive the insertion. - let source = "# Section\n\n| A | B |\n|---|---|\n| 1 | 2 |\n\n## Next\n"; - let state = state_for(source, "markdown"); - - // The table is sect_Sectio.chunk_2 (L3-L5). - let table_chunk = state - .inner() - .chunk("sct_Sec.ch_2") - .expect("table chunk should exist"); - - let result = apply_single_edit(&state, "test.md", EditOperation { - op: ChunkEditOp::After, - sel: Some(table_chunk.path.clone()), - crc: None, - region: None, - content: Some("Extra paragraph.\n".to_owned()), - find: None, - }); - - // Blank line before ## Next must be preserved - assert!( - result.diff_after.contains("Extra paragraph.\n\n## Next"), - "blank line before next heading must be preserved after 'after' insert: {:?}", - result.diff_after - ); - } - - #[test] - fn body_replace_nested_fn_uses_correct_indent() { - let source = "impl Server {\n fn is_running(&self) -> bool {\n true\n }\n}\n"; - let state = state_for(source, "rust"); - let chunk = state - .inner() - .tree - .chunks - .iter() - .find(|c| c.identifier.as_deref() == Some("is") || c.path.contains("fn_is")) - .expect("is_running chunk"); - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Put, - sel: Some(chunk.path.clone()), - crc: Some(chunk.checksum.clone()), - region: Some(ChunkRegion::Body), - content: Some("false\n".to_owned()), - find: None, - }); - // Body should be at 2 levels of indent (8 spaces), not 1 level (4 spaces) - let new_source = &result.diff_after; - assert!( - new_source.contains(" false"), - "expected body at 8-space indent (2 levels), got:\n{new_source}" - ); - } - - #[test] - fn body_replace_preserves_closing_delimiter_on_own_line() { - let source = "fn foo() {\n old_body();\n}\n"; - let state = state_for(source, "rust"); - let chunk = state.inner().chunk("fn_foo").expect("fn_foo"); - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Put, - sel: Some("fn_foo".to_owned()), - crc: Some(chunk.checksum.clone()), - region: Some(ChunkRegion::Body), - content: Some("new_body();".to_owned()), // No trailing newline - find: None, - }); - let new_source = &result.diff_after; - // Closing } should be on its own line, not merged - assert!( - new_source.contains("new_body();\n}"), - "expected closing brace on own line, got:\n{new_source}" - ); - } - - #[test] - fn is_comment_only_line_distinguishes_hash_comment_from_private_field_and_attribute() { - // Shell/Python-style comments are treated as comments. - assert!(is_comment_only_line("")); - assert!(is_comment_only_line("# shell comment")); - assert!(is_comment_only_line("#\tpython-style tab after hash")); - assert!(is_comment_only_line("#")); - assert!(is_comment_only_line("#!/usr/bin/env bash")); - assert!(is_comment_only_line("// line comment")); - assert!(is_comment_only_line("/// doc comment")); - assert!(is_comment_only_line("/* block comment start")); - - // TypeScript / JavaScript private fields are NOT comments. - assert!(!is_comment_only_line("#config: Config;")); - assert!(!is_comment_only_line("#running = false;")); - assert!(!is_comment_only_line("#_internal: number = 0;")); - - // Rust attributes and inner attributes are NOT comments. - assert!(!is_comment_only_line("#[napi]")); - assert!(!is_comment_only_line("#[derive(Debug)]")); - assert!(!is_comment_only_line("#![deny(warnings)]")); - - // Plain code lines are not comments. - assert!(!is_comment_only_line("let x = 1;")); - assert!(!is_comment_only_line("return 0;")); - } - - #[test] - fn deletion_cleanup_collapses_blank_line_before_closing_delimiter() { - // Scenario: deleting the last method in a class leaves }\n\n}. - // The cleanup should collapse to }\n}. - let text = "class Foo {\n\tmethod() {}\n\n}\n"; - let offset = "class Foo {\n\tmethod() {}\n".len(); - let result = cleanup_blank_line_artifacts_at_offset(text, offset); - assert!( - !result.contains("}\n\n}"), - "should collapse blank line before closing brace, got: {result:?}" - ); - assert!( - result.contains("method() {}\n}"), - "last method should be followed directly by class close, got: {result:?}" - ); - } - - #[test] - fn deletion_cleanup_collapses_blank_line_after_opening_delimiter() { - // Scenario: deleting the first child in a container leaves {\n\n\tcontent. - // The cleanup should collapse to {\n\tcontent. - let text = "class Foo {\n\n\tfield: number;\n}\n"; - let offset = "class Foo {\n".len(); - let result = cleanup_blank_line_artifacts_at_offset(text, offset); - assert!( - !result.contains("{\n\n\t"), - "should collapse blank line after opening brace, got: {result:?}" - ); - assert!( - result.contains("{\n\tfield"), - "first child should follow opening brace directly, got: {result:?}" - ); - } - - #[test] - fn body_region_on_leaf_if_is_rejected_python() { - // Python `if` inside a function body is a leaf chunk. Using `~` on it - // used to fall back to a whole-chunk replacement and could corrupt - // indentation around the guard; it is now rejected with a clear - // workaround instead. - let source = "def handle(request):\n x = 1\n y = 2\n if request.ok:\n \ - return \"yes\"\n z = 3\n for item in items:\n process(item)\n \ - return \"no\"\n"; - let state = parsed_state_for(source, "python"); - let if_chunk = state - .inner() - .tree - .chunks - .iter() - .find(|c| { - Path::new(&c.path) - .extension() - .is_some_and(|ext| ext.eq_ignore_ascii_case("if")) - }) - .expect("if chunk should exist"); - assert!(if_chunk.leaf, "if chunk should be leaf"); - - let err = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("{}~", if_chunk.path)), - crc: Some(if_chunk.checksum.clone()), - region: None, - content: Some("if request.ok:\n return \"forced\"\n".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.py".to_owned(), - normalize_indent: None, - }) - .err() - .expect("leaf ~ should be rejected"); - - assert!( - err.contains("Python compound-statement leaf chunks"), - "error should identify the unsafe leaf fallback, got: {err}" - ); - assert!( - err.contains("parent container's `~`"), - "error should mention the parent-body workaround, got: {err}" - ); - } - - #[test] - fn body_region_fallback_that_omits_python_head_is_rejected() { - let source = "def handle(request):\n x = 1\n y = 2\n if request.ok:\n \ - return \"yes\"\n z = 3\n for item in items:\n process(item)\n \ - return \"no\"\n"; - let state = parsed_state_for(source, "python"); - let if_chunk = state - .inner() - .tree - .chunks - .iter() - .find(|c| c.kind == ChunkKind::If) - .expect("if chunk should exist"); - - let err = apply_edits(&state, &EditParams { - operations: vec![EditOperation { - op: ChunkEditOp::Put, - sel: Some(format!("{}~", if_chunk.path)), - crc: Some(if_chunk.checksum.clone()), - region: None, - content: Some("return \"forced\"\n".to_owned()), - find: None, - }], - default_selector: None, - default_crc: None, - anchor_style: None, - cwd: ".".to_owned(), - file_path: "test.py".to_owned(), - normalize_indent: None, - }) - .err() - .expect("fallback body edit should be rejected"); - - assert!( - err.contains("Use the unsuffixed selector with complete replacement content"), - "error should tell the operator to include the complete leaf chunk, got: {err}" - ); - } - - #[test] - fn delete_chunk_produces_removal_diff() { - let source = "fn foo() {\n println!(\"a\");\n}\n\nfn bar() {\n println!(\"b\");\n}\n"; - let state = state_for(source, "rust"); - let foo = state.inner().chunk("fn_foo").expect("fn_foo"); - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Put, - sel: Some("fn_foo".to_owned()), - crc: Some(foo.checksum.clone()), - region: None, - content: Some(String::new()), - find: None, - }); - - assert!(result.changed, "deletion should be marked as changed"); - // The response_text should include a diff showing removed lines, - // not an empty body after the file header. - assert!( - result.response_text.contains("fn foo"), - "deletion response should include a diff showing the removed function: {}", - result.response_text - ); - } - - #[test] - fn delete_first_enum_variant_produces_diff() { - let source = "enum Level {\n Debug,\n Info,\n Warn,\n}\n"; - let state = state_for(source, "rust"); - let debug = state.inner().chunk("en_Lev.vr_Deb").expect("vr_Deb"); - let result = apply_single_edit(&state, "test.rs", EditOperation { - op: ChunkEditOp::Put, - sel: Some("en_Lev.vr_Deb".to_owned()), - crc: Some(debug.checksum.clone()), - region: None, - content: Some(String::new()), - find: None, - }); - - assert!(result.changed, "deletion should be marked as changed"); - // The response should show the removed variant in a diff hunk. - assert!( - result.response_text.contains("Debug"), - "deletion of first enum variant should show a diff with the removed content: {}", - result.response_text - ); - assert!( - result.response_text.contains("@en_Lev.vr_Deb#"), - "deleted variant should keep its chunk anchor with a deletion marker: {}", - result.response_text - ); - } -} diff --git a/crates/pi-natives/src/chunk/indent.rs b/crates/pi-natives/src/chunk/indent.rs deleted file mode 100644 index 1567006a9..000000000 --- a/crates/pi-natives/src/chunk/indent.rs +++ /dev/null @@ -1,618 +0,0 @@ -use crate::chunk::{HASHLINE_BIGRAMS, types::ChunkTree}; - -const DEFAULT_SPACE_INDENT_STEP: usize = 4; -const MAX_REASONABLE_INDENT_STEP: usize = 8; - -pub fn dedent_python_style(text: &str) -> String { - let mut margin: Option<&str> = None; - for line in text.split('\n') { - if line.trim().is_empty() { - continue; - } - let indent = leading_whitespace(line); - margin = Some(match margin { - None => indent, - Some(current) if indent.starts_with(current) => current, - Some(current) if current.starts_with(indent) => indent, - Some(current) => common_prefix(current, indent), - }); - } - - let Some(margin) = margin else { - return text.to_owned(); - }; - if margin.is_empty() { - return text.to_owned(); - } - - text - .split('\n') - .map(|line| line.strip_prefix(margin).unwrap_or(line)) - .collect::>() - .join("\n") -} - -pub fn indent_non_empty_lines(text: &str, prefix: &str) -> String { - if prefix.is_empty() { - return text.to_owned(); - } - text - .split('\n') - .map(|line| { - if line.trim().is_empty() { - line.to_owned() - } else { - format!("{prefix}{line}") - } - }) - .collect::>() - .join("\n") -} - -pub fn detect_space_indent_step(text: &str) -> usize { - let mut min = usize::MAX; - for line in text.split('\n') { - if line.trim().is_empty() { - continue; - } - let count = line.chars().take_while(|ch| *ch == ' ').count(); - if count > 0 { - min = min.min(count); - } - } - if min == usize::MAX || min == 0 || min > MAX_REASONABLE_INDENT_STEP { - DEFAULT_SPACE_INDENT_STEP - } else { - min - } -} - -pub fn count_indent_columns(whitespace: &str, space_step: usize) -> usize { - whitespace - .chars() - .map(|ch| if ch == '\t' { space_step } else { 1 }) - .sum() -} - -pub fn normalize_to_tabs(line: &str, indent_char: char, indent_step: usize) -> String { - if indent_char == '\t' { - return line.to_owned(); - } - - let whitespace = leading_whitespace(line); - if whitespace.is_empty() { - return line.to_owned(); - } - - let step = indent_step.max(1); - let total_columns = count_indent_columns(whitespace, step); - let tabs = total_columns / step; - let remainder = total_columns % step; - format!("{}{}{}", "\t".repeat(tabs), " ".repeat(remainder), &line[whitespace.len()..]) -} - -pub fn denormalize_from_tabs( - line: &str, - file_indent_char: char, - file_indent_step: usize, -) -> String { - if file_indent_char != ' ' && file_indent_char != '\t' { - return line.to_owned(); - } - - let whitespace = leading_whitespace(line); - if whitespace.is_empty() { - return line.to_owned(); - } - - let step = file_indent_step.max(1); - let mut converted = String::with_capacity(whitespace.len() * step.max(1)); - for ch in whitespace.chars() { - match ch { - '\t' if file_indent_char == '\t' => converted.push('\t'), - '\t' => converted.push_str(&file_indent_char.to_string().repeat(step)), - ' ' => converted.push(' '), - _ => converted.push(ch), - } - } - format!("{converted}{}", &line[whitespace.len()..]) -} - -pub fn normalize_target_indent(target_indent: &str, sample_text: &str) -> String { - if target_indent.is_empty() { - return String::new(); - } - let has_tabs = target_indent.contains('\t'); - let has_spaces = target_indent.contains(' '); - if !has_tabs || !has_spaces { - return target_indent.to_owned(); - } - - let space_step = detect_space_indent_step(sample_text); - let total_columns = count_indent_columns(target_indent, space_step); - let normalized_levels = round_to_nearest_step(total_columns, space_step) / space_step; - if normalized_levels == 0 { - return String::new(); - } - - match target_indent.chars().next().unwrap_or(' ') { - '\t' => "\t".repeat(normalized_levels), - _ => " ".repeat(normalized_levels * space_step), - } -} - -pub fn normalize_leading_whitespace_char( - text: &str, - target_char: char, - file_indent_step: Option, -) -> String { - if target_char != ' ' && target_char != '\t' { - return text.to_owned(); - } - - let other_char = if target_char == ' ' { '\t' } else { ' ' }; - let mut needs_conversion = false; - for line in text.split('\n') { - if line.trim().is_empty() { - continue; - } - let ws = leading_whitespace(line); - if ws.is_empty() { - continue; - } - needs_conversion |= ws.contains(other_char); - } - - if !needs_conversion { - return text.to_owned(); - } - - let space_step = if target_char == ' ' { - file_indent_step - .filter(|step| *step > 1) - .unwrap_or_else(|| detect_space_indent_step(text)) - } else { - file_indent_step.unwrap_or_else(|| detect_space_indent_step(text)) - }; - - text - .split('\n') - .map(|line| { - let ws = leading_whitespace(line); - if ws.is_empty() { - return line.to_owned(); - } - let rest = &line[ws.len()..]; - let total_spaces = count_indent_columns(ws, space_step); - if target_char == ' ' { - format!("{}{}", " ".repeat(total_spaces), rest) - } else { - let tabs = total_spaces / space_step; - let remainder = total_spaces % space_step; - format!("{}{}{}", "\t".repeat(tabs), " ".repeat(remainder), rest) - } - }) - .collect::>() - .join("\n") -} - -pub fn reindent_inserted_block( - content: &str, - target_indent: &str, - file_indent_step: Option, -) -> String { - if content.is_empty() { - return String::new(); - } - - let normalized_target_indent = normalize_target_indent(target_indent, content); - let mut dedented = dedent_python_style(content); - - if let Some(target_char) = normalized_target_indent.chars().next() { - let target_step = if target_char == ' ' { - file_indent_step.unwrap_or_else(|| normalized_target_indent.chars().count()) - } else { - file_indent_step.unwrap_or_else(|| detect_space_indent_step(&dedented)) - }; - dedented = normalize_leading_whitespace_char(&dedented, target_char, Some(target_step)); - } - - indent_non_empty_lines(&dedented, &normalized_target_indent) -} - -/// Detect the file's indent character. -/// Prefer chunk metadata, then fall back to scanning source lines. -/// Returns `' '` when the file provides no indentation signal. -pub fn detect_file_indent_char(source: &str, tree: &ChunkTree) -> char { - for chunk in &tree.chunks { - if chunk.indent > 0 && !chunk.indent_char.is_empty() { - return chunk.indent_char.chars().next().unwrap_or(' '); - } - } - - for line in source.split('\n') { - if line.trim().is_empty() { - continue; - } - if let Some(ch) = leading_whitespace(line).chars().next() - && matches!(ch, ' ' | '\t') - { - return ch; - } - } - - ' ' -} - -/// Detect spaces-per-indent-level from parent→child indent differences. -/// Falls back to scanning `source` for the minimum indented-line width, -/// then to `DEFAULT_SPACE_INDENT_STEP` when neither gives a signal. -pub fn detect_file_indent_step(source: &str, tree: &ChunkTree) -> u32 { - for chunk in &tree.chunks { - if chunk.children.is_empty() { - continue; - } - for child_path in &chunk.children { - let Some(child) = tree - .chunks - .iter() - .find(|candidate| &candidate.path == child_path) - else { - continue; - }; - if child.indent <= chunk.indent || child.indent_char != " " { - continue; - } - let step = child.indent - chunk.indent; - if step > 0 && step <= MAX_REASONABLE_INDENT_STEP as u32 { - return step; - } - } - } - detect_space_indent_step(source) as u32 -} - -pub fn strip_content_prefixes(content: &str) -> String { - let lines = content.split('\n').collect::>(); - let mut line_num_count = 0usize; - let mut non_empty = 0usize; - for line in &lines { - if line.trim().is_empty() { - continue; - } - non_empty += 1; - if parse_chunk_gutter_code_row(line).is_some() { - line_num_count += 1; - } - } - - if non_empty == 0 { - return content.to_owned(); - } - - let without_line_numbers = if line_num_count * 10 > non_empty * 6 { - lines - .iter() - .map(|line| strip_chunk_gutter_line(line)) - .collect::>() - } else { - lines - .iter() - .map(|line| (*line).to_owned()) - .collect::>() - }; - - strip_new_line_prefixes(&without_line_numbers).join("\n") -} - -fn leading_whitespace(line: &str) -> &str { - let end = line - .char_indices() - .find_map(|(index, ch)| (!matches!(ch, ' ' | '\t')).then_some(index)) - .unwrap_or(line.len()); - &line[..end] -} - -fn common_prefix<'a>(left: &'a str, right: &'a str) -> &'a str { - let mut matched = 0usize; - for ((left_index, left_char), (_, right_char)) in left.char_indices().zip(right.char_indices()) { - if left_char != right_char { - break; - } - matched = left_index + left_char.len_utf8(); - } - &left[..matched] -} - -const fn round_to_nearest_step(value: usize, step: usize) -> usize { - if step == 0 { - return value; - } - ((value + (step / 2)) / step) * step -} - -fn parse_chunk_gutter_code_row(line: &str) -> Option<&str> { - let trimmed = line.trim_start_matches([' ', '\t']); - let digits = trimmed.chars().take_while(|ch| ch.is_ascii_digit()).count(); - if digits == 0 { - return None; - } - let after_digits = &trimmed[digits..]; - let after_spaces = after_digits.trim_start_matches([' ', '\t']); - let after_pipe = after_spaces - .strip_prefix('|') - .or_else(|| after_spaces.strip_prefix('│'))?; - Some( - after_pipe - .strip_prefix(' ') - .or_else(|| after_pipe.strip_prefix('\t')) - .unwrap_or(after_pipe), - ) -} - -fn strip_chunk_gutter_line(line: &str) -> String { - if let Some(rest) = parse_chunk_gutter_code_row(line) { - return rest.to_owned(); - } - let trimmed = line.trim_start_matches([' ', '\t']); - if trimmed.starts_with('|') || trimmed.starts_with('│') { - return String::new(); - } - line.to_owned() -} - -fn strip_new_line_prefixes(lines: &[String]) -> Vec { - let non_empty = lines.iter().filter(|line| !line.trim().is_empty()).count(); - if non_empty == 0 { - return lines.to_vec(); - } - - let hash_prefixed = lines - .iter() - .filter(|line| !line.trim().is_empty()) - .filter(|line| hashline_prefix_len(line).is_some()) - .count(); - if hash_prefixed == non_empty { - return lines - .iter() - .map(|line| match hashline_prefix_len(line) { - Some(prefix_len) => line[prefix_len..].to_owned(), - None => line.clone(), - }) - .collect(); - } - - lines.to_vec() -} - -fn hashline_prefix_len(line: &str) -> Option { - let mut offset = line.len() - line.trim_start_matches([' ', '\t']).len(); - let mut remainder = &line[offset..]; - - if let Some(stripped) = remainder.strip_prefix(">>>") { - offset += 3; - remainder = stripped; - } else if let Some(stripped) = remainder.strip_prefix(">>") { - offset += 2; - remainder = stripped; - } - - let ws = remainder.len() - remainder.trim_start_matches([' ', '\t']).len(); - offset += ws; - remainder = &remainder[ws..]; - - if let Some(stripped) = remainder.strip_prefix('+') { - offset += 1; - remainder = stripped; - let inner_ws = remainder.len() - remainder.trim_start_matches([' ', '\t']).len(); - offset += inner_ws; - remainder = &remainder[inner_ws..]; - } - - // Line number digits are mandatory (no `#`-only form in the new format). - let digits = remainder - .chars() - .take_while(|ch| ch.is_ascii_digit()) - .count(); - if digits == 0 { - return None; - } - offset += digits; - remainder = &remainder[digits..]; - - // Match exactly one BPE bigram (2 ASCII chars) from HASHLINE_BIGRAMS, - // directly adjacent to the line-number digits (no `#` separator). - // Use char-boundary-safe slicing to avoid panicking on multi-byte content. - let bigram_end = 2; - if remainder.len() < bigram_end || !remainder.is_char_boundary(bigram_end) { - return None; - } - let bigram = &remainder[..bigram_end]; - if !bigram.is_ascii() || !HASHLINE_BIGRAMS.contains(&bigram) { - return None; - } - offset += bigram_end; - remainder = &remainder[bigram_end..]; - - // Anchor terminator is a single colon character. - remainder.strip_prefix(':').map(|_| offset + 1) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::chunk::{kind::ChunkKind, types::ChunkNode}; - - fn chunk( - path: &str, - parent_path: Option<&str>, - children: &[&str], - indent: u32, - indent_char: &str, - ) -> ChunkNode { - let kind = match path.split_once('_').map_or(path, |(prefix, _)| prefix) { - "fn" => ChunkKind::Function, - "class" => ChunkKind::Class, - "stmts" => ChunkKind::Statements, - _ => ChunkKind::Chunk, - }; - ChunkNode { - path: path.to_owned(), - identifier: path - .split_once('_') - .and_then(|(_, identifier)| (!identifier.is_empty()).then_some(identifier.to_owned())), - kind, - leaf: children.is_empty(), - virtual_content: None, - parent_path: parent_path.map(str::to_owned), - children: children.iter().map(|child| (*child).to_owned()).collect(), - signature: None, - start_line: 1, - end_line: 1, - line_count: 1, - start_byte: 0, - end_byte: 0, - checksum_start_byte: 0, - prologue_end_byte: None, - epilogue_start_byte: None, - checksum: "ABCD".to_owned(), - error: false, - indent, - indent_char: indent_char.to_owned(), - group: false, - } - } - - #[test] - fn dedent_preserves_mixed_common_margin() { - let input = "\t foo\n\t bar\n\t baz"; - assert_eq!(dedent_python_style(input), "foo\n bar\nbaz"); - } - - #[test] - fn normalize_leading_whitespace_char_uses_file_step() { - let input = "\t alpha\n\t beta"; - assert_eq!( - normalize_leading_whitespace_char(input, ' ', Some(4)), - " alpha\n beta" - ); - } - - #[test] - fn canonical_indent_round_trips_common_profiles() { - let cases = [ - (" value()", ' ', 4, "\tvalue()", " value()"), - (" value()", ' ', 3, "\t\tvalue()", " value()"), - (" value()", ' ', 2, "\tvalue()", " value()"), - ("\tvalue()", '\t', 4, "\tvalue()", "\tvalue()"), - (" \t value()", ' ', 4, "\t value()", " value()"), - ]; - - for (input, indent_char, indent_step, canonical, restored) in cases { - let normalized = normalize_to_tabs(input, indent_char, indent_step); - assert_eq!(normalized, canonical, "unexpected canonical indent for {input:?}"); - assert_eq!( - denormalize_from_tabs(&normalized, indent_char, indent_step), - restored, - "unexpected restored indent for {input:?}" - ); - } - } - - #[test] - fn reindent_inserted_block_preserves_relative_indentation() { - // Agent sends tab-based content: call(\n\talpha,\n\tbeta,\n) - // After denormalize_from_tabs (4 spaces/tab): call(\n alpha,\n beta,\n) - // Should add target indent to all lines, preserving relative offsets. - let input = "call(\n alpha,\n beta,\n)"; - assert_eq!( - reindent_inserted_block(input, " ", Some(4)), - " call(\n alpha,\n beta,\n )" - ); - } - - #[test] - fn strip_content_prefixes_removes_gutter_and_meta_rows() { - let input = "10 | fn main() {\n │ <.fn_main#ABCD>\n11 | println!(\"hi\");\n12 | }"; - assert_eq!(strip_content_prefixes(input), "fn main() {\n\n println!(\"hi\");\n}"); - } - - #[test] - fn strip_content_prefixes_removes_colon_hashline_prefixes() { - let input = "1th:fn main() {\n2er:\tprintln!(\"hi\");\n3in:}"; - assert_eq!(strip_content_prefixes(input), "fn main() {\n\tprintln!(\"hi\");\n}"); - } - - #[test] - fn detect_file_indent_step_prefers_space_children() { - let tree = ChunkTree { - language: "rust".to_owned(), - checksum: "ABCD".to_owned(), - line_count: 1, - parse_errors: 0, - parse_error_lines: Vec::new(), - fallback: false, - root_path: String::new(), - root_children: vec!["cls_A".to_owned()], - chunks: vec![ - chunk("cls_A", Some(""), &["fn_b"], 0, " "), - chunk("fn_b", Some("cls_A"), &[], 2, " "), - ], - }; - assert_eq!(detect_file_indent_step("", &tree), 2); - } - - #[test] - fn detect_file_indent_step_falls_back_to_source_scan() { - // When the chunk tree has no parent->child pairs (all leaves), - // fall back to scanning source lines for the minimum indent width. - let tree = ChunkTree { - language: "yaml".to_owned(), - checksum: "ABCD".to_owned(), - line_count: 3, - parse_errors: 0, - parse_error_lines: Vec::new(), - fallback: false, - root_path: String::new(), - root_children: vec!["key_ser".to_owned()], - chunks: vec![chunk("key_ser", Some(""), &[], 0, " ")], - }; - let source = "server:\n host: localhost\n port: 5432\n"; - assert_eq!(detect_file_indent_step(source, &tree), 2); - } - - #[test] - fn detect_file_indent_char_falls_back_to_source_lines() { - let tree = ChunkTree { - language: "rust".to_owned(), - checksum: "ABCD".to_owned(), - line_count: 3, - parse_errors: 0, - parse_error_lines: Vec::new(), - fallback: false, - root_path: String::new(), - root_children: vec!["fn_mai".to_owned()], - chunks: vec![chunk("fn_mai", Some(""), &[], 0, "")], - }; - - let source = "fn main() {\n println!(\"hi\");\n}\n"; - assert_eq!(detect_file_indent_char(source, &tree), ' '); - } - - #[test] - fn detect_file_indent_char_defaults_to_spaces_without_signal() { - let tree = ChunkTree { - language: "rust".to_owned(), - checksum: "ABCD".to_owned(), - line_count: 0, - parse_errors: 0, - parse_error_lines: Vec::new(), - fallback: false, - root_path: String::new(), - root_children: Vec::new(), - chunks: Vec::new(), - }; - - assert_eq!(detect_file_indent_char("", &tree), ' '); - } -} diff --git a/crates/pi-natives/src/chunk/kind.rs b/crates/pi-natives/src/chunk/kind.rs deleted file mode 100644 index 96fde3e3a..000000000 --- a/crates/pi-natives/src/chunk/kind.rs +++ /dev/null @@ -1,699 +0,0 @@ -#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] -pub enum ChunkKind { - Add, - After, - Alias, - Algo, - Arg, - Argv, - Array, - At, - Attr, - AttrExpr, - Attrs, - Block, - BlockIf, - BlockLocals, - Body, - Case, - Cases, - Catch, - Cell, - Class, - Clause, - Cmd, - Code, - Cond, - Constructor, - Contract, - Copy, - Custom, - Declarations, - Decl, - DefaultExport, - Define, - Directive, - Either, - Elif, - Else, - Enum, - Env, - Error, - Except, - Exports, - Expose, - Expression, - Field, - Fields, - File, - Frame, - Function, - For, - ForIn, - ForOf, - Form, - Frontmatter, - Group, - GroupBy, - Headers, - Healthcheck, - Html, - Hunks, - Hunk, - If, - Iface, - Impl, - Imports, - Includes, - InlineFragment, - Install, - Interface, - Interpolation, - Item, - Join, - Key, - KeyScripts, - Label, - Let, - List, - Loop, - Macro, - Map, - Markdown, - Match, - Method, - Methods, - Module, - Mustache, - Object, - Operation, - Operator, - Option, - Options, - OrderBy, - Parameters, - Preamble, - Proc, - Process, - Project, - Proto, - Py, - Python, - Query, - Receive, - Recipe, - Relations, - Render, - Return, - Row, - Root, - Rule, - Schema, - Script, - ScriptModule, - ScriptSetup, - Section, - Select, - Setting, - Shebang, - Shell, - Slot, - Snippet, - Source, - Stage, - StaticInit, - Statements, - Struct, - Style, - StyleScoped, - Switch, - Table, - Tag, - Target, - Template, - Text, - Trait, - Translation, - Try, - Ts, - Type, - Typescript, - Union, - User, - Val, - Variable, - Variant, - Variants, - VersionGate, - When, - Where, - While, - With, - Workdir, - Conflict, - Ours, - Theirs, - Chunk, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum SummaryStyle { - Function, - Variable, - Imports, - Default, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct ChunkTraits { - pub groupable: bool, - pub packed: bool, - pub addressable_leaf: bool, - pub always_preserve_children: bool, - pub has_addressable_members: bool, - pub summary: SummaryStyle, - pub container: bool, -} - -const DEFAULT_TRAITS: ChunkTraits = ChunkTraits { - groupable: false, - packed: false, - addressable_leaf: false, - always_preserve_children: false, - has_addressable_members: false, - summary: SummaryStyle::Default, - container: false, -}; - -const GROUP_TRAITS: ChunkTraits = ChunkTraits { groupable: true, ..DEFAULT_TRAITS }; - -const CONTAINER_TRAITS: ChunkTraits = ChunkTraits { container: true, ..DEFAULT_TRAITS }; - -const ADDRESSABLE_CONTAINER_TRAITS: ChunkTraits = - ChunkTraits { container: true, has_addressable_members: true, ..DEFAULT_TRAITS }; - -const PRESERVE_CHILDREN_TRAITS: ChunkTraits = - ChunkTraits { container: true, always_preserve_children: true, ..DEFAULT_TRAITS }; - -const PACKED_LEAF_TRAITS: ChunkTraits = - ChunkTraits { packed: true, addressable_leaf: true, ..DEFAULT_TRAITS }; - -const FUNCTION_TRAITS: ChunkTraits = - ChunkTraits { summary: SummaryStyle::Function, ..DEFAULT_TRAITS }; - -const VARIABLE_TRAITS: ChunkTraits = ChunkTraits { - packed: true, - addressable_leaf: true, - summary: SummaryStyle::Variable, - ..DEFAULT_TRAITS -}; - -const IMPORTS_TRAITS: ChunkTraits = - ChunkTraits { groupable: true, summary: SummaryStyle::Imports, ..DEFAULT_TRAITS }; - -impl ChunkKind { - pub const fn prefix(self) -> &'static str { - match self { - Self::Add => "add", - Self::After => "aft", - Self::Alias => "al", - Self::Algo => "algo", - Self::Arg => "arg", - Self::Argv => "argv", - Self::Array => "ar", - Self::At => "at", - Self::Attr => "attr", - Self::AttrExpr => "aex", - Self::Attrs => "ats", - Self::Block => "blk", - Self::BlockIf => "bif", - Self::BlockLocals => "blc", - Self::Body => "b", - Self::Case => "case", - Self::Cases => "cs", - Self::Catch => "ctc", - Self::Cell => "cell", - Self::Class => "cls", - Self::Clause => "cla", - Self::Cmd => "cmd", - Self::Code => "code", - Self::Cond => "cond", - Self::Constructor => "ctor", - Self::Contract => "ctr", - Self::Copy => "copy", - Self::Custom => "cus", - Self::Declarations => "d", - Self::Decl => "decl", - Self::DefaultExport => "dex", - Self::Define => "def", - Self::Directive => "dir", - Self::Either => "eth", - Self::Elif => "elif", - Self::Else => "else", - Self::Enum => "en", - Self::Env => "env", - Self::Error => "err", - Self::Except => "exc", - Self::Exports => "exp", - Self::Expose => "exo", - Self::Expression => "ex", - Self::Field => "fld", - Self::Fields => "flds", - Self::File => "file", - Self::Frame => "fr", - Self::Function => "fn", - Self::For => "for", - Self::ForIn => "fri", - Self::ForOf => "fro", - Self::Form => "form", - Self::Frontmatter => "fm", - Self::Group => "grp", - Self::GroupBy => "gby", - Self::Headers => "hdrs", - Self::Healthcheck => "hck", - Self::Html => "html", - Self::Hunks => "hks", - Self::Hunk => "hunk", - Self::If => "if", - Self::Iface => "ifc", - Self::Impl => "ipl", - Self::Imports => "imp", - Self::Includes => "incl", - Self::InlineFragment => "ifr", - Self::Install => "inst", - Self::Interface => "intf", - Self::Interpolation => "itp", - Self::Item => "item", - Self::Join => "join", - Self::Key => "key", - Self::KeyScripts => "ksc", - Self::Label => "lbl", - Self::Let => "let", - Self::List => "list", - Self::Loop => "loop", - Self::Macro => "mc", - Self::Map => "map", - Self::Markdown => "md", - Self::Match => "mt", - Self::Method => "m", - Self::Methods => "ms", - Self::Module => "mod", - Self::Mustache => "mst", - Self::Object => "obj", - Self::Operation => "op", - Self::Operator => "oper", - Self::Option => "opt", - Self::Options => "opts", - Self::OrderBy => "oby", - Self::Parameters => "p", - Self::Preamble => "pre", - Self::Proc => "proc", - Self::Process => "prcs", - Self::Project => "proj", - Self::Proto => "pt", - Self::Py => "py", - Self::Python => "pyt", - Self::Query => "qry", - Self::Receive => "recv", - Self::Recipe => "rcp", - Self::Relations => "rels", - Self::Render => "rnd", - Self::Return => "ret", - Self::Row => "row", - Self::Root => "root", - Self::Rule => "rule", - Self::Schema => "sch", - Self::Script => "scr", - Self::ScriptModule => "smo", - Self::ScriptSetup => "sse", - Self::Section => "sct", - Self::Select => "sel", - Self::Setting => "stn", - Self::Shebang => "shb", - Self::Shell => "sh", - Self::Slot => "slot", - Self::Snippet => "snip", - Self::Source => "src", - Self::Stage => "stg", - Self::StaticInit => "sni", - Self::Statements => "st", - Self::Struct => "stc", - Self::Style => "sty", - Self::StyleScoped => "sco", - Self::Switch => "sw", - Self::Table => "tbl", - Self::Tag => "tag", - Self::Target => "tgt", - Self::Template => "tmpl", - Self::Text => "text", - Self::Trait => "tr", - Self::Translation => "trs", - Self::Try => "try", - Self::Ts => "ts", - Self::Type => "ty", - Self::Typescript => "tysc", - Self::Union => "u", - Self::User => "user", - Self::Val => "val", - Self::Variable => "var", - Self::Variant => "vr", - Self::Variants => "vrs", - Self::VersionGate => "vg", - Self::When => "when", - Self::Where => "wh", - Self::While => "wl", - Self::With => "with", - Self::Workdir => "wd", - Self::Conflict => "cfl", - Self::Ours => "ours", - Self::Theirs => "ths", - Self::Chunk => "ch", - } - } - - pub const fn traits(self) -> &'static ChunkTraits { - match self { - Self::Add => &GROUP_TRAITS, - Self::After => &GROUP_TRAITS, - Self::Alias => &DEFAULT_TRAITS, - Self::Algo => &CONTAINER_TRAITS, - Self::Arg => &DEFAULT_TRAITS, - Self::Argv => &DEFAULT_TRAITS, - Self::Array => &DEFAULT_TRAITS, - Self::At => &DEFAULT_TRAITS, - Self::Attr => &DEFAULT_TRAITS, - Self::AttrExpr => &DEFAULT_TRAITS, - Self::Attrs => &CONTAINER_TRAITS, - Self::Block => &DEFAULT_TRAITS, - Self::BlockIf => &DEFAULT_TRAITS, - Self::BlockLocals => &DEFAULT_TRAITS, - Self::Body => &DEFAULT_TRAITS, - Self::Case => &DEFAULT_TRAITS, - Self::Cases => &GROUP_TRAITS, - Self::Catch => &DEFAULT_TRAITS, - Self::Cell => &CONTAINER_TRAITS, - Self::Class => &CONTAINER_TRAITS, - Self::Clause => &DEFAULT_TRAITS, - Self::Cmd => &GROUP_TRAITS, - Self::Code => &GROUP_TRAITS, - Self::Cond => &DEFAULT_TRAITS, - Self::Constructor => &FUNCTION_TRAITS, - Self::Contract => &DEFAULT_TRAITS, - Self::Copy => &GROUP_TRAITS, - Self::Custom => &DEFAULT_TRAITS, - Self::Declarations => &GROUP_TRAITS, - Self::Decl => &DEFAULT_TRAITS, - Self::DefaultExport => &DEFAULT_TRAITS, - Self::Define => &DEFAULT_TRAITS, - Self::Directive => &DEFAULT_TRAITS, - Self::Either => &DEFAULT_TRAITS, - Self::Elif => &DEFAULT_TRAITS, - Self::Else => &DEFAULT_TRAITS, - Self::Enum => &ADDRESSABLE_CONTAINER_TRAITS, - Self::Env => &DEFAULT_TRAITS, - Self::Error => &DEFAULT_TRAITS, - Self::Except => &DEFAULT_TRAITS, - Self::Exports => &GROUP_TRAITS, - Self::Expose => &DEFAULT_TRAITS, - Self::Expression => &DEFAULT_TRAITS, - Self::Field => &PACKED_LEAF_TRAITS, - Self::Fields => &GROUP_TRAITS, - Self::File => &DEFAULT_TRAITS, - Self::Frame => &DEFAULT_TRAITS, - Self::Function => &FUNCTION_TRAITS, - Self::For => &DEFAULT_TRAITS, - Self::ForIn => &DEFAULT_TRAITS, - Self::ForOf => &DEFAULT_TRAITS, - Self::Form => &DEFAULT_TRAITS, - Self::Frontmatter => &DEFAULT_TRAITS, - Self::Group => &DEFAULT_TRAITS, - Self::GroupBy => &DEFAULT_TRAITS, - Self::Headers => &GROUP_TRAITS, - Self::Healthcheck => &DEFAULT_TRAITS, - Self::Html => &GROUP_TRAITS, - Self::Hunks => &GROUP_TRAITS, - Self::Hunk => &DEFAULT_TRAITS, - Self::If => &DEFAULT_TRAITS, - Self::Iface => &PRESERVE_CHILDREN_TRAITS, - Self::Impl => &CONTAINER_TRAITS, - Self::Imports => &IMPORTS_TRAITS, - Self::Includes => &GROUP_TRAITS, - Self::InlineFragment => &DEFAULT_TRAITS, - Self::Install => &DEFAULT_TRAITS, - Self::Interface => &PRESERVE_CHILDREN_TRAITS, - Self::Interpolation => &GROUP_TRAITS, - Self::Item => &DEFAULT_TRAITS, - Self::Join => &DEFAULT_TRAITS, - Self::Key => &PACKED_LEAF_TRAITS, - Self::KeyScripts => &DEFAULT_TRAITS, - Self::Label => &DEFAULT_TRAITS, - Self::Let => &DEFAULT_TRAITS, - Self::List => &DEFAULT_TRAITS, - Self::Loop => &DEFAULT_TRAITS, - Self::Macro => &DEFAULT_TRAITS, - Self::Map => &DEFAULT_TRAITS, - Self::Markdown => &DEFAULT_TRAITS, - Self::Match => &DEFAULT_TRAITS, - Self::Method => &DEFAULT_TRAITS, - Self::Methods => &GROUP_TRAITS, - Self::Module => &CONTAINER_TRAITS, - Self::Mustache => &DEFAULT_TRAITS, - Self::Object => &DEFAULT_TRAITS, - Self::Operation => &DEFAULT_TRAITS, - Self::Operator => &DEFAULT_TRAITS, - Self::Option => &DEFAULT_TRAITS, - Self::Options => &GROUP_TRAITS, - Self::OrderBy => &DEFAULT_TRAITS, - Self::Parameters => &GROUP_TRAITS, - Self::Preamble => &DEFAULT_TRAITS, - Self::Proc => &CONTAINER_TRAITS, - Self::Process => &CONTAINER_TRAITS, - Self::Project => &DEFAULT_TRAITS, - Self::Proto => &DEFAULT_TRAITS, - Self::Py => &DEFAULT_TRAITS, - Self::Python => &DEFAULT_TRAITS, - Self::Query => &DEFAULT_TRAITS, - Self::Receive => &DEFAULT_TRAITS, - Self::Recipe => &DEFAULT_TRAITS, - Self::Relations => &DEFAULT_TRAITS, - Self::Render => &DEFAULT_TRAITS, - Self::Return => &DEFAULT_TRAITS, - Self::Row => &PACKED_LEAF_TRAITS, - Self::Root => &DEFAULT_TRAITS, - Self::Rule => &DEFAULT_TRAITS, - Self::Schema => &DEFAULT_TRAITS, - Self::Script => &DEFAULT_TRAITS, - Self::ScriptModule => &DEFAULT_TRAITS, - Self::ScriptSetup => &DEFAULT_TRAITS, - Self::Section => &DEFAULT_TRAITS, - Self::Select => &DEFAULT_TRAITS, - Self::Setting => &DEFAULT_TRAITS, - Self::Shebang => &DEFAULT_TRAITS, - Self::Shell => &DEFAULT_TRAITS, - Self::Slot => &DEFAULT_TRAITS, - Self::Snippet => &DEFAULT_TRAITS, - Self::Source => &DEFAULT_TRAITS, - Self::Stage => &DEFAULT_TRAITS, - Self::StaticInit => &DEFAULT_TRAITS, - Self::Statements => &GROUP_TRAITS, - Self::Struct => &ADDRESSABLE_CONTAINER_TRAITS, - Self::Style => &DEFAULT_TRAITS, - Self::StyleScoped => &DEFAULT_TRAITS, - Self::Switch => &DEFAULT_TRAITS, - Self::Table => &DEFAULT_TRAITS, - Self::Tag => &DEFAULT_TRAITS, - Self::Target => &DEFAULT_TRAITS, - Self::Template => &DEFAULT_TRAITS, - Self::Text => &GROUP_TRAITS, - Self::Trait => &PRESERVE_CHILDREN_TRAITS, - Self::Translation => &DEFAULT_TRAITS, - Self::Try => &DEFAULT_TRAITS, - Self::Ts => &DEFAULT_TRAITS, - Self::Type => &ADDRESSABLE_CONTAINER_TRAITS, - Self::Typescript => &DEFAULT_TRAITS, - Self::Union => &DEFAULT_TRAITS, - Self::User => &DEFAULT_TRAITS, - Self::Val => &DEFAULT_TRAITS, - Self::Variable => &VARIABLE_TRAITS, - Self::Variant => &PACKED_LEAF_TRAITS, - Self::Variants => &DEFAULT_TRAITS, - Self::VersionGate => &DEFAULT_TRAITS, - Self::When => &DEFAULT_TRAITS, - Self::Where => &DEFAULT_TRAITS, - Self::While => &DEFAULT_TRAITS, - Self::With => &GROUP_TRAITS, - Self::Workdir => &DEFAULT_TRAITS, - Self::Conflict => &DEFAULT_TRAITS, - Self::Ours => &PACKED_LEAF_TRAITS, - Self::Theirs => &PACKED_LEAF_TRAITS, - Self::Chunk => &DEFAULT_TRAITS, - } - } - - pub fn path_segment(self, identifier: Option<&str>) -> String { - match identifier { - Some(identifier) => format!("{}_{identifier}", self.prefix()), - None => self.prefix().to_string(), - } - } - - pub fn from_sanitized_kind(kind: &str) -> Self { - match kind { - "add" => Self::Add, - "after" => Self::After, - "alias" => Self::Alias, - "algo" => Self::Algo, - "arg" => Self::Arg, - "argv" => Self::Argv, - "array" => Self::Array, - "at" => Self::At, - "attr" => Self::Attr, - "attr_expr" => Self::AttrExpr, - "attrs" => Self::Attrs, - "block" => Self::Block, - "block_if" => Self::BlockIf, - "block_locals" => Self::BlockLocals, - "body" => Self::Body, - "case" => Self::Case, - "cases" => Self::Cases, - "catch" => Self::Catch, - "cell" => Self::Cell, - "class" => Self::Class, - "clause" => Self::Clause, - "cmd" => Self::Cmd, - "code" => Self::Code, - "cond" => Self::Cond, - "constructor" => Self::Constructor, - "contract" => Self::Contract, - "copy" => Self::Copy, - "custom" => Self::Custom, - "decls" | "declarations" => Self::Declarations, - "decl" => Self::Decl, - "default_export" => Self::DefaultExport, - "define" => Self::Define, - "directive" => Self::Directive, - "either" => Self::Either, - "elif" => Self::Elif, - "else" => Self::Else, - "enum" => Self::Enum, - "env" => Self::Env, - "error" => Self::Error, - "except" => Self::Except, - "exports" => Self::Exports, - "expose" => Self::Expose, - "expr" | "expression" => Self::Expression, - "field" => Self::Field, - "fields" => Self::Fields, - "file" => Self::File, - "frame" => Self::Frame, - "fn" | "function" => Self::Function, - "for" => Self::For, - "for_in" => Self::ForIn, - "for_of" => Self::ForOf, - "form" => Self::Form, - "frontmatter" => Self::Frontmatter, - "group" => Self::Group, - "group_by" => Self::GroupBy, - "headers" => Self::Headers, - "healthcheck" => Self::Healthcheck, - "html" => Self::Html, - "hunks" => Self::Hunks, - "hunk" => Self::Hunk, - "if" => Self::If, - "iface" => Self::Iface, - "impl" => Self::Impl, - "imports" => Self::Imports, - "includes" => Self::Includes, - "inline_fragment" => Self::InlineFragment, - "install" => Self::Install, - "interface" => Self::Interface, - "interpolation" => Self::Interpolation, - "item" => Self::Item, - "join" => Self::Join, - "key" => Self::Key, - "key_scripts" => Self::KeyScripts, - "label" => Self::Label, - "let" => Self::Let, - "list" => Self::List, - "loop" => Self::Loop, - "macro" => Self::Macro, - "map" => Self::Map, - "markdown" => Self::Markdown, - "match" => Self::Match, - "meth" | "method" => Self::Method, - "methods" => Self::Methods, - "mod" | "module" => Self::Module, - "mustache" => Self::Mustache, - "object" => Self::Object, - "operation" => Self::Operation, - "operator" => Self::Operator, - "option" => Self::Option, - "options" => Self::Options, - "order_by" => Self::OrderBy, - "params" | "parameters" => Self::Parameters, - "preamble" => Self::Preamble, - "proc" => Self::Proc, - "process" => Self::Process, - "project" => Self::Project, - "proto" => Self::Proto, - "py" => Self::Py, - "python" => Self::Python, - "query" => Self::Query, - "receive" => Self::Receive, - "recipe" => Self::Recipe, - "relations" => Self::Relations, - "render" => Self::Render, - "ret" | "return" => Self::Return, - "row" => Self::Row, - "root" => Self::Root, - "rule" => Self::Rule, - "schema" => Self::Schema, - "script" => Self::Script, - "script_module" => Self::ScriptModule, - "script_setup" => Self::ScriptSetup, - "section" => Self::Section, - "select" => Self::Select, - "setting" => Self::Setting, - "shebang" => Self::Shebang, - "shell" => Self::Shell, - "slot" => Self::Slot, - "snippet" => Self::Snippet, - "source" => Self::Source, - "stage" => Self::Stage, - "static_init" => Self::StaticInit, - "stmts" | "statements" => Self::Statements, - "struct" => Self::Struct, - "style" => Self::Style, - "style_scoped" => Self::StyleScoped, - "switch" => Self::Switch, - "table" => Self::Table, - "tag" => Self::Tag, - "target" => Self::Target, - "template" => Self::Template, - "text" => Self::Text, - "trait" => Self::Trait, - "translation" => Self::Translation, - "try" => Self::Try, - "ts" => Self::Ts, - "type" => Self::Type, - "typescript" => Self::Typescript, - "union" => Self::Union, - "user" => Self::User, - "val" => Self::Val, - "var" | "variable" => Self::Variable, - "variant" => Self::Variant, - "variants" => Self::Variants, - "version_gate" => Self::VersionGate, - "when" => Self::When, - "where" => Self::Where, - "while" => Self::While, - "with" => Self::With, - "workdir" => Self::Workdir, - "conflict" => Self::Conflict, - "ours" => Self::Ours, - "theirs" => Self::Theirs, - "chunk" => Self::Chunk, - _ => Self::Chunk, - } - } -} diff --git a/crates/pi-natives/src/chunk/mod.rs b/crates/pi-natives/src/chunk/mod.rs deleted file mode 100644 index c7e3f77a6..000000000 --- a/crates/pi-natives/src/chunk/mod.rs +++ /dev/null @@ -1,3443 +0,0 @@ -//! Chunk-tree parsing powered by tree-sitter with best-effort structural -//! grouping. -//! -//! The module is split into: -//! - `types` — napi-exported data structures -//! - `common` — shared helpers used by all classifiers -//! - `defaults` — default classification logic (language-agnostic node kinds) -//! - `classify` — `LangClassifier` trait and dispatch -//! - `ast_*` — per-language classifier implementations - -mod atom_list; -mod classify; -pub(crate) mod common; -pub(crate) mod conflict; -mod defaults; -pub(crate) mod edit; -pub(crate) mod indent; -mod render; -pub(crate) mod resolve; -mod schema; -mod shape; -pub(crate) mod state; -pub mod types; - -pub mod kind; - -// Per-language classifiers -mod ast_astro; -mod ast_bash_make_diff; -mod ast_c_cpp_objc; -mod ast_clojure; -mod ast_cmake; -mod ast_csharp_java; -mod ast_css; -mod ast_data_formats; -mod ast_dockerfile; -mod ast_elixir; -mod ast_erlang; -mod ast_go; -mod ast_graphql; -mod ast_haskell_scala; -mod ast_html_xml; -mod ast_ini; -pub(crate) mod ast_ipynb; -mod ast_js_ts; -mod ast_just; -mod ast_markup; -mod ast_misc; -mod ast_nix_hcl; -mod ast_ocaml; -mod ast_perl; -mod ast_powershell; -mod ast_proto; -mod ast_python; -mod ast_r; -mod ast_ruby_lua; -mod ast_rust; -mod ast_sql; -mod ast_svelte; -mod ast_tlaplus; -mod ast_vue; - -use std::collections::HashMap; - -use ast_grep_core::tree_sitter::LanguageExt; -use napi::{Error, Result}; -use napi_derive::napi; -use tree_sitter::{Node, Parser, Tree}; -use xxhash_rust::xxh64::xxh64; - -use self::{ - classify::{LangClassifier, classifier_for, classify_with_defaults, structural_overrides}, - common::*, - kind::ChunkKind, -}; -pub use self::{ - state::ChunkState, - types::{ChunkNode, ChunkTree}, -}; -use crate::{chunk::types::ChunkAnchorStyle, language::SupportLang}; - -// ── Napi exports ───────────────────────────────────────────────────────── - -/// Format one chunk anchor string for a node at `depth` using `style` and -/// optional checksum omission. -#[napi] -pub fn format_anchor( - name: String, - checksum: String, - style: ChunkAnchorStyle, - omit_checksum: Option, -) -> String { - style - .with_omit_checksum(omit_checksum.unwrap_or(false)) - .render("", name.as_str(), checksum.as_str()) -} - -// ── Core build logic ───────────────────────────────────────────────────── - -pub(crate) fn build_chunk_tree(source: &str, language: &str) -> Result { - let normalized_language = language.trim().to_ascii_lowercase(); - let total_lines = total_line_count(source); - let root_checksum = chunk_checksum(source.as_bytes()); - - // Notebooks (`.ipynb`) are parsed by `ChunkStateInner::parse`, which - // converts the JSON file to a *virtual source* and then re-enters - // `build_chunk_tree` with the `ipynb` language tag. When we arrive here - // with that tag, `source` is the virtual concatenated cell text and the - // per-cell sub-trees are built via `ast_ipynb`. - if normalized_language == "ipynb" { - return ast_ipynb::build_notebook_tree_from_virtual(source, "python") - .map_err(Error::from_reason); - } - let Some(chunk_lang) = resolve_chunk_lang(normalized_language.as_str()) else { - return Ok(build_blank_line_tree(source, language.to_string(), total_lines, root_checksum)); - }; - - let _schema_language = schema::enter_language(chunk_lang.canonical_name()); - let classifier = classifier_for(normalized_language.as_str()); - let tree = parse_tree(source, chunk_lang)?; - let root = tree.root_node(); - let (parse_errors, parse_error_lines) = count_parse_errors(root); - let mut acc = ChunkAccumulator::default(); - let mut root_children = - collect_children_for_context(root, ChunkContext::Root, source, classifier) - .into_iter() - .map(|candidate| build_chunk(candidate, "", source, &mut acc, classifier)) - .collect::>>()?; - - classifier.post_process(&mut acc.chunks, &mut root_children, source); - - insert_preamble_chunk(source, &mut acc.chunks, &mut root_children); - - acc.chunks.insert(0, ChunkNode { - path: String::new(), - identifier: None, - kind: ChunkKind::Root, - leaf: false, - virtual_content: None, - parent_path: None, - children: root_children.clone(), - signature: None, - start_line: u32::from(total_lines != 0), - end_line: total_lines as u32, - line_count: total_lines as u32, - start_byte: 0, - end_byte: source.len() as u32, - checksum_start_byte: 0, - prologue_end_byte: Some(0), - epilogue_start_byte: Some(source.len() as u32), - checksum: root_checksum.clone(), - error: false, - indent: 0, - indent_char: String::new(), - group: false, - }); - - Ok(ChunkTree { - language: normalized_language, - checksum: root_checksum, - line_count: total_lines as u32, - parse_errors: parse_errors as u32, - parse_error_lines, - fallback: false, - root_path: String::new(), - root_children, - chunks: acc.chunks, - }) -} - -/// Smallest chunk path containing `line` (1-based file line), preferring the -/// innermost leaf when multiple chunks overlap. -pub(crate) fn line_to_chunk_path(tree: &ChunkTree, line: u32) -> Option { - if line == 0 { - return None; - } - - if let Some(chunk) = tree - .chunks - .iter() - .filter(|chunk| { - chunk.leaf && !chunk.path.is_empty() && chunk.start_line <= line && line <= chunk.end_line - }) - .min_by_key(|chunk| chunk.line_count) - { - return Some(chunk.path.clone()); - } - - tree - .chunks - .iter() - .filter(|chunk| chunk.start_line <= line && line <= chunk.end_line) - .min_by_key(|chunk| chunk.line_count) - .map(|chunk| chunk.path.clone()) -} - -fn parse_tree(source: &str, language: SupportLang) -> Result { - let mut parser = Parser::new(); - let ts_language = language.get_ts_language(); - parser - .set_language(&ts_language) - .map_err(|err| Error::from_reason(format!("Failed to set parser language: {err}")))?; - parser - .parse(source, None) - .ok_or_else(|| Error::from_reason("Tree-sitter failed to parse source".to_string())) -} - -fn build_blank_line_tree( - source: &str, - language: String, - total_lines: usize, - checksum: String, -) -> ChunkTree { - let mut chunks = vec![ChunkNode { - path: String::new(), - identifier: None, - kind: ChunkKind::Root, - leaf: false, - virtual_content: None, - parent_path: None, - children: Vec::new(), - signature: None, - start_line: u32::from(total_lines != 0), - end_line: total_lines as u32, - line_count: total_lines as u32, - start_byte: 0, - end_byte: source.len() as u32, - checksum_start_byte: 0, - prologue_end_byte: Some(0), - epilogue_start_byte: Some(source.len() as u32), - checksum: checksum.clone(), - error: false, - indent: 0, - indent_char: String::new(), - group: false, - }]; - let line_starts = line_start_offsets(source); - let mut root_children = Vec::new(); - let mut seen_names = HashMap::::new(); - let lines: Vec<&str> = if source.is_empty() { - Vec::new() - } else { - source.split('\n').collect() - }; - let mut start_line = 0usize; - - while start_line < lines.len() { - while start_line < lines.len() && lines[start_line].trim().is_empty() { - start_line += 1; - } - if start_line >= lines.len() { - break; - } - - let mut end_line = start_line; - while end_line + 1 < lines.len() && !lines[end_line + 1].trim().is_empty() { - end_line += 1; - } - - let name = infer_fallback_block_name(lines[start_line], &mut seen_names); - let start_byte = line_starts[start_line]; - let end_byte = line_end_offset(source, &line_starts, end_line); - root_children.push(name.clone()); - chunks.push(ChunkNode { - path: name.clone(), - identifier: Some(name.clone()), - kind: ChunkKind::Chunk, - leaf: true, - virtual_content: None, - parent_path: Some(String::new()), - children: Vec::new(), - signature: None, - start_line: (start_line + 1) as u32, - end_line: (end_line + 1) as u32, - line_count: (end_line - start_line + 1) as u32, - start_byte: start_byte as u32, - end_byte: end_byte as u32, - checksum_start_byte: start_byte as u32, - prologue_end_byte: None, - epilogue_start_byte: None, - checksum: chunk_checksum( - source - .as_bytes() - .get(start_byte..end_byte) - .unwrap_or_default(), - ), - error: false, - indent: 0, - indent_char: String::new(), - group: false, - }); - start_line = end_line + 1; - } - - if let Some(root) = chunks.first_mut() { - root.children.clone_from(&root_children); - } - - ChunkTree { - language, - checksum, - line_count: total_lines as u32, - parse_errors: 0, - parse_error_lines: Vec::new(), - fallback: true, - root_path: String::new(), - root_children, - chunks, - } -} - -// ── Chunk building ─────────────────────────────────────────────────────── - -fn build_chunk( - candidate: RawChunkCandidate<'_>, - parent_path: &str, - source: &str, - acc: &mut ChunkAccumulator, - classifier: &dyn classify::LangClassifier, -) -> Result { - let segment = candidate.kind.path_segment(candidate.identifier.as_deref()); - let path = if parent_path.is_empty() { - segment - } else { - format!("{parent_path}.{segment}") - }; - let line_count = candidate - .range_end_line - .saturating_sub(candidate.range_start_line) - + 1; - let checksum = chunk_checksum( - source - .as_bytes() - .get(candidate.checksum_start_byte..candidate.range_end_byte) - .unwrap_or_default(), - ); - let recurse = candidate.recurse; - let injected = candidate.injected; - let chunk_start = candidate.range_start_byte; - let mut chunk_end = candidate.range_end_byte; - let region_boundaries = candidate.region_node.map(|region_node| { - let (pro_end, epi_start) = - compute_body_inner_boundaries(source, region_node.start_byte(), region_node.end_byte()); - // For indent-based languages (Python, Ruby, etc.) the body boundary - // computation may extend past the tree-sitter node to include a - // trailing newline that logically terminates the last body line. - // When this happens and the source byte at chunk_end is indeed a - // newline, extend the chunk's end_byte to match so that: - // - ~ covers complete lines including their trailing newline - // - ^ and ~ are the only supported sub-chunk regions - if epi_start > chunk_end - && epi_start <= source.len() - && source.as_bytes().get(chunk_end) == Some(&b'\n') - { - chunk_end = epi_start; - } - let pro_end = pro_end.max(chunk_start).min(chunk_end); - let epi_start = epi_start.max(pro_end).min(chunk_end); - (pro_end, epi_start) - }); - let child_candidates = recurse - .map(|recurse| { - collect_children_for_context(recurse.node, recurse.context, source, classifier) - }) - .unwrap_or_default(); - let recurse_parse_errors = recurse.map_or(0, |recurse| count_parse_errors(recurse.node).0); - let has_injected_children = injected.is_some(); - let should_collapse = !has_injected_children - && !classifier.preserve_children(&candidate, &child_candidates) - && recurse.is_some() - && recurse_parse_errors == 0 - && should_collapse_trivial_children(&candidate, &child_candidates); - let always_recurse = !candidate.groupable && !child_candidates.is_empty(); - // A child that already committed to splitting further (force_recurse + - // recurse) should always pull its parent along. Otherwise a small - // wrapper parent would keep the child's sub-structure hidden just - // because the wrapper itself fits under LEAF_THRESHOLD — e.g. a tiny - // function whose body is one JSX return. - let has_forced_child = child_candidates - .iter() - .any(|c| c.force_recurse && c.recurse.is_some()); - let should_recurse = !has_injected_children - && !candidate.error - && recurse.is_some() - && !should_collapse - && (candidate.force_recurse - || always_recurse - || recurse_parse_errors > 0 - || has_forced_child - || (line_count > *LEAF_THRESHOLD - && recursion_narrows_scope(line_count, &child_candidates))); - let children = if let Some(injected) = injected { - translate_injected_subtree(path.as_str(), injected, source, acc)? - } else if should_recurse { - child_candidates - .into_iter() - .map(|child| build_chunk(child, path.as_str(), source, acc, classifier)) - .collect::>>()? - } else { - Vec::new() - }; - - let leaf = children.is_empty() && (!candidate.force_recurse || should_collapse); - let (indent, indent_char) = detect_indent(source, candidate.range_start_byte); - acc.chunks.push(ChunkNode { - path: path.clone(), - identifier: candidate.identifier, - kind: candidate.kind, - leaf, - virtual_content: None, - parent_path: Some(parent_path.to_string()), - children, - signature: candidate.signature, - start_line: candidate.range_start_line as u32, - end_line: candidate.range_end_line as u32, - line_count: line_count as u32, - start_byte: candidate.range_start_byte as u32, - end_byte: chunk_end as u32, - checksum_start_byte: candidate.checksum_start_byte as u32, - prologue_end_byte: region_boundaries.map(|(start, _)| start as u32), - epilogue_start_byte: region_boundaries.map(|(_, end)| end as u32), - checksum, - error: candidate.error, - indent, - indent_char, - group: candidate.groupable, - }); - Ok(path) -} - -fn translate_injected_subtree( - parent_path: &str, - injected: InjectedChunkSpec<'_>, - source: &str, - acc: &mut ChunkAccumulator, -) -> Result> { - let content_start = injected.content_node.start_byte(); - let content_end = injected.content_node.end_byte(); - let content = node_text(source, content_start, content_end); - if content.is_empty() { - return Ok(Vec::new()); - } - - let sub_tree = build_chunk_tree(content, injected.language.canonical_name())?; - let content_line_shift = injected.content_node.start_position().row as u32; - let mut translated_root_children = Vec::new(); - - for sub_chunk in sub_tree.chunks.into_iter().skip(1) { - let translated_path = format!("{parent_path}.{}", sub_chunk.path); - let translated_parent = match sub_chunk.parent_path.as_deref() { - Some("") | None => Some(parent_path.to_string()), - Some(other) => Some(format!("{parent_path}.{other}")), - }; - let translated_children = sub_chunk - .children - .iter() - .map(|child| format!("{parent_path}.{child}")) - .collect(); - acc.chunks.push(ChunkNode { - path: translated_path, - identifier: sub_chunk.identifier, - kind: sub_chunk.kind, - leaf: sub_chunk.leaf, - virtual_content: sub_chunk.virtual_content, - parent_path: translated_parent, - children: translated_children, - signature: sub_chunk.signature, - start_line: sub_chunk.start_line.saturating_add(content_line_shift), - end_line: sub_chunk.end_line.saturating_add(content_line_shift), - line_count: sub_chunk.line_count, - start_byte: sub_chunk.start_byte.saturating_add(content_start as u32), - end_byte: sub_chunk.end_byte.saturating_add(content_start as u32), - checksum_start_byte: sub_chunk - .checksum_start_byte - .saturating_add(content_start as u32), - prologue_end_byte: sub_chunk - .prologue_end_byte - .map(|byte| byte.saturating_add(content_start as u32)), - epilogue_start_byte: sub_chunk - .epilogue_start_byte - .map(|byte| byte.saturating_add(content_start as u32)), - checksum: sub_chunk.checksum, - error: sub_chunk.error, - indent: sub_chunk.indent, - indent_char: sub_chunk.indent_char, - group: sub_chunk.group, - }); - } - - for root_child in sub_tree.root_children { - translated_root_children.push(format!("{parent_path}.{root_child}")); - } - - Ok(translated_root_children) -} - -// ── Child collection ───────────────────────────────────────────────────── - -pub(crate) fn collect_children_for_context<'tree>( - container: Node<'tree>, - context: ChunkContext, - source: &str, - classifier: &dyn LangClassifier, -) -> Vec> { - let named_children_list = children_for_context(container, context, classifier); - let overrides = structural_overrides(classifier); - let mut raw = Vec::new(); - - for (index, child) in named_children_list.iter().enumerate() { - let is_error_node = child.is_error() || child.kind() == "ERROR"; - let is_skippable_trivia = - !is_error_node && is_trivia_for_classifier(*child, classifier, overrides); - let is_absorbable_attr = !is_error_node - && (is_absorbable_attribute(child.kind()) - || overrides.is_absorbable_attr(child.kind()) - || classifier.is_absorbable_attr(child.kind())); - let is_skipped = !is_error_node && classifier.should_skip_child(child.kind()); - if is_skipped - || is_skippable_trivia - || is_absorbable_attr - || (child.is_missing() && !is_error_node) - { - continue; - } - - let mut candidate = classify_node(*child, context, source, classifier); - attach_leading_trivia(&mut candidate, &named_children_list, index, classifier); - raw.push(candidate); - } - - group_candidates(raw) -} - -fn children_for_context<'tree>( - container: Node<'tree>, - context: ChunkContext, - classifier: &dyn LangClassifier, -) -> Vec> { - match context { - ChunkContext::Root => flatten_root_children(container, classifier), - ChunkContext::ClassBody | ChunkContext::FunctionBody => named_children(container), - } -} - -fn flatten_root_children<'tree>( - container: Node<'tree>, - classifier: &dyn LangClassifier, -) -> Vec> { - let children = named_children(container); - let overrides = structural_overrides(classifier); - if children.len() == 1 && is_root_wrapper_for_classifier(children[0], classifier, overrides) { - return flatten_root_children(children[0], classifier); - } - // When a root wrapper's only non-trivia child is another wrapper, - // flatten through it. Handles YAML's `document` containing a leading - // comment alongside a single `block_node`. - if children.len() > 1 { - let non_trivia: Vec<_> = children - .iter() - .filter(|child| !is_trivia_for_classifier(**child, classifier, overrides)) - .collect(); - if non_trivia.len() == 1 - && is_root_wrapper_for_classifier(*non_trivia[0], classifier, overrides) - { - return flatten_root_children(*non_trivia[0], classifier); - } - } - children -} - -fn classify_node<'tree>( - node: Node<'tree>, - context: ChunkContext, - source: &str, - classifier: &dyn LangClassifier, -) -> RawChunkCandidate<'tree> { - classify_with_defaults(classifier, context, node, source) -} - -fn attach_leading_trivia<'tree>( - candidate: &mut RawChunkCandidate<'tree>, - named_children_list: &[Node<'tree>], - index: usize, - classifier: &dyn LangClassifier, -) { - let overrides = structural_overrides(classifier); - let mut cursor = index; - while cursor > 0 { - let prev = named_children_list[cursor - 1]; - if !is_trivia_for_classifier(prev, classifier, overrides) - && !is_absorbable_attribute(prev.kind()) - && !overrides.is_absorbable_attr(prev.kind()) - && !classifier.is_absorbable_attr(prev.kind()) - { - break; - } - - let prev_end_line = prev.end_position().row + 1; - if candidate.range_start_line > prev_end_line + 1 { - break; - } - - candidate.range_start_byte = prev.start_byte(); - candidate.range_start_line = prev.start_position().row + 1; - if prev.kind() == "comment" { - candidate.has_leading_comment = true; - } - cursor -= 1; - } -} - -fn is_trivia_for_classifier( - node: Node<'_>, - classifier: &dyn LangClassifier, - overrides: classify::StructuralOverrides, -) -> bool { - let kind = node.kind(); - ((is_trivia_node(node) || classifier.is_trivia(kind)) - && !overrides.preserves_trivia(kind) - && !classifier.preserve_trivia(kind)) - || (overrides.is_extra_trivia(kind) - && !overrides.preserves_trivia(kind) - && !classifier.preserve_trivia(kind)) -} - -fn is_root_wrapper_for_classifier( - node: Node<'_>, - classifier: &dyn LangClassifier, - overrides: classify::StructuralOverrides, -) -> bool { - let kind = node.kind(); - if overrides.preserves_root_wrapper(kind) || classifier.preserve_root_wrapper(kind) { - return false; - } - overrides.is_extra_root_wrapper(kind) - || classifier.is_root_wrapper(kind) - || is_root_wrapper_node(node) -} - -// ── Grouping / deduplication ───────────────────────────────────────────── - -fn group_candidates(candidates: Vec>) -> Vec> { - let mut grouped: Vec> = Vec::new(); - - for candidate in candidates { - if let Some(last) = grouped.last_mut() { - let last_line_count = line_span(last.range_start_line, last.range_end_line); - let next_line_count = line_span(candidate.range_start_line, candidate.range_end_line); - let can_merge = last.groupable - && candidate.groupable - && last.kind == candidate.kind - && last.identifier == candidate.identifier - && !candidate.has_leading_comment - && candidate.range_start_line <= last.range_end_line + 1 - && last_line_count + next_line_count <= *MAX_CHUNK_LINES; - if can_merge { - last.range_end_byte = candidate.range_end_byte; - last.range_end_line = candidate.range_end_line; - continue; - } - } - grouped.push(candidate); - } - - assign_unique_names(grouped) -} - -/// Truncate a chunk identifier to at most `MAX_IDENT_CHARS` characters for -/// compact path segments. Trailing underscores left by mid-word truncation -/// are stripped. -fn truncate_path_name(name: &str) -> String { - const MAX_IDENT_CHARS: usize = 3; - if name.len() <= MAX_IDENT_CHARS { - return name.to_string(); - } - let end = name - .char_indices() - .nth(MAX_IDENT_CHARS) - .map_or(name.len(), |(idx, _)| idx); - name[..end].trim_end_matches('_').to_string() -} - -fn assign_unique_names(mut candidates: Vec>) -> Vec> { - // Truncate identifiers for path brevity before grouping. - for candidate in &mut candidates { - candidate.identifier = candidate - .identifier - .take() - .map(|id| truncate_path_name(&id)); - } - - let mut totals = HashMap::::new(); - for candidate in &candidates { - let key = candidate.kind.path_segment(candidate.identifier.as_deref()); - *totals.entry(key).or_insert(0) += 1; - } - let mut seen = HashMap::::new(); - - for candidate in &mut candidates { - let key = candidate.kind.path_segment(candidate.identifier.as_deref()); - let count = seen.entry(key.clone()).or_insert(0); - *count += 1; - let occurrence = *count; - let total = *totals.get(key.as_str()).unwrap_or(&1); - - candidate.identifier = match candidate.name_style { - NameStyle::Error => { - if total > 1 { - Some(occurrence.to_string()) - } else { - None - } - }, - NameStyle::Named => { - if total > 1 { - match candidate.identifier.as_deref() { - Some(identifier) => Some(format!("{identifier}_{occurrence}")), - None => Some(occurrence.to_string()), - } - } else { - candidate.identifier.clone() - } - }, - NameStyle::Group => { - if total == 1 || occurrence == 1 { - candidate.identifier.clone() - } else { - match candidate.identifier.as_deref() { - Some(identifier) => Some(format!("{identifier}_{occurrence}")), - None => Some(occurrence.to_string()), - } - } - }, - }; - } - - candidates -} - -// ── Collapse heuristics ────────────────────────────────────────────────── - -/// Returns `true` when splitting a parent into children actually provides -/// meaningful scope narrowing. Recursion is only worthwhile if addressing -/// the largest child saves at least `PI_CHUNK_MIN_SAVINGS` lines compared -/// to addressing the parent directly — or when a child already wants to -/// recurse further (in which case the scope narrowing happens at the next -/// level and should not be cut off here). -fn recursion_narrows_scope(parent_lines: usize, children: &[RawChunkCandidate<'_>]) -> bool { - if children.is_empty() { - return false; - } - // If any child has already been marked as needing its own recursion, - // always recurse through it. Otherwise a wrapper parent whose single - // child covers almost the whole body (function -> return_statement, - // arrow body -> JSX element, etc.) would fail the simple savings - // heuristic even though splitting down the chain exposes real - // structure. - if children - .iter() - .any(|c| c.force_recurse && c.recurse.is_some()) - { - return true; - } - let max_child_lines = children - .iter() - .map(|c| line_span(c.range_start_line, c.range_end_line)) - .max() - .unwrap_or(0); - parent_lines.saturating_sub(max_child_lines) >= *MIN_RECURSE_SAVINGS -} - -fn should_collapse_trivial_children( - parent: &RawChunkCandidate<'_>, - children: &[RawChunkCandidate<'_>], -) -> bool { - if children.is_empty() { - return false; - } - - let has_addressable_leaf_members = children.iter().all(|child| child.kind.traits().packed); - if has_addressable_leaf_members && parent.kind.traits().has_addressable_members { - return false; - } - if parent.kind.traits().always_preserve_children { - return false; - } - - if children.len() == 1 && is_collapsible_flat_child(&children[0]) { - return true; - } - - if !children.iter().all(is_collapsible_flat_child) { - return false; - } - let total_lines: usize = children - .iter() - .map(|c| line_span(c.range_start_line, c.range_end_line)) - .sum(); - total_lines <= *LEAF_THRESHOLD -} - -const fn is_trivial_child_candidate(candidate: &RawChunkCandidate<'_>) -> bool { - !candidate.error - && !candidate.has_leading_comment - && candidate.injected.is_none() - && candidate.recurse.is_none() - && line_span(candidate.range_start_line, candidate.range_end_line) == 1 -} - -const fn is_collapsible_flat_child(candidate: &RawChunkCandidate<'_>) -> bool { - (candidate.groupable || is_trivial_child_candidate(candidate)) - && !candidate.error - && !candidate.has_leading_comment - && candidate.injected.is_none() - && candidate.recurse.is_none() -} - -// ── Utility ────────────────────────────────────────────────────────────── - -fn count_parse_errors(node: Node<'_>) -> (usize, Vec) { - let mut count = 0; - let mut lines = Vec::new(); - collect_parse_errors(node, &mut count, &mut lines); - lines.sort_unstable(); - lines.dedup(); - (count, lines) -} - -fn collect_parse_errors(node: Node<'_>, count: &mut usize, lines: &mut Vec) { - if node.is_error() || node.is_missing() || node.kind() == "ERROR" { - *count += 1; - lines.push(node.start_position().row as u32 + 1); - } - for child in named_children(node) { - collect_parse_errors(child, count, lines); - } -} - -fn resolve_chunk_lang(language: &str) -> Option { - SupportLang::from_alias(language) -} - -fn infer_fallback_block_name(first_line: &str, seen: &mut HashMap) -> String { - let trimmed = first_line.trim(); - let base = trimmed - .split(|c: char| !c.is_alphanumeric() && c != '_' && c != '-') - .next() - .unwrap_or("") - .trim_matches(|c: char| !c.is_alphanumeric() && c != '_'); - let base = if base.is_empty() { "chunk" } else { base }; - let count = seen.entry(base.to_string()).or_insert(0); - *count += 1; - if *count == 1 { - base.to_string() - } else { - format!("{base}#{count}") - } -} - -pub(crate) fn line_start_offsets(source: &str) -> Vec { - let mut starts = vec![0usize]; - for (index, byte) in source.bytes().enumerate() { - if byte == b'\n' { - starts.push(index + 1); - } - } - starts -} - -fn line_end_offset(source: &str, line_starts: &[usize], line_index: usize) -> usize { - if line_index + 1 < line_starts.len() { - line_starts[line_index + 1] - } else { - source.len() - } -} - -/// 647 single-token BPE bigrams (lowercase) for hashline anchors. Mirrors -/// `packages/coding-agent/src/edit/line-hash.ts::HASHLINE_BIGRAMS`. -pub(crate) const HASHLINE_BIGRAMS: [&str; 647] = [ - "aa", "ab", "ac", "ad", "ae", "af", "ag", "ah", "ai", "aj", "ak", "al", "am", "an", "ao", "ap", - "aq", "ar", "as", "at", "au", "av", "aw", "ax", "ay", "az", "ba", "bb", "bc", "bd", "be", "bf", - "bg", "bh", "bi", "bj", "bk", "bl", "bm", "bn", "bo", "bp", "br", "bs", "bt", "bu", "bv", "bw", - "bx", "by", "bz", "ca", "cb", "cc", "cd", "ce", "cf", "cg", "ch", "ci", "cj", "ck", "cl", "cm", - "cn", "co", "cp", "cq", "cr", "cs", "ct", "cu", "cv", "cw", "cx", "cy", "cz", "da", "db", "dc", - "dd", "de", "df", "dg", "dh", "di", "dj", "dk", "dl", "dm", "dn", "do", "dp", "dq", "dr", "ds", - "dt", "du", "dv", "dw", "dx", "dy", "dz", "ea", "eb", "ec", "ed", "ee", "ef", "eg", "eh", "ei", - "ej", "ek", "el", "em", "en", "eo", "ep", "eq", "er", "es", "et", "eu", "ev", "ew", "ex", "ey", - "ez", "fa", "fb", "fc", "fd", "fe", "ff", "fg", "fh", "fi", "fj", "fk", "fl", "fm", "fn", "fo", - "fp", "fq", "fr", "fs", "ft", "fu", "fv", "fw", "fx", "fy", "fz", "ga", "gb", "gc", "gd", "ge", - "gf", "gg", "gh", "gi", "gj", "gl", "gm", "gn", "go", "gp", "gr", "gs", "gt", "gu", "gv", "gw", - "gx", "gy", "gz", "ha", "hb", "hc", "hd", "he", "hf", "hg", "hh", "hi", "hj", "hk", "hl", "hm", - "hn", "ho", "hp", "hq", "hr", "hs", "ht", "hu", "hv", "hw", "hx", "hy", "hz", "ia", "ib", "ic", - "id", "ie", "if", "ig", "ih", "ii", "ij", "ik", "il", "im", "in", "io", "ip", "iq", "ir", "is", - "it", "iu", "iv", "iw", "ix", "iy", "iz", "ja", "jb", "jc", "jd", "je", "jf", "jg", "jh", "ji", - "jj", "jk", "jl", "jm", "jn", "jo", "jp", "jq", "jr", "js", "jt", "ju", "jw", "jx", "jy", "ka", - "kb", "kc", "kd", "ke", "kf", "kg", "kh", "ki", "kj", "kk", "kl", "km", "kn", "ko", "kp", "kr", - "ks", "kt", "ku", "kv", "kw", "kx", "ky", "la", "lb", "lc", "ld", "le", "lf", "lg", "lh", "li", - "lj", "lk", "ll", "lm", "ln", "lo", "lp", "lr", "ls", "lt", "lu", "lv", "lw", "lx", "ly", "lz", - "ma", "mb", "mc", "md", "me", "mf", "mg", "mh", "mi", "mj", "mk", "ml", "mm", "mn", "mo", "mp", - "mq", "mr", "ms", "mt", "mu", "mv", "mw", "mx", "my", "mz", "na", "nb", "nc", "nd", "ne", "nf", - "ng", "nh", "ni", "nj", "nk", "nl", "nm", "nn", "no", "np", "nr", "ns", "nt", "nu", "nv", "nw", - "nx", "ny", "nz", "oa", "ob", "oc", "od", "oe", "of", "og", "oh", "oi", "oj", "ok", "ol", "om", - "on", "oo", "op", "oq", "or", "os", "ot", "ou", "ov", "ow", "ox", "oy", "oz", "pa", "pb", "pc", - "pd", "pe", "pf", "pg", "ph", "pi", "pj", "pk", "pl", "pm", "pn", "po", "pp", "pq", "pr", "ps", - "pt", "pu", "pv", "pw", "px", "py", "pz", "qa", "qb", "qc", "qd", "qe", "qh", "qi", "ql", "qm", - "qn", "qo", "qp", "qq", "qr", "qs", "qt", "qu", "qw", "qx", "qy", "ra", "rb", "rc", "rd", "re", - "rf", "rg", "rh", "ri", "rk", "rl", "rm", "rn", "ro", "rp", "rq", "rr", "rs", "rt", "ru", "rv", - "rw", "rx", "ry", "rz", "sa", "sb", "sc", "sd", "se", "sf", "sg", "sh", "si", "sj", "sk", "sl", - "sm", "sn", "so", "sp", "sq", "sr", "ss", "st", "su", "sv", "sw", "sx", "sy", "sz", "ta", "tb", - "tc", "td", "te", "tf", "tg", "th", "ti", "tj", "tk", "tl", "tm", "tn", "to", "tp", "tr", "ts", - "tt", "tu", "tv", "tw", "tx", "ty", "tz", "ua", "ub", "uc", "ud", "ue", "uf", "ug", "uh", "ui", - "uj", "uk", "ul", "um", "un", "uo", "up", "uq", "ur", "us", "ut", "uu", "uv", "uw", "ux", "uy", - "uz", "va", "vb", "vc", "vd", "ve", "vf", "vg", "vh", "vi", "vj", "vk", "vl", "vm", "vn", "vo", - "vp", "vq", "vr", "vs", "vt", "vu", "vv", "vw", "vx", "vy", "vz", "wa", "wb", "wc", "wd", "we", - "wf", "wg", "wh", "wi", "wj", "wk", "wl", "wm", "wn", "wo", "wp", "wr", "ws", "wt", "wu", "wv", - "ww", "wx", "wy", "xa", "xb", "xc", "xd", "xe", "xf", "xh", "xi", "xl", "xm", "xn", "xo", "xp", - "xr", "xs", "xt", "xu", "xx", "xy", "xz", "ya", "yb", "yc", "yd", "ye", "yf", "yg", "yh", "yi", - "yj", "yk", "yl", "ym", "yn", "yo", "yp", "yr", "ys", "yt", "yu", "yv", "yw", "yx", "yy", "yz", - "za", "zb", "zc", "zd", "ze", "zf", "zg", "zh", "zi", "zk", "zl", "zm", "zn", "zo", "zp", "zr", - "zs", "zt", "zu", "zw", "zx", "zy", "zz", -]; - -/// 40 common English BPE bigrams (lowercase) for chunk checksums. Mirrors -/// `packages/coding-agent/src/edit/line-hash.ts::CHUNK_BIGRAMS`. -/// Independent of `HASHLINE_BIGRAMS` — chunk checksum format is -/// `path#bigram1bigram2` (4 chars from a 1600-code namespace) and is unaffected -/// by the line-anchor format. Order is stable forever — changing it invalidates -/// every saved chunk path. -pub(crate) const CHUNK_BIGRAMS: [&str; 40] = [ - "th", "he", "in", "er", "an", "re", "on", "at", "en", "nd", "ti", "es", "or", "te", "of", "ed", - "is", "it", "al", "ar", "st", "to", "nt", "ng", "se", "ha", "as", "ou", "io", "le", "ve", "co", - "me", "de", "hi", "ri", "ro", "ic", "ne", "ea", -]; - -/// Encode a chunk checksum as 2 BPE bigrams (4 lowercase chars). -/// -/// Maps `xxh64(bytes) % (40*40)` into `BIGRAMS[i0] + BIGRAMS[i1]` where -/// `i0 = h % 40` and `i1 = (h / 40) % 40`. Total namespace: 1,600 codes. -pub(crate) fn chunk_checksum(bytes: &[u8]) -> String { - let h = xxh64(bytes, 0); - let n = CHUNK_BIGRAMS.len() as u64; - let i0 = (h % n) as usize; - let i1 = ((h / n) % n) as usize; - let mut out = String::with_capacity(4); - out.push_str(CHUNK_BIGRAMS[i0]); - out.push_str(CHUNK_BIGRAMS[i1]); - out -} - -/// When the first structural chunk begins after line 1, insert a leaf chunk -/// `preamble` covering leading comments/whitespace so they stay -/// addressable via chunk paths (not only raw line ops). -fn insert_preamble_chunk( - source: &str, - chunks: &mut Vec, - root_children: &mut Vec, -) { - if root_children.is_empty() { - return; - } - if chunks.iter().any(|c| c.path == "preamble") || root_children.iter().any(|p| p == "preamble") { - return; - } - let mut min_start = u32::MAX; - for path in root_children.iter() { - if let Some(chunk) = chunks.iter().find(|c| c.path == *path) { - min_start = min_start.min(chunk.start_line); - } - } - if min_start <= 1 { - return; - } - let line_starts = line_start_offsets(source); - let start_byte: u32 = 0; - let end_byte = line_starts - .get(min_start as usize - 1) - .copied() - .unwrap_or(source.len()) as u32; - if end_byte <= start_byte { - return; - } - let preamble_end_line = min_start - 1; - let line_count = preamble_end_line; - let checksum = chunk_checksum( - source - .as_bytes() - .get(start_byte as usize..end_byte as usize) - .unwrap_or_default(), - ); - let preamble = ChunkNode { - path: "preamble".to_string(), - identifier: None, - kind: ChunkKind::Preamble, - leaf: true, - virtual_content: None, - parent_path: Some(String::new()), - children: Vec::new(), - signature: None, - start_line: 1, - end_line: preamble_end_line, - line_count, - start_byte, - end_byte, - checksum_start_byte: start_byte, - prologue_end_byte: None, - epilogue_start_byte: None, - checksum, - error: false, - indent: 0, - indent_char: String::new(), - group: false, - }; - chunks.push(preamble); - root_children.insert(0, "preamble".to_string()); -} - -// ── Tests ──────────────────────────────────────────────────────────────── - -#[cfg(test)] -mod tests { - use std::fmt::Write as _; - - use super::{ - build_chunk_tree, line_to_chunk_path, resolve_chunk_lang, - state::ChunkState, - types::{ChunkAnchorStyle, ReadRenderParams}, - }; - use crate::{chunk::ChunkKind, language::SupportLang}; - - fn assert_supported_sample(language: &str, source: &str) { - let tree = build_chunk_tree(source, language) - .unwrap_or_else(|err| panic!("expected {language} sample to parse: {err}")); - assert!(!tree.fallback, "{language} unexpectedly fell back to blank-line chunking"); - assert_eq!(tree.parse_errors, 0, "{language} sample should parse cleanly"); - assert!( - !tree.root_children.is_empty(), - "{language} should expose at least one structural chunk" - ); - } - - #[test] - fn resolves_every_supported_canonical_language() { - for language in SupportLang::all_langs() { - assert_eq!( - resolve_chunk_lang(language.canonical_name()), - Some(*language), - "missing canonical alias for {}", - language.canonical_name() - ); - } - } - - #[test] - fn resolves_handlebars_and_tlaplus_aliases() { - assert_eq!(resolve_chunk_lang("handlebars"), Some(SupportLang::Handlebars)); - assert_eq!(resolve_chunk_lang("hbs"), Some(SupportLang::Handlebars)); - assert_eq!(resolve_chunk_lang("hsb"), Some(SupportLang::Handlebars)); - assert_eq!(resolve_chunk_lang("tla"), Some(SupportLang::Tlaplus)); - assert_eq!(resolve_chunk_lang("pluscal"), Some(SupportLang::Tlaplus)); - } - - #[test] - fn builds_structural_tree_for_each_supported_language() { - let cases = [ - ( - "astro", - "---\nconst title = \"Hello\";\n---\n

{title}

\n", - ), - ("bash", "build() { echo ok; }\n"), - ("c", "#include \nint main(void) { return 0; }\n"), - ( - "cmake", - "cmake_minimum_required(VERSION 3.28)\nproject(App)\nfunction(run_it NAME)\n message(STATUS ${NAME})\nendfunction()\n", - ), - ("cpp", "#include \nclass App {};\nint main() { return 0; }\n"), - ("csharp", "using System;\nclass App { void Run() {} }\n"), - ("clojure", "(ns demo.core)\n(defn greet [x] x)\n"), - ("css", "@import \"a.css\";\n.app { color: red; }\n"), - ("diff", "@@ -1,1 +1,1 @@\n-a\n+b\n"), - ( - "dockerfile", - "FROM alpine AS base\nARG PORT=3000\nRUN echo hi\nCMD [\"sh\", \"-c\", \"echo ok\"]\n", - ), - ("elixir", "defmodule App do\n def run(x) do\n x\n end\nend\n"), - ( - "erlang", - "-module(app).\n-export([run/1]).\nrun(X) ->\n case X of\n ok -> ok;\n _ -> error\n end.\n", - ), - ("go", "package main\nimport \"fmt\"\nfunc main() { fmt.Println(\"ok\") }\n"), - ("graphql", "type Query { hello: String }\nquery AppQuery { hello }\n"), - ("handlebars", "{{#if ready}}
{{name}}
{{/if}}\n"), - ("haskell", "module App where\nimport Data.List\nmain = putStrLn \"ok\"\n"), - ("hcl", "locals { foo = 1 }\n"), - ("html", "
ok
\n"), - ("ini", "[app]\nname=demo\nport=3000\n"), - ("java", "import java.util.*;\nclass App { void run() {} }\n"), - ("javascript", "import x from \"x\";\nexport function run() {}\n"), - ("json", "{\"name\":\"app\",\"scripts\":{\"start\":\"bun\"}}\n"), - ("just", "set shell := [\"bash\", \"-cu\"]\nrun name:\n echo {{name}}\n"), - ("julia", "module App\nfunction run(x)\n x\nend\nend\n"), - ("kotlin", "package app\nclass App { fun run() {} }\n"), - ("lua", "local function run(x) return x end\n"), - ("make", "all:\n\t@echo hi\n"), - ("markdown", "# Title\n\n## Child\n\ntext\n"), - ("nix", "{ hello = \"world\"; }\n"), - ( - "objc", - "#import \n@interface App : NSObject\n- (void)run;\n@end\n", - ), - ("ocaml", "open Printf\nlet run x = x + 1\nmodule App = struct let value = 1 end\n"), - ("odin", "package main\nmain :: proc() {}\n"), - ("perl", "package App;\nuse strict;\nsub run { return 1; }\n"), - ("php", "let count = 0;\n{#if count}

{count}

{/if}\n"), - ("swift", "import Foundation\nclass App { func run() {} }\n"), - ("toml", "[package]\nname = \"app\"\n"), - ( - "tlaplus", - "---- MODULE Spec ----\nVARIABLE x\n\n(* --algorithm Demo\nvariables x = 0;\nbegin\n Inc:\n x := x + 1;\nend algorithm; *)\n====\n", - ), - ("tsx", "export function App() { return
; }\n"), - ("typescript", "export function run(): void {}\n"), - ("verilog", "module app; endmodule\n"), - ( - "vue", - "\n\n", - ), - ("xml", "\n"), - ("yaml", "apiVersion: v1\nmetadata:\n name: app\n"), - ("zig", "const std = @import(\"std\");\npub fn main() void {}\n"), - ]; - - for (language, source) in cases { - assert_supported_sample(language, source); - } - } - - #[test] - fn tlaplus_keeps_module_and_hides_translation_generated_chunks() { - let tree = build_chunk_tree( - "---- MODULE Spec ----\nVARIABLE x\n\nInit == x = 0\n\n(* --algorithm Demo\nvariables x \ - = 0;\nbegin\n Inc:\n x := x + 1;\nend algorithm; *)\n\\* BEGIN \ - TRANSLATION\nVARIABLES pc\nNext == pc' = pc\n\\* END TRANSLATION\n====\n", - "tlaplus", - ) - .expect("tlaplus tree should build"); - - assert_eq!(tree.root_children, vec!["mod_Spe"]); - - let module = tree - .chunks - .iter() - .find(|chunk| chunk.path == "mod_Spe") - .expect("mod_Spe chunk should exist"); - assert!( - module - .children - .iter() - .any(|child| child == "mod_Spe.oper_Ini"), - "expected Init operator child, got {:?}", - module.children - ); - assert!( - module - .children - .iter() - .any(|child| child == "mod_Spe.translation_12"), - "expected synthetic translation chunk, got {:?}", - module.children - ); - assert!( - tree - .chunks - .iter() - .all(|chunk| !chunk.path.ends_with("oper_Nex")), - "translation-generated operator should be hidden: {:?}", - tree - .chunks - .iter() - .map(|chunk| chunk.path.as_str()) - .collect::>() - ); - } - - #[test] - fn json_and_hcl_chunk_names_are_structural() { - let json = build_chunk_tree("{\"scripts\":{\"start\":\"bun\"}}\n", "json") - .expect("json tree should build"); - assert!( - json.root_children.contains(&"key_scr".to_string()), - "expected key_scr, got {:?}", - json.root_children - ); - - let hcl = build_chunk_tree("locals { foo = 1 }\n", "hcl").expect("hcl tree should build"); - assert!( - hcl.root_children.contains(&"blk_loc".to_string()), - "expected blk_loc, got {:?}", - hcl.root_children - ); - } - - #[test] - fn yaml_nested_keys_produce_sub_chunks() { - // YAML keys with container values always recurse (force_recurse=true), - // so even small mappings produce sub-chunks. - let source = "database:\n host: localhost\n port: 5432\n credentials:\n username: \ - admin\n password: secret\n"; - let tree = build_chunk_tree(source, "yaml").expect("yaml tree should build"); - - assert!( - tree.root_children.contains(&"key_dat".to_string()), - "expected key_dat, got {:?}", - tree.root_children - ); - - let db = tree - .chunks - .iter() - .find(|c| c.path == "key_dat") - .expect("key_dat"); - assert!(!db.leaf, "key_dat should have children: {:?}", db.children); - assert!( - db.children.iter().any(|c| c.contains("key_hos")), - "expected key_hos child, got {:?}", - db.children - ); - assert!( - db.children.iter().any(|c| c.contains("key_cre")), - "expected key_cre child, got {:?}", - db.children - ); - - // 3-level deep: credentials should also have sub-chunks. - let creds = tree - .chunks - .iter() - .find(|c| c.path == "key_dat.key_cre") - .expect("key_cre"); - assert!(!creds.leaf, "key_cre should have children: {:?}", creds.children); - assert!( - creds.children.iter().any(|c| c.contains("key_use")), - "expected key_use child of credentials, got {:?}", - creds.children - ); - } - - #[test] - fn yaml_key_region_boundaries_separate_key_from_value() { - use super::{resolve::chunk_region_range, types::ChunkRegion}; - - let source = "server:\n host: 0.0.0.0\n port: 8080\n"; - let tree = build_chunk_tree(source, "yaml").expect("yaml tree should build"); - let server = tree - .chunks - .iter() - .find(|c| c.path == "key_ser") - .expect("key_ser"); - - // ^ should contain "server:" but not the nested keys. - let (head_s, head_e) = chunk_region_range(server, ChunkRegion::Head); - let head = &source[head_s..head_e]; - assert!(head.contains("server"), "^ should contain the key, got {head:?}"); - assert!(!head.contains("host"), "^ should not contain value content, got {head:?}"); - - // ~ should contain the nested keys but not "server:". - let (body_s, body_e) = chunk_region_range(server, ChunkRegion::Body); - let body = &source[body_s..body_e]; - assert!(body.contains("host"), "~ should contain nested keys, got {body:?}"); - assert!(!body.contains("server"), "~ should not contain the key header, got {body:?}"); - } - - #[test] - fn yaml_leading_comment_does_not_prevent_sub_chunks() { - let source = "# Global settings\napp:\n name: my-app\n debug: true\n features:\n - \ - auth\n - logging\n"; - let tree = build_chunk_tree(source, "yaml").expect("yaml tree should build"); - - // The leading comment becomes a preamble; key_app is a separate chunk. - let app = tree - .chunks - .iter() - .find(|c| c.path == "key_app") - .expect("key_app"); - - // Sub-keys should be individually addressable. - assert!(!app.leaf, "key_app should have children: {:?}", app.children); - assert!( - app.children.iter().any(|c| c.contains("key_nam")), - "expected key_nam child, got {:?}", - app.children - ); - assert!( - app.children.iter().any(|c| c.contains("key_deb")), - "expected key_deb child, got {:?}", - app.children - ); - assert!( - app.children.iter().any(|c| c.contains("key_fea")), - "expected key_fea child, got {:?}", - app.children - ); - } - - #[test] - fn handlebars_chunks_blocks_and_tags() { - let tree = - build_chunk_tree("{{#if ready}}
{{name}}
{{/if}}\n", "handlebars") - .expect("handlebars tree should build"); - assert!( - tree.root_children.contains(&"blk_if".to_string()), - "expected blk_if, got {:?}", - tree.root_children - ); - let block = tree - .chunks - .iter() - .find(|chunk| chunk.path == "blk_if") - .expect("blk_if chunk should exist"); - assert!(!block.leaf); - assert!( - block.children.iter().any(|child| child == "blk_if.tag_div"), - "expected nested div tag, got {:?}", - block.children - ); - } - - #[test] - fn builds_typescript_chunk_tree() { - let source = format!( - r#"import a from "a"; -import b from "b"; - -class Bla extends Base {{ - value = 1; - - constructor(config: Config) {{ - this.value = config.value; - }} - - async onEvent(ev: Event, ctx?: Context): Promise {{ - if (!ev) return; -{body} - }} -}} - -function main(): void {{ - console.log("ok"); -}} -"#, - body = (0..60) - .map(|index| format!("\t\tthis.value += {index};")) - .collect::>() - .join("\n"), - ); - - let tree = build_chunk_tree(source.as_str(), "typescript").expect("tree should build"); - let child_names = tree - .root_children - .iter() - .map(std::string::String::as_str) - .collect::>(); - assert_eq!(child_names, vec!["imp", "cls_Bla", "fn_mai"]); - - let class_chunk = tree - .chunks - .iter() - .find(|chunk| chunk.path == "cls_Bla") - .expect("class chunk should exist"); - assert!(!class_chunk.leaf); - assert!( - class_chunk - .children - .iter() - .any(|child| child == "cls_Bla.ctor") - ); - assert!( - class_chunk - .children - .iter() - .any(|child| child == "cls_Bla.fn_onE") - ); - - let line_path = line_to_chunk_path(&tree, 15).expect("line should resolve"); - assert!(line_path.starts_with("cls_Bla.fn_onE")); - } - - #[test] - fn call_with_trailing_callback_promotes_to_named_expression() { - // Test that `describe(...)` / `it(...)` patterns with trailing callback - // arguments are promoted to named expression chunks with children, - // rather than being flat groupable stmts leaves. - let source = "import { describe, it } from \"bun:test\";\n\ndescribe(\"suite\", () => \ - {\n\tit(\"does a\", () => {\n\t\texpect(1).toBe(1);\n\t});\n\n\tit(\"does \ - b\", () => {\n\t\texpect(2).toBe(2);\n\t});\n});\n"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - - // describe(...) should be promoted to a named expr chunk, not grouped into - // stmts. - let describe_chunk = tree - .chunks - .iter() - .find(|c| c.path == "ex_des") - .expect("describe should be a named chunk"); - assert!(!describe_chunk.leaf, "describe chunk should have children (not a leaf)"); - assert!(!describe_chunk.group, "describe chunk should not be groupable"); - - // The nested calls inside should stay addressable under the promoted parent. - let it_chunks = tree - .chunks - .iter() - .filter(|c| c.path.starts_with("ex_des.ex")) - .count(); - assert_eq!( - it_chunks, - 2, - "nested calls under describe() should stay addressable; chunks: {:?}", - tree.chunks.iter().map(|c| &c.path).collect::>() - ); - } - - #[test] - fn call_with_trailing_callback_works_for_member_expressions() { - // Test member expression calls like `describe.serial(...)` or `app.use(...)`. - let source = "describe.serial(\"ordered\", () => {\n\tit(\"first\", () => \ - {\n\t\texpect(true).toBe(true);\n\t});\n});\n"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - - let describe_chunk = tree - .chunks - .iter() - .find(|c| c.path == "ex_des") - .expect("describe.serial should be a named chunk"); - assert!(!describe_chunk.group, "describe.serial chunk should not be groupable"); - } - - #[test] - fn call_without_callback_stays_grouped() { - // Plain call expressions without trailing callbacks should remain as - // groupable stmts, not promoted. - let source = "console.log(\"a\");\nconsole.log(\"b\");\n"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - - let stmts_chunk = tree - .chunks - .iter() - .find(|c| c.path == "st") - .expect("plain calls should be grouped into st"); - assert!(stmts_chunk.group, "stmts should be a group"); - assert!(stmts_chunk.leaf, "stmts with no callback should be a leaf"); - } - - #[test] - fn nested_call_with_callback_has_body_region() { - // Nested test()/it() calls inside describe() should be promoted with - // prologue/epilogue set so that `~` targets the callback body, not the - // entire chunk. - let source = "\ -describe(\"suite\", () => { -\ttest(\"my test\", () => { -\t\tconst x = 1; -\t\tconst y = 2; -\t\tconst z = 3; -\t\texpect(x + y).toBe(z); -\t}); - -\ttest(\"other test\", () => { -\t\tconst a = 10; -\t\tconst b = 20; -\t\tconst c = 30; -\t\texpect(a + b).toBe(c); -\t}); -}); -"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - - let test_chunk = tree - .chunks - .iter() - .find(|c| c.path.starts_with("ex_des.ex_tes")) - .expect("test() should be a promoted named chunk under describe"); - assert!( - test_chunk.prologue_end_byte.is_some(), - "test() chunk should have prologue_end_byte for ~ region support" - ); - assert!( - test_chunk.epilogue_start_byte.is_some(), - "test() chunk should have epilogue_start_byte for ~ region support" - ); - } - - #[test] - fn jsx_return_with_map_callback_exposes_nested_chunks() { - // A React component body that returns `items.map(item => - // ...children...)` used to collapse into a single opaque return - // chunk, forcing any edit to replace the full return body. The pipeline must - // now surface: - // * the `.map()` call as a promoted `expr_*` container - // * the inner `return (...)` as a `ret` container - // * every direct JSX child of the returned element as its own `tag_*` chunk - let source = r#" - const runsBody = () => { - return items.map(run => { - return ( - -
- {run.uid} -
-
- {run.name} -
-
- {run.state} -
- - ); - }); - }; - "#; - let tree = build_chunk_tree(source, "tsx").expect("tsx tree should build"); - let paths: Vec<&str> = tree.chunks.iter().map(|c| c.path.as_str()).collect(); - - let ret = tree - .chunks - .iter() - .find(|c| c.path == "fn_run.ex_ite.ret") - .unwrap_or_else(|| panic!("expected fn_run.ex_ite.ret; chunks: {paths:?}")); - assert!(!ret.leaf, "JSX return chunk must expose children, got leaf"); - - let div_children: Vec<&str> = tree - .chunks - .iter() - .filter(|c| c.path.starts_with("fn_run.ex_ite.ret.tag_div")) - .map(|c| c.path.as_str()) - .collect(); - assert_eq!( - div_children.len(), - 3, - "expected 3 top-level div chunks inside the returned , got {div_children:?}" - ); - - // The JSX opening/closing elements must not leak into the chunk tree. - assert!( - !paths - .iter() - .any(|p| p.contains("ch_jsx") || p.contains("ch_jsx")), - "jsx opening/closing elements must be filtered out; chunks: {paths:?}" - ); - } - - #[test] - fn jsx_return_does_not_absorb_opening_tag_into_first_child() { - // When the outer `` has a multi-line opening element, the first - // `
` child must still report its own real start line and not be - // extended backward to swallow the opening element. - let source = r#" - const row = (run: Run) => { - return ( - -
- {run.uid} -
-
- {run.name} -
- - ); - }; - "#; - let tree = build_chunk_tree(source, "tsx").expect("tsx tree should build"); - let first_div = tree - .chunks - .iter() - .find(|c| c.path == "fn_row.ret.tag_div_1") - .expect("first tag_div_1 should exist"); - // The `
` starts well below the `= 9, - "first div chunk must start at its own opening line (>=9), got {}", - first_div.start_line - ); - } - - #[test] - fn jsx_return_directly_returning_element_exposes_children() { - // `return ...` without parens should still recurse into the JSX. - let source = r#" - const header = () => { - return
-
title
-
subtitle
-
body
-
footer
-
; - }; - "#; - let tree = build_chunk_tree(source, "tsx").expect("tsx tree should build"); - let ret = tree - .chunks - .iter() - .find(|c| c.path == "fn_hea.ret") - .expect("ret chunk should exist"); - assert!(!ret.leaf, "bare JSX return should expose its children"); - } - - #[test] - fn small_jsx_return_stays_collapsed() { - // Short JSX returns (well under the leaf threshold) should NOT explode - // into per-element chunks — otherwise small React components get drowned - // in noise. - let source = r" - const Loading = () => { - return
Loading…
; - }; - "; - let tree = build_chunk_tree(source, "tsx").expect("tsx tree should build"); - let has_tag_children = tree - .chunks - .iter() - .any(|c| c.path.starts_with("fn_Loading.") && c.path.contains("tag_")); - assert!( - !has_tag_children, - "tiny JSX components should not explode; chunks: {:?}", - tree.chunks.iter().map(|c| &c.path).collect::>() - ); - } - - #[test] - fn surfaces_error_chunks() { - let source = r"class Broken { - method() { - if ( - } - - ok(): void { - return; - } -} -"; - - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - assert!(tree.parse_errors > 0); - assert!( - tree - .chunks - .iter() - .any(|chunk| chunk.kind == ChunkKind::Error && chunk.identifier.is_none()), - "expected error chunk, got {:?}", - tree - .chunks - .iter() - .map(|chunk| (&chunk.path, chunk.kind, chunk.identifier.as_deref())) - .collect::>() - ); - } - - #[test] - fn falls_back_to_blank_line_blocks() { - let source = "A=1\nB=2\n\nC=3\n"; - let tree = build_chunk_tree(source, "env").expect("fallback tree should build"); - assert!(tree.fallback); - assert_eq!(tree.root_children, vec!["A", "C"]); - } - - #[test] - fn always_recurses_small_class() { - let source = r"class Tiny { - foo() { return 1; } - bar() { return 2; } -}"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - let class_chunk = tree - .chunks - .iter() - .find(|c| c.path == "cls_Tin") - .expect("cls_Tin"); - assert!(!class_chunk.leaf); - assert!(class_chunk.children.iter().any(|c| c == "cls_Tin.fn_foo")); - assert!(class_chunk.children.iter().any(|c| c == "cls_Tin.fn_bar")); - } - - #[test] - fn empty_class_is_a_branch() { - let source = r"class Empty {}"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - let class_chunk = tree - .chunks - .iter() - .find(|c| c.path == "cls_Emp") - .expect("cls_Emp"); - assert!(!class_chunk.leaf); - } - - #[test] - fn promotes_arrow_function_to_fn_chunk() { - let source = r"const handler = (ev) => { - console.log(ev); - return ev; -};"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - assert!(tree.chunks.iter().any(|c| c.path == "fn_han"), "expected fn_han chunk"); - assert!( - !tree.root_children.contains(&"decls".to_string()), - "arrow fn should not be grouped as decls" - ); - } - - #[test] - fn promotes_const_class_expression() { - let source = r"const Foo = class { - method() { return 42; } -};"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - assert!(tree.chunks.iter().any(|c| c.path == "cls_Foo"), "expected cls_Foo chunk"); - assert!( - !tree.root_children.contains(&"decls".to_string()), - "class expr should not be grouped as decls" - ); - } - - #[test] - fn promotes_exported_arrow_function_and_preserves_wrapper_range() { - let source = r#"export const handler = () => { - console.log("handled"); -};"#; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - let chunk = tree - .chunks - .iter() - .find(|c| c.path == "fn_han") - .expect("fn_han"); - assert!(chunk.leaf); - assert_eq!(chunk.start_line, 1); - assert_eq!(chunk.end_line, 3); - assert!( - !tree.root_children.contains(&"decls".to_string()), - "exported arrow fn should not fall back to decls" - ); - } - - #[test] - fn promotes_exported_const_class_expression() { - let source = r"export const Foo = class { - method() { return 42; } -};"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - let chunk = tree - .chunks - .iter() - .find(|c| c.path == "cls_Foo") - .expect("cls_Foo"); - assert!(!chunk.leaf); - assert_eq!(chunk.start_line, 1); - assert_eq!(chunk.end_line, 3); - } - - #[test] - fn promotes_export_default_class_to_default_export_chunk() { - let source = r"export default class Foo { - method() { return 42; } -}"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - let chunk = tree.chunks.iter().find(|c| c.path == "dex").expect("dex"); - assert_eq!(chunk.start_line, 1); - assert_eq!(chunk.end_line, 3); - assert!( - !tree.root_children.contains(&"cls_Foo".to_string()), - "default export should be remapped to defexp" - ); - } - - #[test] - fn small_interfaces_keep_children() { - let source = r"interface Config { - name: string; - getValue(): number; -}"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - let iface = tree - .chunks - .iter() - .find(|c| c.path == "intf_Con") - .expect("intf_Con"); - assert!(!iface.children.is_empty(), "interface members should be addressable as children"); - } - - #[test] - fn unicode_identifiers_preserved() { - let source = r"class 服务器 { - 启动() { return true; } -}"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - assert!(tree.chunks.iter().any(|c| c.path == "cls_服务器"), "expected cls_服务器 chunk"); - } - - #[test] - fn python_chunk_tree() { - let source = r"import os -import sys - -class Server: - def __init__(self): - self.running = False - - def start(self): - self.running = True - -def main(): - s = Server() - s.start() -" - .to_string(); - let tree = build_chunk_tree(source.as_str(), "python").expect("tree should build"); - let names: Vec<&str> = tree.root_children.iter().map(String::as_str).collect(); - assert!(names.contains(&"imp"), "expected imports, got {names:?}"); - assert!(names.contains(&"cls_Ser"), "expected cls_Ser, got {names:?}"); - assert!(names.contains(&"fn_mai"), "expected fn_mai, got {names:?}"); - let cls = tree - .chunks - .iter() - .find(|c| c.path == "cls_Ser") - .expect("cls_Ser"); - assert!(!cls.leaf); - assert!( - cls.children.iter().any(|c| c == "cls_Ser.fn_ini"), - "expected fn_ini (__init__ sanitized)" - ); - assert!(cls.children.iter().any(|c| c == "cls_Ser.fn_sta"), "expected fn_sta"); - assert_eq!(cls.signature.as_deref(), Some("class Server")); - } - - #[test] - fn python_loops_are_named_loop() { - let mut body = String::new(); - body.push_str(" total = 0\n"); - body.push_str(" for item in range(3):\n"); - body.push_str(" total += item\n"); - for index in 0..55 { - let _ = writeln!(body, " filler_{index} = {index}"); - } - body.push_str(" return total\n"); - let source = format!("def worker():\n{body}"); - let tree = build_chunk_tree(source.as_str(), "python").expect("tree should build"); - let worker = tree - .chunks - .iter() - .find(|c| c.path == "fn_wor") - .expect("fn_wor"); - assert!(!worker.leaf); - assert!(tree.chunks.iter().any(|c| c.path == "fn_wor.loop"), "expected loop chunk"); - } - - #[test] - fn python_class_signature_strips_colon() { - let source = r"class Foo(Base): - pass -"; - let tree = build_chunk_tree(source, "python").expect("tree should build"); - let class_chunk = tree - .chunks - .iter() - .find(|c| c.path == "cls_Foo") - .expect("cls_Foo"); - assert_eq!(class_chunk.signature.as_deref(), Some("class Foo(Base)")); - assert!(class_chunk.leaf); - } - - #[test] - fn rust_chunk_tree() { - let source = r#"use std::io; - -struct Config { - name: String, -} - -impl Config { - fn new(name: String) -> Self { - Config { name } - } - - fn name(&self) -> &str { - &self.name - } -} - -fn main() { - let c = Config::new("test".into()); - println!("{}", c.name()); -}"#; - let tree = build_chunk_tree(source, "rust").expect("tree should build"); - let names: Vec<&str> = tree.root_children.iter().map(String::as_str).collect(); - assert!(names.contains(&"imp"), "expected imports, got {names:?}"); - assert!(names.contains(&"stc_Con"), "expected stc_Con, got {names:?}"); - assert!(names.contains(&"ipl_Con"), "expected ipl_Con, got {names:?}"); - assert!(names.contains(&"fn_mai"), "expected fn_mai, got {names:?}"); - let impl_chunk = tree - .chunks - .iter() - .find(|c| c.path == "ipl_Con") - .expect("ipl_Con"); - assert!(!impl_chunk.leaf); - assert!(impl_chunk.children.iter().any(|c| c == "ipl_Con.fn_new"), "expected fn_new"); - assert!(impl_chunk.children.iter().any(|c| c == "ipl_Con.fn_nam"), "expected fn_nam"); - } - - #[test] - fn rust_trait_impl_naming() { - let source = r#"use std::fmt; - -struct Config { - name: String, -} - -impl fmt::Display for Config { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(f, "{}", self.name) - } -} - -impl Config { - fn new(name: String) -> Self { - Config { name } - } -}"#; - let tree = build_chunk_tree(source, "rust").expect("tree should build"); - let names: Vec<&str> = tree.root_children.iter().map(String::as_str).collect(); - assert!(names.contains(&"ipl_Dis"), "expected ipl_Dis, got {names:?}"); - assert!(names.contains(&"ipl_Con"), "expected ipl_Con, got {names:?}"); - } - - #[test] - fn rust_field_naming() { - let fields: Vec = (0..32).map(|i| format!(" field_{i}: u32,")).collect(); - let source = format!("struct Server {{\n{}\n}}\n", fields.join("\n")); - let tree = build_chunk_tree(&source, "rust").expect("tree should build"); - let server = tree - .chunks - .iter() - .find(|c| c.path == "stc_Ser") - .expect("stc_Ser should exist"); - assert!(!server.leaf, "large struct should be a branch"); - assert!( - server.children.iter().any(|c| c == "stc_Ser.fld_fie_1"), - "expected fld_fie_1 in children: {:?}", - server.children - ); - } - - #[test] - fn go_chunk_tree() { - let source = r#"package main - - import "fmt" - - type Config struct { - Name string - } - - type Reader interface { - Read(p []byte) (int, error) - } - - func main() { - fmt.Println("hello") - }"#; - let tree = build_chunk_tree(source, "go").expect("tree should build"); - let names: Vec<&str> = tree.root_children.iter().map(String::as_str).collect(); - assert!(names.contains(&"mod_mai"), "expected package module, got {names:?}"); - assert!(names.contains(&"imp"), "expected imports, got {names:?}"); - assert!(names.contains(&"ty_Con"), "expected ty_Con, got {names:?}"); - assert!(names.contains(&"ty_Rea"), "expected ty_Rea, got {names:?}"); - assert!(names.contains(&"fn_mai"), "expected fn_mai, got {names:?}"); - let config = tree - .chunks - .iter() - .find(|c| c.path == "ty_Con") - .expect("ty_Con"); - assert!(!config.leaf); - assert!( - config - .children - .iter() - .any(|child| child == "ty_Con.fld_Nam"), - "expected ty_Con.fld_Nam, got {:?}", - config.children - ); - let reader = tree - .chunks - .iter() - .find(|c| c.path == "ty_Rea") - .expect("ty_Rea"); - assert!(reader.leaf); - assert!(reader.children.is_empty(), "single-line interfaces should render inline"); - } - - #[test] - fn go_method_summary_omits_receiver_duplication() { - let source = r"package main - -type MemorySink struct{} - -func (s *MemorySink) Write(p []byte) (int, error) { - return len(p), nil -} -"; - let tree = build_chunk_tree(source, "go").expect("tree should build"); - let method = tree - .chunks - .iter() - .find(|chunk| chunk.path == "fn_Wri") - .expect("Write method"); - - assert_eq!(method.signature.as_deref(), Some("fn Write(p []byte) (int, error)")); - } - - #[test] - fn nix_chunk_tree_exposes_attr_bindings() { - let source = r#"{ - hello = "world"; - nested = { - value = 1; - }; - } - "#; - let tree = build_chunk_tree(source, "nix").expect("tree should build"); - let attrset = tree - .chunks - .iter() - .find(|chunk| chunk.path == "ats") - .expect("ats chunk"); - assert!(!tree.fallback, "nix should use tree-sitter chunking"); - assert!(!attrset.leaf, "top-level attrset should recurse into bindings"); - assert!( - attrset.children.iter().any(|child| child == "ats.attr_hel"), - "expected attr_hel child, got {:?}", - attrset.children - ); - assert!( - attrset.children.iter().any(|child| child == "ats.attr_nes"), - "expected attr_nes child, got {:?}", - attrset.children - ); - } - - #[test] - fn preamble_chunk_covers_leading_lines_before_first_item() { - let source = "// header\n// second\n\nfn main() {}\n"; - let tree = build_chunk_tree(source, "rust").expect("tree should build"); - assert!( - tree.root_children.iter().any(|c| c == "preamble"), - "expected preamble in {:?}", - tree.root_children - ); - let preamble = tree - .chunks - .iter() - .find(|c| c.path == "preamble") - .expect("preamble"); - assert_eq!(preamble.start_line, 1); - assert_eq!(preamble.end_line, 3); - let main_fn = tree - .chunks - .iter() - .find(|c| c.path == "fn_mai") - .expect("fn_mai"); - assert!( - main_fn.start_line > preamble.end_line, - "first structural chunk should start after preamble" - ); - } - - #[test] - fn indent_fields_populated() { - let source = "class Foo {\n\tbar() {\n\t\treturn 1;\n\t}\n}"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - let method = tree - .chunks - .iter() - .find(|c| c.path == "cls_Foo.fn_bar") - .expect("fn_bar"); - assert_eq!(method.indent, 1, "method should have indent=1"); - assert_eq!(method.indent_char, "\t", "method should use tab indentation"); - } - - #[test] - fn keeps_trivial_rust_enum_variants_addressable() { - let source = r"pub enum LogLevel { - Debug, - Info, - Warn, - Error, - }"; - let tree = build_chunk_tree(source, "rust").expect("tree should build"); - let enum_chunk = tree - .chunks - .iter() - .find(|c| c.path == "en_Log") - .expect("en_Log"); - assert!(!enum_chunk.leaf); - assert!( - enum_chunk - .children - .iter() - .any(|child| child == "en_Log.vr_Deb") - ); - assert!( - enum_chunk - .children - .iter() - .any(|child| child == "en_Log.vr_Err") - ); - } - - #[test] - fn rust_trait_members_stay_addressable() { - let source = r"trait Handler { - fn handle(&self, method: &str, path: &str) -> OpResult; - }"; - let tree = build_chunk_tree(source, "rust").expect("tree should build"); - let trait_chunk = tree - .chunks - .iter() - .find(|c| c.path == "tr_Han") - .expect("tr_Han"); - assert!(!trait_chunk.children.is_empty(), "trait members should be addressable as children"); - } - - #[test] - fn collapses_trivial_go_interface_children() { - let source = r"package main - - type Handler interface { - Handle(method, path string) Result - }"; - let tree = build_chunk_tree(source, "go").expect("tree should build"); - let iface = tree - .chunks - .iter() - .find(|c| c.path == "ty_Han") - .expect("ty_Han"); - assert!(iface.leaf); - assert!(iface.children.is_empty(), "single-line interface methods should render inline"); - } - - #[test] - fn typescript_interfaces_use_interface_prefix() { - let source = r"interface Settings { - enabled: boolean; -} -"; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - assert!( - tree.chunks.iter().any(|chunk| chunk.path == "intf_Set"), - "expected intf_Set in {:?}", - tree - .chunks - .iter() - .map(|chunk| chunk.path.as_str()) - .collect::>() - ); - assert!( - !tree.chunks.iter().any(|chunk| chunk.path == "ifc_Set"), - "legacy ifc_ prefix should not remain addressable" - ); - } - - #[test] - fn read_resolves_partial_selectors_and_bare_checksums() { - let filler = (0..60) - .map(|index| format!(" const value{index} = {index};")) - .collect::>() - .join("\n"); - let source = format!( - "function handleTerraform() {{\n{filler}\n try {{\n if (ready) {{\n \ - work();\n }}\n }} catch (error) {{\n throw error;\n }}\n}}\n" - ); - let state = ChunkState::parse(source, "typescript".to_string()).expect("state should parse"); - let chunk = state - .chunks() - .into_iter() - .find(|candidate| candidate.path == "fn_han.try") - .expect("try chunk path should exist"); - let selectors = vec![ - format!("sample.ts:{}", "fn_han.try"), - format!("sample.ts:{}", "handle.try"), - format!("sample.ts:{}", "try"), - format!("sample.ts:try#{}", chunk.checksum), - format!("sample.ts:#{}", chunk.checksum), - format!("sample.ts:{}", chunk.checksum), - ]; - for selector in selectors { - let result = state - .render_read(ReadRenderParams { - read_path: selector.clone(), - display_path: "sample.ts".to_string(), - language_tag: Some("ts".to_string()), - omit_checksum: false, - anchor_style: Some(ChunkAnchorStyle::Full), - absolute_line_range: None, - tab_replacement: Some(" ".to_string()), - normalize_indent: Some(true), - }) - .unwrap_or_else(|err| panic!("selector {selector} should resolve: {err}")); - let resolved = result - .chunk - .expect("selector read should resolve a chunk target"); - assert_eq!(resolved.selector, format!("fn_han.try#{}", chunk.checksum)); - } - } - - #[test] - fn read_lists_chunks_for_question_selector() { - let source = "function run() {\n return 1;\n}\n"; - let state = ChunkState::parse(source.to_string(), "typescript".to_string()) - .expect("state should parse"); - let result = state - .render_read(ReadRenderParams { - read_path: "sample.ts:?".to_string(), - display_path: "sample.ts".to_string(), - language_tag: Some("ts".to_string()), - omit_checksum: false, - anchor_style: Some(ChunkAnchorStyle::Full), - absolute_line_range: None, - tab_replacement: Some(" ".to_string()), - normalize_indent: Some(true), - }) - .expect("listing should succeed"); - assert!(result.text.contains("sample.ts chunks"), "{}", result.text); - assert!(result.text.contains("fn_run#")); - // Region listing removed — all chunks accept all regions now. - assert!(!result.text.contains("return 1")); - } - - #[test] - fn read_renders_full_chunk_paths_in_full_anchor_style() { - let source = "class Worker { - run(): void { - work(); - } -} -"; - let state = ChunkState::parse(source.to_string(), "typescript".to_string()) - .expect("state should parse"); - let result = state - .render_read(ReadRenderParams { - read_path: "sample.ts".to_string(), - display_path: "sample.ts".to_string(), - language_tag: Some("ts".to_string()), - omit_checksum: false, - anchor_style: Some(ChunkAnchorStyle::Full), - absolute_line_range: None, - tab_replacement: Some(" ".to_string()), - normalize_indent: Some(true), - }) - .expect("root read should succeed"); - assert!(result.text.contains("cls_Wor.fn_run#"), "{}", result.text); - } - - #[test] - fn read_missing_chunk_returns_error_with_suggestions() { - let filler = (0..60) - .map(|index| format!(" const value{index} = {index};")) - .collect::>() - .join("\n"); - let source = format!( - "function loadSkills() {{\n{filler}\n try {{\n work();\n }} catch (error) \ - {{\n throw error;\n }}\n}}\n\nfunction handleTerraform() {{\n{filler}\n \ - try {{\n work();\n }} catch (error) {{\n throw error;\n }}\n}}\n" - ); - let state = ChunkState::parse(source, "typescript".to_string()).expect("state should parse"); - let result = state - .render_read(ReadRenderParams { - read_path: "sample.ts:fn_loa.try_2".to_string(), - display_path: "sample.ts".to_string(), - language_tag: Some("ts".to_string()), - omit_checksum: false, - anchor_style: Some(ChunkAnchorStyle::Full), - absolute_line_range: None, - tab_replacement: Some(" ".to_string()), - normalize_indent: Some(true), - }) - .expect("rnd_rea should succeed"); - - let chunk = result.chunk.expect("should have a chunk target"); - assert_eq!(chunk.status, super::types::ChunkReadStatus::NotFound); - - let text = &result.text; - assert!(text.contains("Chunk path not found: \"fn_loa.try_2\""), "{text}"); - assert!(text.contains("Direct children of \"fn_loa\""), "{text}"); - assert!(text.contains("fn_loa.try"), "{text}"); - } - - #[test] - fn read_reports_unsupported_region_distinctly() { - let source = "function run() {\n return 1;\n}\n"; - let state = ChunkState::parse(source.to_string(), "typescript".to_string()) - .expect("state should parse"); - let result = state - .render_read(ReadRenderParams { - read_path: "sample.ts:fn_run@unknown".to_string(), - display_path: "sample.ts".to_string(), - language_tag: Some("ts".to_string()), - omit_checksum: false, - anchor_style: Some(ChunkAnchorStyle::Full), - absolute_line_range: None, - tab_replacement: Some(" ".to_string()), - normalize_indent: Some(true), - }) - .expect("rnd_rea should succeed"); - - let read_target = result.chunk.expect("should include read target"); - assert_eq!(read_target.status, super::types::ChunkReadStatus::UnsupportedRegion); - assert_eq!(read_target.selector, "sample.ts:fn_run@unknown"); - assert!(result.text.contains("Unknown chunk region"), "{}", result.text); - } - - #[test] - fn read_body_region_returns_only_body_content() { - let source = "/// A doc.\nfunction run() {\n return 1;\n}\n"; - let state = ChunkState::parse(source.to_string(), "typescript".to_string()) - .expect("state should parse"); - let result = state - .render_read(ReadRenderParams { - read_path: "sample.ts:fn_run~".to_string(), - display_path: "sample.ts".to_string(), - language_tag: Some("ts".to_string()), - omit_checksum: false, - anchor_style: Some(ChunkAnchorStyle::Full), - absolute_line_range: None, - tab_replacement: Some(" ".to_string()), - normalize_indent: Some(true), - }) - .expect("rnd_rea should succeed"); - - // Should contain only the body, not the signature or doc comment. - assert!( - !result.text.contains("/// A doc"), - "body read should not contain the doc comment: {}", - result.text - ); - assert!( - !result.text.contains("function run"), - "body read should not contain the signature: {}", - result.text - ); - assert!( - result.text.contains("return 1"), - "body read should contain the body content: {}", - result.text - ); - } - - #[test] - fn python_prologue_read_has_consistent_indentation() { - let source = - "class Server:\n @property\n def address(self) -> str:\n return self._addr\n"; - let state = - ChunkState::parse(source.to_string(), "python".to_string()).expect("state should parse"); - let result = state - .render_read(ReadRenderParams { - read_path: "test.py:cls_Ser.fn_add^".to_string(), - display_path: "test.py".to_string(), - language_tag: Some("py".to_string()), - omit_checksum: false, - anchor_style: Some(ChunkAnchorStyle::Full), - absolute_line_range: None, - tab_replacement: Some(" ".to_string()), - normalize_indent: Some(true), - }) - .expect("rnd_rea should succeed"); - - // Both lines of the prologue should have the same indent depth. - // Skip the first line (selector_ref header). - let content_lines: Vec<&str> = result - .text - .split('\n') - .filter(|l| !l.trim().is_empty()) - .skip(1) - .collect(); - assert!( - content_lines.len() >= 2, - "prologue should have at least 2 lines (decorator + def): {content_lines:?}" - ); - let decorator_tabs = content_lines[0].chars().take_while(|c| *c == '\t').count(); - let def_tabs = content_lines[1].chars().take_while(|c| *c == '\t').count(); - assert_eq!( - decorator_tabs, def_tabs, - "decorator and def should have same indent: decorator={decorator_tabs} tabs, \ - def={def_tabs} tabs in {content_lines:?}" - ); - } - - #[test] - fn go_struct_checksum_ignores_method_body_changes() { - let before = r"package main - -type Server struct { - Addr string -} - -func (s *Server) Start() string { - return s.Addr -} -"; - let after = r#"package main - -type Server struct { - Addr string -} - -func (s *Server) Start() string { - return s.Addr + ":80" -} -"#; - let before_tree = build_chunk_tree(before, "go").expect("before tree should build"); - let after_tree = build_chunk_tree(after, "go").expect("after tree should build"); - let before_struct = before_tree - .chunks - .iter() - .find(|chunk| chunk.path == "ty_Ser") - .expect("before struct chunk"); - let after_struct = after_tree - .chunks - .iter() - .find(|chunk| chunk.path == "ty_Ser") - .expect("after struct chunk"); - let before_method = before_tree - .chunks - .iter() - .find(|chunk| chunk.path == "fn_Sta") - .expect("before method chunk"); - let after_method = after_tree - .chunks - .iter() - .find(|chunk| chunk.path == "fn_Sta") - .expect("after method chunk"); - assert_eq!(before_struct.checksum, after_struct.checksum); - assert_ne!(before_method.checksum, after_method.checksum); - } - - #[test] - fn keeps_trivial_typescript_enum_variants_addressable() { - let source = r#"enum Status { - Idle = "idle", - Busy = "busy", - }"#; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - let enum_chunk = tree - .chunks - .iter() - .find(|c| c.path == "en_Sta") - .expect("en_Sta"); - assert!(!enum_chunk.leaf); - assert!( - enum_chunk - .children - .iter() - .any(|child| child == "en_Sta.vr_Idl") - ); - assert!( - enum_chunk - .children - .iter() - .any(|child| child == "en_Sta.vr_Bus") - ); - } - - #[test] - fn rust_attribute_absorbed_into_struct_chunk() { - let source = r"#[derive(Debug, Clone)] -struct Record { - name: String, -} -"; - let tree = build_chunk_tree(source, "rust").expect("tree should build"); - let struct_chunk = tree - .chunks - .iter() - .find(|c| c.path == "stc_Rec") - .expect("stc_Rec"); - assert_eq!(struct_chunk.start_line, 1, "struct chunk should start at attribute line"); - } - - #[test] - fn rust_multi_attribute_absorbed_into_struct_chunk() { - let source = r#"#[derive(Debug)] -#[serde(rename_all = "camelCase")] -struct Config { - name: String, -} -"#; - let tree = build_chunk_tree(source, "rust").expect("tree should build"); - let struct_chunk = tree - .chunks - .iter() - .find(|c| c.path == "stc_Con") - .expect("stc_Con"); - assert_eq!(struct_chunk.start_line, 1, "struct chunk should start at first attribute line"); - } - - #[test] - fn rust_struct_checksum_ignores_leading_attributes_absorbed_into_display_range() { - let one_attr = r"#[derive(Debug)] -struct Config { - name: String, -} -"; - let two_attrs = r"#[derive(Debug, Clone)] -struct Config { - name: String, -} -"; - let ta = build_chunk_tree(one_attr, "rust").expect("tree"); - let tb = build_chunk_tree(two_attrs, "rust").expect("tree"); - let ca = ta - .chunks - .iter() - .find(|c| c.path == "stc_Con") - .expect("stc_Con"); - let cb = tb - .chunks - .iter() - .find(|c| c.path == "stc_Con") - .expect("stc_Con"); - assert_eq!( - ca.checksum, cb.checksum, - "checksum hashes from the struct item, not absorbed outer attributes" - ); - } - - #[test] - fn rust_enum_variant_naming() { - let source = r"enum Message { - Ok, - Error { - code: u32, - message: String, - }, -} -"; - let tree = build_chunk_tree(source, "rust").expect("tree should build"); - let enum_chunk = tree - .chunks - .iter() - .find(|c| c.path == "en_Mes") - .expect("en_Mes"); - assert!(!enum_chunk.children.is_empty(), "non-trivial enum should have children"); - assert!( - tree.chunks.iter().any(|c| c.path == "en_Mes.vr_Ok"), - "expected vr_Ok, got children: {:?}", - enum_chunk.children - ); - assert!( - tree.chunks.iter().any(|c| c.path == "en_Mes.vr_Err"), - "expected vr_Err, got children: {:?}", - enum_chunk.children - ); - } - - #[test] - fn ruby_class_methods_chunked() { - let source = r#"module PaymentProcessing - class Money - include Comparable - - attr_reader :amount, :currency - - def initialize(amount, currency = :usd) - @amount = amount - @currency = currency - end - - def self.zero(currency = :usd) - new(0, currency) - end - - def to_s - "$#{amount}" - end - - private - - def validate! - raise "Invalid" if amount < 0 - end - end -end -"#; - let tree = build_chunk_tree(source, "ruby").expect("tree should build"); - assert_eq!(tree.root_children, vec!["mod_Pay"]); - let module = tree - .chunks - .iter() - .find(|c| c.path == "mod_Pay") - .expect("mod_Pay"); - assert!(!module.leaf); - assert!( - module.children.iter().any(|c| c == "mod_Pay.cls_Mon"), - "expected cls_Mon inside module, got {:?}", - module.children - ); - let class = tree - .chunks - .iter() - .find(|c| c.path == "mod_Pay.cls_Mon") - .expect("cls_Mon"); - assert!(!class.leaf); - assert!( - class.children.iter().any(|c| c == "mod_Pay.cls_Mon.ctor"), - "expected constructor in class children: {:?}", - class.children - ); - assert!( - class.children.iter().any(|c| c == "mod_Pay.cls_Mon.fn_zer"), - "expected fn_zer in class children: {:?}", - class.children - ); - assert!( - class.children.iter().any(|c| c == "mod_Pay.cls_Mon.fn_to"), - "expected fn_to in class children: {:?}", - class.children - ); - assert!( - class.children.iter().any(|c| c == "mod_Pay.cls_Mon.fn_val"), - "expected fn_val in class children: {:?}", - class.children - ); - } - - #[test] - fn keeps_mixed_enum_children_addressable() { - let source = r"enum Message { - Ok, - Error { - code: u32, - }, - }"; - let tree = build_chunk_tree(source, "rust").expect("tree should build"); - let enum_chunk = tree - .chunks - .iter() - .find(|c| c.path == "en_Mes") - .expect("en_Mes"); - assert!(!enum_chunk.leaf); - assert!(!enum_chunk.children.is_empty(), "mixed-size variants should stay addressable"); - } - - #[test] - fn typescript_namespace_members_stay_addressable() { - let source = r"namespace Foo { - export function bar() { - return 1; - } - } - "; - let tree = build_chunk_tree(source, "typescript").expect("tree should build"); - - let module = tree - .chunks - .iter() - .find(|c| c.path == "mod_Foo") - .expect("mod_Foo"); - assert!(!module.leaf); - assert!( - module - .children - .iter() - .any(|child| child == "mod_Foo.fn_bar"), - "expected fn_bar inside namespace, got {:?}", - module.children - ); - } - - #[test] - fn php_namespace_definition_keeps_inner_members_addressable() { - let source = " LEAF_THRESHOLD) - // with multiple control-flow children that narrow scope enough to trigger - // recursion, plus a `return` statement that must NOT become a standalone - // chunk. - let mut body = String::new(); - body.push_str("class Server:\n"); - body.push_str(" @property\n"); - body.push_str(" def address(self) -> str:\n"); - body.push_str(" if self._host:\n"); - for i in 0..20 { - let _ = writeln!(body, " x{i} = {i}"); - } - body.push_str(" if self._port:\n"); - for i in 0..20 { - let _ = writeln!(body, " y{i} = {i}"); - } - body.push_str(" return f\"{{self._host}}:{{self._port}}\"\n"); - let source = body; - let tree = build_chunk_tree(source.as_str(), "python").expect("tree should build"); - - // Dump all chunk paths for debugging - for chunk in &tree.chunks { - eprintln!( - "chunk: path={:?} kind={:?} leaf={} lines={}-{}", - chunk.path, chunk.kind, chunk.leaf, chunk.start_line, chunk.end_line - ); - } - - // The function body should recurse (it's large enough) but the return - // statement should NOT become a standalone addressable chunk. - let fn_chunk = tree - .chunks - .iter() - .find(|c| c.path == "cls_Ser.fn_add") - .expect("fn_add should exist"); - - let orphan_ret = tree.chunks.iter().find(|c| c.path.contains("ret")); - assert!( - orphan_ret.is_none(), - "return statement inside property method should not be a separate chunk, found: {:?}", - orphan_ret.map(|c| (&c.path, &c.kind)) - ); - - // Verify the function actually recursed (has children) - assert!(!fn_chunk.leaf, "fn_add should recurse into children for this test to be meaningful"); - } - - #[test] - fn python_region_boundaries_correct_for_function() { - use super::{resolve::chunk_region_range, types::ChunkRegion}; - - let source = "def run():\n return 1\n"; - let tree = build_chunk_tree(source, "python").expect("tree should build"); - let fn_chunk = tree - .chunks - .iter() - .find(|c| c.path == "fn_run") - .expect("fn_run"); - - let (head_s, head_e) = chunk_region_range(fn_chunk, ChunkRegion::Head); - let head = &source[head_s..head_e]; - assert!(head.contains("def run():"), "fn ^ should contain def signature, got {head:?}"); - assert!(!head.contains("return"), "fn ^ should not contain body, got {head:?}"); - - let (body_s, body_e) = chunk_region_range(fn_chunk, ChunkRegion::Body); - let body = &source[body_s..body_e]; - assert!(body.contains("return 1"), "fn ~ should contain body, got {body:?}"); - assert!(!body.contains("def run"), "fn ~ should not contain head, got {body:?}"); - assert!(body.ends_with('\n'), "fn ~ should end with newline, got {body:?}"); - } - - #[test] - fn python_region_boundaries_correct_for_class() { - use super::{resolve::chunk_region_range, types::ChunkRegion}; - - let source = - "class Server:\n def start(self) -> None:\n self.running = True\n \ - print('ok')\n\n def stop(self):\n pass\n"; - let tree = build_chunk_tree(source, "python").expect("tree should build"); - let class_chunk = tree - .chunks - .iter() - .find(|c| c.path == "cls_Ser") - .expect("cls_Ser"); - - let (head_s, head_e) = chunk_region_range(class_chunk, ChunkRegion::Head); - let head = &source[head_s..head_e]; - assert!(head.contains("class Server:"), "class ^ should contain class def, got {head:?}"); - assert!(!head.contains("def start"), "class ^ should not contain methods, got {head:?}"); - - let (body_s, body_e) = chunk_region_range(class_chunk, ChunkRegion::Body); - let body = &source[body_s..body_e]; - assert!(body.contains("def start"), "class ~ should contain methods, got {body:?}"); - assert!(body.contains("def stop"), "class ~ should contain all methods, got {body:?}"); - assert!(!body.contains("class Server"), "class ~ should not contain header, got {body:?}"); - } - - #[test] - fn python_decorated_function_region_boundaries() { - use super::{resolve::chunk_region_range, types::ChunkRegion}; - - let source = - "class Server:\n @property\n def address(self) -> str:\n return self._addr\n"; - let tree = build_chunk_tree(source, "python").expect("tree should build"); - let fn_chunk = tree - .chunks - .iter() - .find(|c| c.path == "cls_Ser.fn_add") - .expect("fn_add"); - - let (head_s, head_e) = chunk_region_range(fn_chunk, ChunkRegion::Head); - let head = &source[head_s..head_e]; - assert!(head.contains("@property"), "^ should include decorator, got {head:?}"); - assert!(head.contains("def address"), "^ should include def, got {head:?}"); - assert!(!head.contains("return"), "^ should not include body, got {head:?}"); - - let (body_s, body_e) = chunk_region_range(fn_chunk, ChunkRegion::Body); - let body = &source[body_s..body_e]; - assert!(body.contains("return self._addr"), "~ should contain return, got {body:?}"); - assert!(!body.contains("@property"), "~ should not contain decorator, got {body:?}"); - assert!(!body.contains("def address"), "~ should not contain def, got {body:?}"); - } - - #[test] - fn python_body_replace_preserves_surrounding_code() { - use super::{resolve::chunk_region_range, types::ChunkRegion}; - - let source = - "import os\n\ndef main():\n x = 1\n print(x)\n\ndef helper():\n return 42\n"; - let tree = build_chunk_tree(source, "python").expect("tree should build"); - let fn_main = tree - .chunks - .iter() - .find(|c| c.path == "fn_mai") - .expect("fn_mai"); - - let (body_s, body_e) = chunk_region_range(fn_main, ChunkRegion::Body); - let body = &source[body_s..body_e]; - assert!(!body.contains("def main"), "body should not include def"); - assert!(body.contains("x = 1"), "body should contain body lines"); - assert!(!body.contains("def helper"), "body should not leak into next function"); - assert!(!body.contains("return 42"), "body should not include helper's code"); - assert!(!body.contains("import"), "body should not include imports"); - - // Simulate body replace and verify the result - let replacement = " y = 2\n print(y)\n"; - let mut result = String::new(); - result.push_str(&source[..body_s]); - result.push_str(replacement); - result.push_str(&source[body_e..]); - assert!( - result.starts_with("import os"), - "import should remain at column 0 after body replace: {:?}", - &result[..40.min(result.len())] - ); - assert!(result.contains("def helper"), "helper fn should survive body replace: {result:?}"); - assert!(result.contains("return 42"), "helper's body should survive: {result:?}"); - } - - #[test] - fn python_class_body_replace_does_not_corrupt() { - use super::{resolve::chunk_region_range, types::ChunkRegion}; - - let source = - "class Server:\n def __init__(self, host: str, port: int):\n self.host = \ - host\n self.port = port\n\n def start(self) -> None:\n if \ - self.running:\n raise RuntimeError(\"already running\")\n \ - self.running = True\n print(f\"Started on {self.host}:{self.port}\")\n\n \ - @property\n def address(self) -> str:\n return f\"{self.host}:{self.port}\"\n"; - let tree = build_chunk_tree(source, "python").expect("tree should build"); - - // Verify class regions - let cls = tree - .chunks - .iter() - .find(|c| c.path == "cls_Ser") - .expect("cls_Ser"); - let (body_s, body_e) = chunk_region_range(cls, ChunkRegion::Body); - let body = &source[body_s..body_e]; - assert!(body.contains("def __init__"), "class body should contain __init__"); - assert!(body.contains("def start"), "class body should contain start"); - assert!(body.contains("@property"), "class body should contain @property"); - assert!(body.contains("def address"), "class body should contain address"); - assert!(!body.contains("class Server"), "class body should not contain header"); - - // Verify fn_start regions - let fn_start = tree - .chunks - .iter() - .find(|c| c.path == "cls_Ser.fn_sta") - .expect("fn_sta"); - let (fn_body_s, fn_body_e) = chunk_region_range(fn_start, ChunkRegion::Body); - let fn_body = &source[fn_body_s..fn_body_e]; - assert!(fn_body.contains("if self.running"), "fn_sta ~ should contain body, got {fn_body:?}"); - assert!( - fn_body.contains("self.running = True"), - "fn_sta ~ should contain all lines, got {fn_body:?}" - ); - assert!(!fn_body.contains("def start"), "fn_sta ~ should not include head, got {fn_body:?}"); - assert!( - !fn_body.contains("@property"), - "fn_sta ~ should not leak into next method, got {fn_body:?}" - ); - } - - #[test] - fn diff_block_produces_file_chunks() { - let source = "diff --git a/src/foo.ts b/src/foo.ts\nindex abcdef0..1234567 100644\n--- \ - a/src/foo.ts\n+++ b/src/foo.ts\n@@ -1,3 +1,4 @@\n line1\n+added\n line2\n \ - line3\ndiff --git a/src/bar.ts b/src/bar.ts\nindex 1111111..2222222 \ - 100644\n--- a/src/bar.ts\n+++ b/src/bar.ts\n@@ -5,2 +5,3 @@\n x\n+y\n z\n"; - let tree = build_chunk_tree(source, "diff").expect("diff tree should build"); - assert!( - tree.root_children.contains(&"file_src_1".to_string()), - "expected file_src_1, got {:?}", - tree.root_children - ); - assert!( - tree.root_children.contains(&"file_src_1".to_string()), - "expected file_src_1, got {:?}", - tree.root_children - ); - } - - #[test] - fn diff_hunks_individually_addressable() { - let source = "diff --git a/app.rs b/app.rs\nindex abcdef0..1234567 100644\n--- \ - a/app.rs\n+++ b/app.rs\n@@ -1,3 +1,4 @@\n line1\n+added1\n line2\n line3\n@@ \ - -10,3 +11,4 @@\n line10\n+added2\n line11\n line12\n"; - let tree = build_chunk_tree(source, "diff").expect("diff tree should build"); - let file_chunk = tree - .chunks - .iter() - .find(|c| c.path == "file_app") - .expect("file_app chunk should exist"); - assert!(!file_chunk.leaf, "file chunk with hunks should not be leaf"); - assert!( - file_chunk.children.iter().any(|c| c == "file_app.hunk_1"), - "expected hunk_1 child, got {:?}", - file_chunk.children - ); - assert!( - file_chunk.children.iter().any(|c| c == "file_app.hunk_2"), - "expected hunk_2 child, got {:?}", - file_chunk.children - ); - } - - #[test] - fn diff_deleted_file_uses_old_path() { - let source = "diff --git a/old.txt b/old.txt\ndeleted file mode 100644\nindex \ - abcdef0..0000000 100644\n--- a/old.txt\n+++ /dev/null\n@@ -1,2 +0,0 \ - @@\n-line1\n-line2\n"; - let tree = build_chunk_tree(source, "diff").expect("diff tree should build"); - assert!( - tree.root_children.contains(&"file_old".to_string()), - "expected file_old for deleted file, got {:?}", - tree.root_children - ); - } - - #[test] - fn diff_single_hunk_has_no_suffix() { - let source = "diff --git a/one.rs b/one.rs\nindex abcdef0..1234567 100644\n--- \ - a/one.rs\n+++ b/one.rs\n@@ -1,2 +1,3 @@\n line1\n+added\n line2\n"; - let tree = build_chunk_tree(source, "diff").expect("diff tree should build"); - let file_chunk = tree - .chunks - .iter() - .find(|c| c.path == "file_one") - .expect("file_one should exist"); - assert!(!file_chunk.leaf); - // Single hunk should be named "hunk" without a numeric suffix - assert!( - file_chunk.children.iter().any(|c| c == "file_one.hunk"), - "expected hunk child (no suffix), got {:?}", - file_chunk.children - ); - } - - #[test] - fn visible_range_clip_shows_truncation_markers() { - // Build a TypeScript file with a function spanning lines 1-10. - let source = "function longFunc() {\nlet a = 1;\nlet b = 2;\nlet c = 3;\nlet d = 4;\nlet e \ - = 5;\nlet f = 6;\nlet g = 7;\nlet h = 8;\nreturn a + b + c + d + e + f + g + \ - h;\n}\n"; - let state = ChunkState::parse(source.to_string(), "typescript".to_string()) - .expect("state should parse"); - - // Read only lines 3-7 — the function spans L1-L11, so it should be clipped. - let result = state - .render_read(ReadRenderParams { - read_path: "test.ts:L3-L7".to_string(), - display_path: "test.ts".to_string(), - language_tag: Some("ts".to_string()), - omit_checksum: true, - anchor_style: Some(ChunkAnchorStyle::FullOmit), - absolute_line_range: None, - tab_replacement: Some(" ".to_string()), - normalize_indent: Some(true), - }) - .expect("rnd_rea should succeed"); - - // The output should keep the chunk head and tail context and collapse the - // omitted middle ranges with generic expansion markers. - assert!( - result.text.contains("1^|function longFunc() {"), - "should keep the chunk signature when the visible range clips the head: {}", - result.text - ); - assert!( - result.text.contains("9 |let h = 8;"), - "should keep tail context when the visible range clips the body: {}", - result.text - ); - assert!( - result.text.contains("let c = 3"), - "visible content should be rendered: {}", - result.text - ); - assert!( - result.text.contains("[truncated… sel=L2-L2 to expand]"), - "should show a generic truncation marker above the requested lines: {}", - result.text - ); - assert!( - result.text.contains("[truncated… sel=L8-L8 to expand]"), - "should show a generic truncation marker below the requested lines: {}", - result.text - ); - } - - #[test] - fn visible_range_no_clip_markers_when_chunk_fits() { - // A small file fully contained in the visible range should have no clip - // markers. - let source = "const x = 1;\nconst y = 2;\n"; - let state = ChunkState::parse(source.to_string(), "typescript".to_string()) - .expect("state should parse"); - - let result = state - .render_read(ReadRenderParams { - read_path: "test.ts:L1-L2".to_string(), - display_path: "test.ts".to_string(), - language_tag: Some("ts".to_string()), - omit_checksum: true, - anchor_style: Some(ChunkAnchorStyle::FullOmit), - absolute_line_range: None, - tab_replacement: Some(" ".to_string()), - normalize_indent: Some(true), - }) - .expect("rnd_rea should succeed"); - - // Should NOT have any truncation markers. - assert!( - !result.text.contains("[truncated…"), - "no clip marker when chunk fits: {}", - result.text - ); - } - - #[test] - fn nix_let_expression_bindings_are_individually_addressable() { - // A Nix file with { args }: let bindings in body should produce - // individual chunks for each binding, not a single opaque chunk. - let source = r"{ pkgs }: - let - foo = 1; - bar = pkgs.hello; - baz = { - x = 1; - y = 2; - }; - in - { - inherit foo bar baz; - } - "; - let tree = build_chunk_tree(source, "nix").expect("nix tree should build"); - - // There should be chunks for individual bindings, not just a single - // opaque chunk containing all of them. - let has_foo = tree.chunks.iter().any(|c| c.path.contains("foo")); - let has_bar = tree.chunks.iter().any(|c| c.path.contains("bar")); - let has_baz = tree.chunks.iter().any(|c| c.path.contains("baz")); - - assert!( - has_foo && has_bar && has_baz, - "individual bindings should be addressable chunks. Chunks: {:?}", - tree.chunks.iter().map(|c| &c.path).collect::>() - ); - } - - #[test] - fn md_fenced_block_caret_read_falls_back_to_whole_chunk() { - // `^` on a markdown fenced code block should NOT return [Empty @^ region]. - // Fenced blocks have no meaningful head/body distinction, so `^` should - // fall back to whole-chunk rendering, matching the documented behavior. - let source = "# Title\n\nIntro text.\n\n```python\ndef hello():\n return 1\n```\n"; - let state = - ChunkState::parse(source.to_owned(), "markdown".to_owned()).expect("state should parse"); - let tree = state.inner().tree(); - let fence = tree - .chunks - .iter() - .find(|c| { - !c.path.is_empty() - && state - .inner() - .source() - .lines() - .nth(c.start_line.saturating_sub(1) as usize) - .is_some_and(|line| line.trim_start().starts_with("```")) - }) - .expect("fenced code block chunk should exist"); - - let read_path = format!("sample.md:{}#{}^", fence.path, fence.checksum); - let result = state - .render_read(ReadRenderParams { - read_path, - display_path: "sample.md".to_owned(), - language_tag: Some("md".to_owned()), - omit_checksum: false, - anchor_style: Some(ChunkAnchorStyle::Full), - absolute_line_range: None, - tab_replacement: Some(" ".to_owned()), - normalize_indent: Some(true), - }) - .expect("rnd_rea should succeed"); - - assert!( - !result.text.contains("[Empty @^ region]"), - "^ on fenced block must not return an empty region, got: {}", - result.text - ); - assert!( - result.text.contains("```python") || result.text.contains("def hello"), - "^ fallback must show fenced block content, got: {}", - result.text - ); - } - - #[test] - fn md_fenced_block_display_preserves_space_indentation() { - // When rendering a markdown fenced code block, the 4-space indentation - // inside the fence must NOT be normalized to tabs — code-block content is - // opaque to the chunk renderer. - let source = "# Title\n\n```python\ndef hello():\n return 1\n```\n"; - let state = - ChunkState::parse(source.to_owned(), "markdown".to_owned()).expect("state should parse"); - let result = state - .render_read(ReadRenderParams { - read_path: "sample.md".to_owned(), - display_path: "sample.md".to_owned(), - language_tag: Some("md".to_owned()), - omit_checksum: false, - anchor_style: Some(ChunkAnchorStyle::Full), - absolute_line_range: None, - tab_replacement: Some(" ".to_owned()), - normalize_indent: Some(true), - }) - .expect("rnd_rea should succeed"); - - assert!( - result.text.contains(" return 1"), - "fenced block must keep 4-space indentation in display, got:\n{}", - result.text - ); - assert!( - !result.text.contains("\treturn 1"), - "fenced block must NOT normalize spaces to tabs in display, got:\n{}", - result.text - ); - } - - #[test] - fn markdown_fenced_code_blocks_build_injected_subtrees() { - let source = "# Title\n\n```js\nfunction hello(name) {\n return name;\n}\n```\n"; - let tree = build_chunk_tree(source, "markdown").expect("markdown tree"); - let fence = tree - .chunks - .iter() - .find(|chunk| chunk.path == "sct_Tit.code_js") - .expect("expected semantic fenced-code chunk"); - - assert!(!fence.leaf, "fenced block should recurse into the injected JS tree"); - assert!( - tree - .chunks - .iter() - .any(|chunk| chunk.path == "sct_Tit.code_js.fn_hel"), - "expected translated JS descendant under code_js, got {:?}", - tree - .chunks - .iter() - .map(|chunk| &chunk.path) - .collect::>() - ); - assert!( - fence.prologue_end_byte.is_none() && fence.epilogue_start_byte.is_none(), - "markdown fenced host should stay regionless" - ); - } - - #[test] - fn html_script_and_style_hosts_recurse_into_embedded_languages() { - let source = "
\n\n\n
\n"; - let tree = build_chunk_tree(source, "html").expect("html tree"); - let script = tree - .chunks - .iter() - .find(|chunk| chunk.path == "tag_div.scr") - .expect("expected script host"); - let style = tree - .chunks - .iter() - .find(|chunk| chunk.path == "tag_div.sty") - .expect("expected style host"); - - assert!(!script.leaf, "script host should expose nested JS chunks"); - assert!(!style.leaf, "style host should expose nested CSS chunks"); - assert!( - tree - .chunks - .iter() - .any(|chunk| chunk.path.starts_with("tag_div.scr.") && chunk.path != "tag_div.scr"), - "expected translated JS descendants, got {:?}", - tree - .chunks - .iter() - .map(|chunk| &chunk.path) - .collect::>() - ); - assert!( - tree - .chunks - .iter() - .any(|chunk| chunk.path.starts_with("tag_div.sty.") && chunk.path != "tag_div.sty"), - "expected translated CSS descendants, got {:?}", - tree - .chunks - .iter() - .map(|chunk| &chunk.path) - .collect::>() - ); - assert!(script.prologue_end_byte.is_some(), "script host should expose a body region"); - assert!(style.prologue_end_byte.is_some(), "style host should expose a body region"); - } - - #[test] - fn small_html_wrappers_do_not_hide_embedded_script_hosts() { - let source = "
\n"; - let tree = build_chunk_tree(source, "html").expect("html tree"); - - assert!( - tree.chunks.iter().any(|chunk| chunk.path == "tag_div.scr"), - "script host should stay addressable inside a small wrapper, got {:?}", - tree - .chunks - .iter() - .map(|chunk| &chunk.path) - .collect::>() - ); - } - - #[test] - fn framework_block_hosts_expose_injected_descendants() { - let cases = [ - ( - "astro", - "---\nconst title: string = \"Hello\";\n---\n\n\n", - &["fm", "scr", "sty"][..], - ), - ( - "svelte", - "\n\n
{count}
\n", - &["scr", "sty"][..], - ), - ( - "vue", - "\n\n\n", - &["sse", "sco"][..], - ), - ]; - - for (language, source, hosts) in cases { - let tree = build_chunk_tree(source, language).unwrap_or_else(|err| { - panic!("expected {language} tree to build: {err}"); - }); - for host in hosts { - let chunk = tree - .chunks - .iter() - .find(|chunk| chunk.path == *host) - .unwrap_or_else(|| panic!("missing {language} host {host}")); - assert!(!chunk.leaf, "{language} host {host} should expose embedded descendants"); - } - } - } -} diff --git a/crates/pi-natives/src/chunk/render.rs b/crates/pi-natives/src/chunk/render.rs deleted file mode 100644 index 8ecd66aa5..000000000 --- a/crates/pi-natives/src/chunk/render.rs +++ /dev/null @@ -1,1299 +0,0 @@ -use std::collections::{HashMap, HashSet}; - -use crate::{ - chunk::{ - indent::{detect_file_indent_char, detect_file_indent_step, normalize_to_tabs}, - state::{ChunkStateInner, mask_chunk_display_source}, - types::{ - ChunkAnchorStyle, ChunkFocusMode, ChunkNode, ChunkTree, RenderParams, VisibleLineRange, - }, - }, - env_uint, -}; - -type ChunkLookup<'a> = HashMap<&'a str, &'a ChunkNode>; - -#[derive(Clone)] -pub struct InlineHunkLine { - /// Fully indented line text ready to push as a meta line. - pub text: String, - /// Optional gutter marker (`*`, `-`, etc.) for the rendered meta line. - pub marker: Option, -} - -/// A pre-formatted diff hunk ready for inline display inside a chunk block. -pub struct InlineHunk { - /// Fully indented lines (header + diff lines) ready to push as meta lines. - pub lines: Vec, -} - -env_uint! { - // Configured full display threshold. - static FULL_DISPLAY_THRESHOLD: usize = "PI_CHUNK_FULL_DISPLAY_THRESHOLD" or 80 => [1, usize::MAX]; - // Configured preview head lines. - static PREVIEW_HEAD_LINES: usize = "PI_CHUNK_PREVIEW_HEAD_LINES" or 12 => [1, usize::MAX]; - // Configured preview tail lines. - static PREVIEW_TAIL_LINES: usize = "PI_CHUNK_PREVIEW_TAIL_LINES" or 4 => [1, usize::MAX]; -} - -const CLIPPED_TAIL_CONTEXT_LINES: u32 = 3; - -fn normalize_rendered_line( - line: &str, - normalize_indent: Option<(char, usize)>, - tab_replacement: &str, -) -> String { - match normalize_indent { - Some((indent_char, indent_step)) => normalize_to_tabs(line, indent_char, indent_step), - None => line.replace('\t', tab_replacement), - } -} - -/// Detect a `CommonMark` fence marker (three backticks or three tildes) at the -/// start of trimmed text. Returns the marker character and marker length when -/// at least 3 consecutive markers begin the line. -fn fence_marker(trimmed: &str) -> Option<(char, usize)> { - let first = trimmed.as_bytes().first().copied()?; - if first != b'`' && first != b'~' { - return None; - } - let len = trimmed.bytes().take_while(|&b| b == first).count(); - (len >= 3).then_some((first as char, len)) -} - -/// Returns 1-indexed line numbers that fall strictly inside a fenced code -/// block in a markdown / handlebars file. Opening and closing fence lines -/// are excluded (only opaque content lines are returned). -/// For non-prose languages the set is always empty. -fn compute_fenced_code_lines(source_lines: &[&str], language: &str) -> HashSet { - if !matches!(language, "markdown" | "handlebars") { - return HashSet::new(); - } - let mut fenced = HashSet::new(); - let mut open_fence: Option<(u8, usize)> = None; - for (idx, line) in source_lines.iter().enumerate() { - let line_no = idx as u32 + 1; - let trimmed = line.trim_start(); - match open_fence { - None => { - if let Some((marker, len)) = fence_marker(trimmed) { - open_fence = Some((marker as u8, len)); - } - }, - Some((marker, min_len)) => { - // Closing fence: same char, length >= opening, only whitespace after. - if let Some((m, len)) = fence_marker(trimmed) - && m as u8 == marker - && len >= min_len - && trimmed[len..].trim().is_empty() - { - open_fence = None; - } else { - fenced.insert(line_no); - } - }, - } - } - fenced -} - -pub fn render_state(state: &ChunkStateInner, params: &RenderParams) -> String { - render_state_impl(state, params, HashMap::new(), HashSet::new(), false) -} - -pub fn render_state_with_hunks( - state: &ChunkStateInner, - params: &RenderParams, - inline_hunks: HashMap>, - changed_anchor_paths: HashSet, -) -> String { - render_state_impl(state, params, inline_hunks, changed_anchor_paths, true) -} - -fn render_state_impl( - state: &ChunkStateInner, - params: &RenderParams, - inline_hunks: HashMap>, - changed_anchor_paths: HashSet, - compact_meta: bool, -) -> String { - let tree = state.tree(); - let lookup = build_lookup(tree); - let chunk_path = params - .chunk_path - .as_deref() - .unwrap_or(tree.root_path.as_str()); - let Some(chunk) = get_chunk(&lookup, chunk_path) else { - return String::new(); - }; - let masked_source = mask_chunk_display_source(state.source(), state.language()); - let source_lines = masked_source.split('\n').collect::>(); - let full_display_threshold = *FULL_DISPLAY_THRESHOLD; - let preview_head_lines = *PREVIEW_HEAD_LINES; - let preview_tail_lines = *PREVIEW_TAIL_LINES; - let tab_replacement = params.tab_replacement.as_deref().unwrap_or(" "); - let normalize_indent = params.normalize_indent.unwrap_or(false).then(|| { - ( - detect_file_indent_char(state.source(), tree), - detect_file_indent_step(state.source(), tree) as usize, - ) - }); - - let fenced_lines = compute_fenced_code_lines(&source_lines, &tree.language); - - let anchor_style = params.anchor_style.unwrap_or_default(); - let focus: Option> = params - .focused_paths - .as_ref() - .map(|paths| paths.iter().map(|fp| (fp.path.as_str(), fp.mode)).collect()); - let num_width = compute_num_width( - tree, - chunk, - &lookup, - params.visible_range.as_ref(), - params.render_children_only, - params.show_leaf_preview, - &masked_source, - &source_lines, - tab_replacement, - normalize_indent, - &fenced_lines, - full_display_threshold, - preview_head_lines, - preview_tail_lines, - ); - let rendered_line_count = compute_rendered_line_count( - tree, - chunk, - &lookup, - params.visible_range.as_ref(), - params.render_children_only, - params.show_leaf_preview, - &masked_source, - &source_lines, - tab_replacement, - normalize_indent, - &fenced_lines, - full_display_threshold, - preview_head_lines, - preview_tail_lines, - ); - - let mut ctx = RenderCtx { - out: String::new(), - tree, - lookup: &lookup, - source: &masked_source, - source_lines: &source_lines, - num_width, - visible_range: params.visible_range.as_ref(), - omit_checksum: params.omit_checksum, - anchor_style, - show_leaf_preview: params.show_leaf_preview, - last_was_blank_meta: false, - full_display_threshold, - preview_head_lines, - preview_tail_lines, - tab_replacement, - normalize_indent, - fenced_lines, - focus, - inline_hunks, - compact_meta, - changed_anchor_paths, - }; - - push_meta( - &mut ctx, - format_header_meta( - params.title.as_str(), - rendered_line_count, - params.language_tag.as_deref(), - chunk.checksum.as_str(), - params.omit_checksum, - ), - ); - push_blank_meta(&mut ctx); - - if params.render_children_only { - let focus_ref = ctx.focus.as_ref(); - let children = - visible_children_for_chunk(tree, chunk, &lookup, params.visible_range.as_ref(), focus_ref); - for (index, child) in children.iter().enumerate() { - emit_chunk_subtree(&mut ctx, child, 0, ChunkSubtreeOptions { - is_first_top_level: index == 0, - between_top_level_definitions: true, - }); - } - emit_inline_hunks_for(&mut ctx, ""); - return ctx.out; - } - - if chunk.children.is_empty() && ctx.focus.is_none() { - if params.show_leaf_preview - && intersect_visible_span(chunk, params.visible_range.as_ref()).is_some() - { - emit_chunk_subtree(&mut ctx, chunk, 0, ChunkSubtreeOptions { - is_first_top_level: true, - between_top_level_definitions: false, - }); - } - return ctx.out; - } - - emit_chunk_subtree(&mut ctx, chunk, 0, ChunkSubtreeOptions { - is_first_top_level: true, - between_top_level_definitions: false, - }); - ctx.out -} - -fn build_lookup(tree: &ChunkTree) -> ChunkLookup<'_> { - tree - .chunks - .iter() - .map(|chunk| (chunk.path.as_str(), chunk)) - .collect() -} - -fn get_chunk<'a>(lookup: &ChunkLookup<'a>, chunk_path: &str) -> Option<&'a ChunkNode> { - lookup.get(chunk_path).copied() -} - -fn line_to_chunk_path_leaf(tree: &ChunkTree, line: u32) -> Option<&ChunkNode> { - if line == 0 { - return None; - } - - tree - .chunks - .iter() - .filter(|chunk| { - chunk.leaf - && chunk.virtual_content.is_none() - && chunk.start_line <= line - && line <= chunk.end_line - && !chunk.path.is_empty() - }) - .min_by_key(|chunk| chunk.line_count) -} - -fn smallest_containing_chunk(tree: &ChunkTree, line: u32) -> Option<&ChunkNode> { - let mut best: Option<&ChunkNode> = None; - for chunk in &tree.chunks { - if chunk.path.is_empty() - || chunk.virtual_content.is_some() - || chunk.start_line > line - || line > chunk.end_line - { - continue; - } - if best.is_none_or(|current| chunk.line_count < current.line_count) { - best = Some(chunk); - } - } - best -} - -fn line_to_containing_chunk(tree: &ChunkTree, line: u32) -> Option<&ChunkNode> { - if let Some(chunk) = line_to_chunk_path_leaf(tree, line) { - return Some(chunk); - } - smallest_containing_chunk(tree, line) -} - -const fn chunk_intersects_line_range(chunk: &ChunkNode, visible_range: &VisibleLineRange) -> bool { - chunk.start_line <= visible_range.end_line && visible_range.start_line <= chunk.end_line -} - -fn chunk_or_descendant_intersects_line_range( - tree: &ChunkTree, - chunk: &ChunkNode, - lookup: &ChunkLookup<'_>, - visible_range: &VisibleLineRange, -) -> bool { - if chunk_intersects_line_range(chunk, visible_range) { - return true; - } - for child_path in &chunk.children { - if let Some(child) = get_chunk(lookup, child_path.as_str()) - && chunk_or_descendant_intersects_line_range(tree, child, lookup, visible_range) - { - return true; - } - } - let _ = tree; - false -} - -fn visible_children_for_chunk<'a>( - tree: &'a ChunkTree, - chunk: &'a ChunkNode, - lookup: &ChunkLookup<'a>, - visible_range: Option<&VisibleLineRange>, - focus: Option<&HashMap<&str, ChunkFocusMode>>, -) -> Vec<&'a ChunkNode> { - let mut children = chunk - .children - .iter() - .filter_map(|child_path| get_chunk(lookup, child_path.as_str())) - .filter(|child| { - visible_range.is_none_or(|range| { - chunk_or_descendant_intersects_line_range(tree, child, lookup, range) - }) - }) - .filter(|child| focus.is_none_or(|map| map.contains_key(child.path.as_str()))) - .collect::>(); - children.sort_by(|left, right| { - left - .start_line - .cmp(&right.start_line) - .then_with(|| left.start_byte.cmp(&right.start_byte)) - .then_with(|| left.path.cmp(&right.path)) - }); - children -} - -fn leading_whitespace(line: &str) -> &str { - let count = line - .chars() - .take_while(|ch| *ch == ' ' || *ch == '\t') - .map(char::len_utf8) - .sum::(); - &line[..count] -} - -fn chunk_body_anchor_indent( - source_lines: &[&str], - chunk: &ChunkNode, - tab_replacement: &str, - normalize_indent: Option<(char, usize)>, -) -> String { - source_lines - .get(chunk.start_line.saturating_sub(1) as usize) - .map_or(String::new(), |line| { - leading_whitespace(&normalize_rendered_line(line, normalize_indent, tab_replacement)) - .to_owned() - }) -} - -fn chunk_anchor_label(chunk: &ChunkNode, style: ChunkAnchorStyle) -> String { - match style { - ChunkAnchorStyle::Full | ChunkAnchorStyle::FullOmit => chunk.path.clone(), - ChunkAnchorStyle::Kind - | ChunkAnchorStyle::KindOmit - | ChunkAnchorStyle::Bare - | ChunkAnchorStyle::None => chunk.kind.path_segment(chunk.identifier.as_deref()), - } -} - -/// Compute head and body line counts for a chunk. -/// Head = lines covered by the prologue region (signature through opening -/// delimiter). Body = remaining lines (interior through closing delimiter). -/// When the chunk has no region boundaries, head = total, body = 0. -fn chunk_head_body_lines(source: &str, chunk: &ChunkNode) -> (u32, u32) { - let total = chunk.line_count; - if total == 0 { - return (0, 0); - } - let Some(pro_end) = chunk.prologue_end_byte else { - return (total, 0); - }; - let start = chunk.start_byte as usize; - let end = chunk.end_byte as usize; - let pro_end = (pro_end as usize).clamp(start, end); - if pro_end <= start { - return (0, total); - } - let bytes = source.as_bytes(); - // Count newlines strictly within [start, pro_end). - #[expect(clippy::naive_bytecount, reason = "small head region, memchr dep not wired in")] - let head_newlines = bytes[start..pro_end] - .iter() - .filter(|&&b| b == b'\n') - .count() as u32; - // When the prologue ends exactly on a newline, the newline terminates the - // last head line, so head_lines == head_newlines. Otherwise the prologue - // ends mid-line and we're on the line following the last newline. - let head_ends_at_newline = bytes.get(pro_end - 1).copied() == Some(b'\n'); - let raw_head_lines = if head_ends_at_newline { - head_newlines.max(1) - } else { - head_newlines + 1 - }; - let head_lines = raw_head_lines.min(total); - let body_lines = total.saturating_sub(head_lines); - (head_lines, body_lines) -} - -#[derive(Clone, Copy)] -struct VisibleSpan { - start: u32, - end: u32, -} - -#[derive(Clone, Copy)] -struct LineSegment { - start: u32, - end: u32, -} - -fn intersect_visible_span( - chunk: &ChunkNode, - visible_range: Option<&VisibleLineRange>, -) -> Option { - let low = chunk.start_line; - let high = chunk.end_line; - match visible_range { - None => Some(VisibleSpan { start: low, end: high }), - Some(range) => { - let start = low.max(range.start_line); - let end = high.min(range.end_line); - (start <= end).then_some(VisibleSpan { start, end }) - }, - } -} - -fn line_in_file_scope(line: u32, visible_range: Option<&VisibleLineRange>) -> bool { - visible_range.is_none_or(|range| line >= range.start_line && line <= range.end_line) -} - -#[derive(Clone)] -enum LeafEntry { - Line { abs_line: u32, text: String }, - Ellipsis { start_abs: u32, end_abs: u32 }, -} - -fn clipped_head_context_line_count( - source: &str, - chunk: &ChunkNode, - preview_head_lines: usize, -) -> u32 { - let (head_lines, body_lines) = chunk_head_body_lines(source, chunk); - let fallback = preview_head_lines.max(1) as u32; - if head_lines == 0 { - return fallback; - } - if body_lines == 0 { - return head_lines.min(fallback); - } - head_lines -} - -fn push_segment(segments: &mut Vec, start: u32, end: u32) { - if start > end { - return; - } - if let Some(last) = segments.last_mut() - && start <= last.end.saturating_add(1) - { - last.end = last.end.max(end); - return; - } - segments.push(LineSegment { start, end }); -} - -fn clipped_leaf_segments( - source: &str, - chunk: &ChunkNode, - span: VisibleSpan, - preview_head_lines: usize, -) -> Vec { - let mut segments = Vec::new(); - if span.start > chunk.start_line { - let head_context_lines = clipped_head_context_line_count(source, chunk, preview_head_lines); - let preview_end = chunk - .start_line - .saturating_add(head_context_lines.saturating_sub(1)) - .min(chunk.end_line) - .min(span.start.saturating_sub(1)); - push_segment(&mut segments, chunk.start_line, preview_end); - } - push_segment(&mut segments, span.start, span.end); - if span.end < chunk.end_line { - let preview_start = chunk - .end_line - .saturating_add(1) - .saturating_sub(CLIPPED_TAIL_CONTEXT_LINES) - .max(chunk.start_line) - .max(span.end.saturating_add(1)); - push_segment(&mut segments, preview_start, chunk.end_line); - } - segments -} - -fn build_leaf_entries( - source_lines: &[&str], - source: &str, - chunk: &ChunkNode, - span: VisibleSpan, - tab_replacement: &str, - normalize_indent: Option<(char, usize)>, - fenced_lines: &HashSet, - full_display_threshold: usize, - preview_head_lines: usize, - preview_tail_lines: usize, -) -> Vec { - let low = span.start; - let high = span.end; - let visible_line_count = (high - low + 1) as usize; - let make_line = |line: u32| { - let normalize = if fenced_lines.contains(&line) { - None - } else { - normalize_indent - }; - LeafEntry::Line { - abs_line: line, - text: source_lines - .get(line.saturating_sub(1) as usize) - .map_or(String::new(), |text| { - normalize_rendered_line(text, normalize, tab_replacement) - }), - } - }; - - if span.start > chunk.start_line || span.end < chunk.end_line { - let segments = clipped_leaf_segments(source, chunk, span, preview_head_lines); - let mut entries = Vec::new(); - let mut last_end: Option = None; - for segment in segments { - if let Some(previous_end) = last_end { - let gap_start = previous_end.saturating_add(1); - let gap_end = segment.start.saturating_sub(1); - if gap_start <= gap_end { - entries.push(LeafEntry::Ellipsis { start_abs: gap_start, end_abs: gap_end }); - } - } - for line in segment.start..=segment.end { - entries.push(make_line(line)); - } - last_end = Some(segment.end); - } - return entries; - } - - let raw = (low..=high).map(make_line).collect::>(); - - if visible_line_count <= full_display_threshold { - return raw; - } - - let head = raw - .iter() - .take(preview_head_lines) - .cloned() - .collect::>(); - let tail = raw - .iter() - .rev() - .take(preview_tail_lines) - .cloned() - .collect::>() - .into_iter() - .rev() - .collect::>(); - let _omitted = visible_line_count.saturating_sub(head.len() + tail.len()); - let first_omitted = low + head.len() as u32; - let last_omitted = high.saturating_sub(tail.len() as u32); - let mut entries = Vec::with_capacity(head.len() + tail.len() + 1); - entries.extend(head); - entries.push(LeafEntry::Ellipsis { start_abs: first_omitted, end_abs: last_omitted }); - entries.extend(tail); - entries -} - -fn format_header_meta( - title: &str, - line_count: usize, - language_tag: Option<&str>, - checksum: &str, - omit_checksum: bool, -) -> String { - let language = language_tag.unwrap_or("text"); - let checksum_part = if omit_checksum { - String::new() - } else { - format!("·#{checksum}") - }; - format!("{title}·{line_count}L·{language}{checksum_part}") -} - -fn should_render_gap_line( - tree: &ChunkTree, - chunk: &ChunkNode, - lookup: &ChunkLookup<'_>, - line: u32, -) -> bool { - if chunk.path.is_empty() { - return true; - } - let children = visible_children_for_chunk(tree, chunk, lookup, None, None); - let has_out_of_span_child = children - .iter() - .any(|child| child.start_line < chunk.start_line || child.end_line > chunk.end_line); - if !has_out_of_span_child { - return true; - } - let Some(owner) = line_to_containing_chunk(tree, line) else { - return true; - }; - owner.path == chunk.path || owner.path.starts_with(&format!("{}.", chunk.path)) -} - -fn for_each_rendered_source_line( - tree: &ChunkTree, - chunk: &ChunkNode, - lookup: &ChunkLookup<'_>, - visible_range: Option<&VisibleLineRange>, - show_leaf_preview: bool, - source: &str, - source_lines: &[&str], - tab_replacement: &str, - normalize_indent: Option<(char, usize)>, - fenced_lines: &HashSet, - full_display_threshold: usize, - preview_head_lines: usize, - preview_tail_lines: usize, - visit: &mut impl FnMut(u32), -) { - let children = visible_children_for_chunk(tree, chunk, lookup, visible_range, None); - let span = intersect_visible_span(chunk, visible_range); - let has_kids = !children.is_empty(); - - if !chunk.path.is_empty() && span.is_none() && !has_kids { - return; - } - - if !has_kids { - if !show_leaf_preview { - return; - } - if let Some(span) = span { - for entry in build_leaf_entries( - source_lines, - source, - chunk, - span, - tab_replacement, - normalize_indent, - fenced_lines, - full_display_threshold, - preview_head_lines, - preview_tail_lines, - ) { - if let LeafEntry::Line { abs_line, .. } = entry { - visit(abs_line); - } - } - } - return; - } - - if let Some(span) = span { - if span.start > chunk.start_line { - let head_context_lines = - clipped_head_context_line_count(source, chunk, preview_head_lines); - let first_child_start = children - .first() - .map_or_else(|| chunk.end_line.saturating_add(1), |child| child.start_line); - let preview_end = chunk - .start_line - .saturating_add(head_context_lines.saturating_sub(1)) - .min(chunk.end_line) - .min(first_child_start.saturating_sub(1)) - .min(span.start.saturating_sub(1)); - for line in chunk.start_line..=preview_end { - if should_render_gap_line(tree, chunk, lookup, line) { - visit(line); - } - } - } - let mut cursor = chunk.start_line; - for child in &children { - let gap_end = child.start_line.saturating_sub(1); - if gap_end >= cursor { - for line in cursor..=gap_end { - if line_in_file_scope(line, visible_range) - && should_render_gap_line(tree, chunk, lookup, line) - { - visit(line); - } - } - } - for_each_rendered_source_line( - tree, - child, - lookup, - visible_range, - show_leaf_preview, - source, - source_lines, - tab_replacement, - normalize_indent, - fenced_lines, - full_display_threshold, - preview_head_lines, - preview_tail_lines, - visit, - ); - cursor = cursor.max(child.end_line.saturating_add(1)); - } - if cursor <= span.end { - for line in cursor..=span.end { - if line_in_file_scope(line, visible_range) { - visit(line); - } - } - } - if span.end < chunk.end_line { - let last_child_end = children.last().map_or(0, |child| child.end_line); - let preview_start = chunk - .end_line - .saturating_add(1) - .saturating_sub(CLIPPED_TAIL_CONTEXT_LINES) - .max(chunk.start_line) - .max(last_child_end.saturating_add(1)) - .max(span.end.saturating_add(1)); - for line in preview_start..=chunk.end_line { - if should_render_gap_line(tree, chunk, lookup, line) { - visit(line); - } - } - } - return; - } - - for child in children { - for_each_rendered_source_line( - tree, - child, - lookup, - visible_range, - show_leaf_preview, - source, - source_lines, - tab_replacement, - normalize_indent, - fenced_lines, - full_display_threshold, - preview_head_lines, - preview_tail_lines, - visit, - ); - } -} - -fn compute_rendered_line_count( - tree: &ChunkTree, - chunk: &ChunkNode, - lookup: &ChunkLookup<'_>, - visible_range: Option<&VisibleLineRange>, - render_children_only: bool, - show_leaf_preview: bool, - source: &str, - source_lines: &[&str], - tab_replacement: &str, - normalize_indent: Option<(char, usize)>, - fenced_lines: &HashSet, - full_display_threshold: usize, - preview_head_lines: usize, - preview_tail_lines: usize, -) -> usize { - if visible_range.is_none() { - if render_children_only { - // Root reads render child chunks only, but the header should still report the - // file's true total line count. - return if chunk.path.is_empty() { - tree.line_count as usize - } else { - chunk.line_count as usize - }; - } - let children = visible_children_for_chunk(tree, chunk, lookup, visible_range, None); - let has_out_of_span_child = children - .iter() - .any(|child| child.start_line < chunk.start_line || child.end_line > chunk.end_line); - if !has_out_of_span_child { - return chunk.line_count as usize; - } - } - let mut rendered_lines = std::collections::BTreeSet::new(); - for_each_rendered_source_line( - tree, - chunk, - lookup, - visible_range, - show_leaf_preview, - source, - source_lines, - tab_replacement, - normalize_indent, - fenced_lines, - full_display_threshold, - preview_head_lines, - preview_tail_lines, - &mut |line| { - rendered_lines.insert(line); - }, - ); - rendered_lines.len().max(1) -} - -struct RenderCtx<'a> { - out: String, - tree: &'a ChunkTree, - lookup: &'a ChunkLookup<'a>, - source: &'a str, - source_lines: &'a [&'a str], - num_width: usize, - visible_range: Option<&'a VisibleLineRange>, - omit_checksum: bool, - anchor_style: ChunkAnchorStyle, - show_leaf_preview: bool, - last_was_blank_meta: bool, - full_display_threshold: usize, - preview_head_lines: usize, - preview_tail_lines: usize, - tab_replacement: &'a str, - normalize_indent: Option<(char, usize)>, - fenced_lines: HashSet, - focus: Option>, - inline_hunks: HashMap>, - compact_meta: bool, - changed_anchor_paths: HashSet, -} - -fn push_line(out: &mut String, line: String) { - if !out.is_empty() { - out.push('\n'); - } - out.push_str(&line); -} - -fn push_blank_meta(ctx: &mut RenderCtx<'_>) { - if ctx.last_was_blank_meta { - return; - } - push_line(&mut ctx.out, format!("{} |", " ".repeat(ctx.num_width))); - ctx.last_was_blank_meta = true; -} - -fn push_meta_marked(ctx: &mut RenderCtx<'_>, body: String, marker: Option) { - ctx.last_was_blank_meta = false; - let gutter = match marker { - Some(marker) => format!("{marker}{}", " ".repeat(ctx.num_width)), - None => " ".repeat(ctx.num_width + 1), - }; - let separator = if ctx.compact_meta { "|" } else { "| " }; - push_line(&mut ctx.out, format!("{gutter}{separator}{body}")); -} - -fn push_meta(ctx: &mut RenderCtx<'_>, body: String) { - push_meta_marked(ctx, body, None); -} - -fn push_anchor(ctx: &mut RenderCtx<'_>, body: String, marker: Option) { - ctx.last_was_blank_meta = false; - let line = match marker { - Some(m) => format!("{m}{body}"), - None => body, - }; - push_line(&mut ctx.out, line); -} - -fn line_is_in_head(source: &str, chunk: &ChunkNode, abs_line: u32) -> bool { - let (head_lines, body_lines) = chunk_head_body_lines(source, chunk); - body_lines > 0 && abs_line >= chunk.start_line && abs_line < chunk.start_line + head_lines -} - -fn push_code(ctx: &mut RenderCtx<'_>, abs_line: u32, source_text: &str, head: bool) { - ctx.last_was_blank_meta = false; - let marker = if head { '^' } else { ' ' }; - push_line( - &mut ctx.out, - format!("{}{}|{}", abs_line.to_string().pad_start(ctx.num_width, ' '), marker, source_text), - ); -} - -trait PadStart { - fn pad_start(&self, width: usize, ch: char) -> String; -} - -impl PadStart for String { - fn pad_start(&self, width: usize, ch: char) -> String { - if self.len() >= width { - return self.clone(); - } - format!("{}{}", ch.to_string().repeat(width - self.len()), self) - } -} - -fn emit_line_gap(ctx: &mut RenderCtx<'_>, from: u32, to: u32, chunk: &ChunkNode) { - for line in from..=to { - if !line_in_file_scope(line, ctx.visible_range) { - continue; - } - let normalize = if ctx.fenced_lines.contains(&line) { - None - } else { - ctx.normalize_indent - }; - let text = ctx - .source_lines - .get(line.saturating_sub(1) as usize) - .map_or(String::new(), |text| { - normalize_rendered_line(text, normalize, ctx.tab_replacement) - }); - let head = line_is_in_head(ctx.source, chunk, line); - push_code(ctx, line, &text, head); - } -} - -fn push_truncation_marker(ctx: &mut RenderCtx<'_>, chunk: &ChunkNode, start: u32, end: u32) { - if start > end { - return; - } - let indent = - chunk_body_anchor_indent(ctx.source_lines, chunk, ctx.tab_replacement, ctx.normalize_indent); - push_meta(ctx, format!("{indent}[truncated\u{2026} sel=L{start}-L{end} to expand]")); -} - -fn emit_explicit_gap_lines(ctx: &mut RenderCtx<'_>, chunk: &ChunkNode, from: u32, to: u32) { - if from > to { - return; - } - for line in from..=to { - if !should_render_gap_line(ctx.tree, chunk, ctx.lookup, line) { - continue; - } - let normalize = if ctx.fenced_lines.contains(&line) { - None - } else { - ctx.normalize_indent - }; - let text = ctx - .source_lines - .get(line.saturating_sub(1) as usize) - .map_or(String::new(), |source_text| { - normalize_rendered_line(source_text, normalize, ctx.tab_replacement) - }); - let head = line_is_in_head(ctx.source, chunk, line); - push_code(ctx, line, &text, head); - } -} - -fn emit_container_clip_above( - ctx: &mut RenderCtx<'_>, - chunk: &ChunkNode, - span: &VisibleSpan, - children: &[&ChunkNode], -) { - if ctx.visible_range.is_none() || span.start <= chunk.start_line { - return; - } - let head_context_lines = - clipped_head_context_line_count(ctx.source, chunk, ctx.preview_head_lines); - let first_child_start = children - .first() - .map_or_else(|| chunk.end_line.saturating_add(1), |child| child.start_line); - let preview_end = chunk - .start_line - .saturating_add(head_context_lines.saturating_sub(1)) - .min(chunk.end_line) - .min(first_child_start.saturating_sub(1)) - .min(span.start.saturating_sub(1)); - if preview_end >= chunk.start_line { - emit_explicit_gap_lines(ctx, chunk, chunk.start_line, preview_end); - } - push_truncation_marker(ctx, chunk, preview_end.saturating_add(1), span.start.saturating_sub(1)); -} - -fn emit_container_clip_below( - ctx: &mut RenderCtx<'_>, - chunk: &ChunkNode, - span: &VisibleSpan, - children: &[&ChunkNode], -) { - if ctx.visible_range.is_none() || span.end >= chunk.end_line { - return; - } - let last_child_end = children.last().map_or(0, |child| child.end_line); - let preview_start = chunk - .end_line - .saturating_add(1) - .saturating_sub(CLIPPED_TAIL_CONTEXT_LINES) - .max(chunk.start_line) - .max(last_child_end.saturating_add(1)) - .max(span.end.saturating_add(1)); - push_truncation_marker(ctx, chunk, span.end.saturating_add(1), preview_start.saturating_sub(1)); - if preview_start <= chunk.end_line { - emit_explicit_gap_lines(ctx, chunk, preview_start, chunk.end_line); - } -} - -fn virtual_render_lines( - content: &str, - tab_replacement: &str, - normalize_indent: Option<(char, usize)>, -) -> Vec { - if content.is_empty() { - return Vec::new(); - } - let content = content.strip_suffix('\n').unwrap_or(content); - content - .split('\n') - .map(|line| normalize_rendered_line(line, normalize_indent, tab_replacement)) - .collect() -} - -fn emit_leaf_body(ctx: &mut RenderCtx<'_>, chunk: &ChunkNode, span: VisibleSpan) { - if let Some(content) = chunk.virtual_content.as_deref() { - for line in virtual_render_lines(content, ctx.tab_replacement, ctx.normalize_indent) { - push_meta(ctx, line); - } - return; - } - - for entry in build_leaf_entries( - ctx.source_lines, - ctx.source, - chunk, - span, - ctx.tab_replacement, - ctx.normalize_indent, - &ctx.fenced_lines, - ctx.full_display_threshold, - ctx.preview_head_lines, - ctx.preview_tail_lines, - ) { - match entry { - LeafEntry::Line { abs_line, text } => { - let head = line_is_in_head(ctx.source, chunk, abs_line); - push_code(ctx, abs_line, &text, head); - }, - LeafEntry::Ellipsis { start_abs, end_abs, .. } => { - push_truncation_marker(ctx, chunk, start_abs, end_abs); - }, - } - } -} - -fn chunk_depth_dashes(chunk: &ChunkNode) -> String { - let depth = chunk.path.chars().filter(|&c| c == '.').count(); - "-".repeat(depth) -} - -fn render_open_anchor_line(ctx: &RenderCtx<'_>, chunk: &ChunkNode) -> String { - let style = ctx.anchor_style.with_omit_checksum(ctx.omit_checksum); - let anchor_label = chunk_anchor_label(chunk, style); - let dashes = chunk_depth_dashes(chunk); - style.render(&dashes, anchor_label.as_str(), chunk.checksum.as_str()) -} - -fn emit_inline_hunks_for(ctx: &mut RenderCtx<'_>, chunk_path: &str) { - let lines = match ctx.inline_hunks.get(chunk_path) { - Some(hunks) => hunks - .iter() - .flat_map(|hunk| hunk.lines.iter().cloned()) - .collect::>(), - None => return, - }; - for line in lines { - push_meta_marked(ctx, line.text, line.marker); - } -} - -struct ChunkSubtreeOptions { - is_first_top_level: bool, - between_top_level_definitions: bool, -} - -fn emit_chunk_subtree( - ctx: &mut RenderCtx<'_>, - chunk: &ChunkNode, - depth: usize, - options: ChunkSubtreeOptions, -) { - // Focus mode gate: skip unfocused chunks, collapse siblings, pass through - // containers and expanded. - if let Some(focus_map) = ctx.focus.as_ref() - && !chunk.path.is_empty() - { - match focus_map.get(chunk.path.as_str()) { - None => return, - Some(ChunkFocusMode::Collapsed) => { - if options.between_top_level_definitions && depth == 0 && !options.is_first_top_level { - push_blank_meta(ctx); - } - push_anchor( - ctx, - render_open_anchor_line(ctx, chunk), - ctx.changed_anchor_paths - .contains(chunk.path.as_str()) - .then_some('*'), - ); - return; - }, - Some(ChunkFocusMode::Container | ChunkFocusMode::Expanded) => { - // fall through to normal rendering - }, - } - } - - let focus_ref = ctx.focus.as_ref(); - let children = - visible_children_for_chunk(ctx.tree, chunk, ctx.lookup, ctx.visible_range, focus_ref); - let span = intersect_visible_span(chunk, ctx.visible_range); - let has_kids = !children.is_empty(); - - if !chunk.path.is_empty() && span.is_none() && !has_kids { - return; - } - if options.between_top_level_definitions && depth == 0 && !options.is_first_top_level { - push_blank_meta(ctx); - } - if !chunk.path.is_empty() { - push_anchor( - ctx, - render_open_anchor_line(ctx, chunk), - ctx.changed_anchor_paths - .contains(chunk.path.as_str()) - .then_some('*'), - ); - } - - if !has_kids { - if !chunk.path.is_empty() && ctx.inline_hunks.contains_key(chunk.path.as_str()) { - emit_inline_hunks_for(ctx, &chunk.path); - return; - } - if ctx.show_leaf_preview - && let Some(span) = span - { - emit_leaf_body(ctx, chunk, span); - } - return; - } - - let is_container = ctx - .focus - .as_ref() - .and_then(|f| f.get(chunk.path.as_str())) - .copied() - == Some(ChunkFocusMode::Container); - - if let Some(span) = span { - emit_container_clip_above(ctx, chunk, &span, &children); - let mut cursor = chunk.start_line; - for child in &children { - let gap_end = child.start_line.saturating_sub(1); - if gap_end >= cursor && !is_container { - for line in cursor..=gap_end { - if line_in_file_scope(line, ctx.visible_range) - && should_render_gap_line(ctx.tree, chunk, ctx.lookup, line) - { - emit_line_gap(ctx, line, line, chunk); - } - } - } - emit_chunk_subtree(ctx, child, depth + 1, ChunkSubtreeOptions { - is_first_top_level: false, - between_top_level_definitions: false, - }); - cursor = cursor.max(child.end_line.saturating_add(1)); - } - if cursor <= span.end && !is_container { - emit_line_gap(ctx, cursor, span.end, chunk); - } - emit_container_clip_below(ctx, chunk, &span, &children); - if !chunk.path.is_empty() { - emit_inline_hunks_for(ctx, &chunk.path); - } - return; - } - for (index, child) in children.iter().enumerate() { - emit_chunk_subtree(ctx, child, depth + 1, ChunkSubtreeOptions { - is_first_top_level: index == 0, - between_top_level_definitions: true, - }); - } -} - -fn compute_num_width( - tree: &ChunkTree, - chunk: &ChunkNode, - lookup: &ChunkLookup<'_>, - visible_range: Option<&VisibleLineRange>, - render_children_only: bool, - show_leaf_preview: bool, - source: &str, - source_lines: &[&str], - tab_replacement: &str, - normalize_indent: Option<(char, usize)>, - fenced_lines: &HashSet, - full_display_threshold: usize, - preview_head_lines: usize, - preview_tail_lines: usize, -) -> usize { - if visible_range.is_none() { - if render_children_only { - return tree.line_count.to_string().len().max(1); - } - let children = visible_children_for_chunk(tree, chunk, lookup, visible_range, None); - let has_out_of_span_child = children - .iter() - .any(|child| child.start_line < chunk.start_line || child.end_line > chunk.end_line); - if !has_out_of_span_child { - return chunk.end_line.to_string().len().max(1); - } - } - let mut max_line = 1usize; - for_each_rendered_source_line( - tree, - chunk, - lookup, - visible_range, - show_leaf_preview, - source, - source_lines, - tab_replacement, - normalize_indent, - fenced_lines, - full_display_threshold, - preview_head_lines, - preview_tail_lines, - &mut |line| { - max_line = max_line.max(line as usize); - }, - ); - max_line.to_string().len().max(1) -} - -/// Find the deepest visible chunk that should own a diff hunk for inline -/// display. -pub fn find_hunk_owner_chunk<'a>( - tree: &'a ChunkTree, - _lookup: &ChunkLookup<'a>, - line: u32, -) -> Option<&'a str> { - line_to_containing_chunk(tree, line).map(|chunk| chunk.path.as_str()) -} - -pub fn hunk_indent_for_chunk( - lookup: &ChunkLookup<'_>, - chunk_path: &str, - source: &str, - tab_replacement: &str, - normalize_indent: Option<(char, usize)>, -) -> String { - let source_lines: Vec<&str> = source.split('\n').collect(); - let Some(chunk) = lookup.get(chunk_path) else { - return String::new(); - }; - let base = chunk_body_anchor_indent(&source_lines, chunk, tab_replacement, normalize_indent); - match normalize_indent { - Some(_) => format!("{base}\t"), - None => format!("{base}{tab_replacement}"), - } -} diff --git a/crates/pi-natives/src/chunk/resolve.rs b/crates/pi-natives/src/chunk/resolve.rs deleted file mode 100644 index 172cd7c56..000000000 --- a/crates/pi-natives/src/chunk/resolve.rs +++ /dev/null @@ -1,1257 +0,0 @@ -use std::{cmp::Ordering, collections::BTreeSet}; - -use crate::chunk::{ - CHUNK_BIGRAMS, - state::ChunkStateInner, - types::{ChunkNode, ChunkRegion, ChunkTree}, -}; - -pub struct ResolvedChunk<'a> { - pub chunk: &'a ChunkNode, - pub crc: Option, -} - -fn parse_region_name(value: &str) -> Option { - match value.trim() { - "^" => Some(ChunkRegion::Head), - "~" => Some(ChunkRegion::Body), - _ => None, - } -} - -fn is_known_region_name(value: &str) -> bool { - matches!(value.trim(), "^" | "~") -} - -pub fn split_region_suffix(selector: &str) -> (&str, bool, Option) { - let Some(suffix) = selector.chars().last() else { - return (selector, false, None); - }; - let suffix = suffix.to_string(); - if !is_known_region_name(&suffix) { - return (selector, false, None); - } - let prefix = &selector[..selector.len() - suffix.len()]; - (prefix.trim_end(), true, parse_region_name(&suffix)) -} - -pub struct ParsedSelector { - pub selector: Option, - pub crc: Option, - pub region: Option, - /// All checksum tokens discovered in the selector (trailing and - /// per-segment), in the order they appeared. Used for lenient multi-CRC - /// matching where at least one CRC in the resolved ancestor chain must - /// match. - pub all_crcs: Vec, - /// `true` when the resolved primary CRC targets the trailing (leaf) - /// segment — either because the last path segment had `#XXXX` or the - /// caller passed `crc` explicitly. `false` when CRCs only appeared on - /// intermediate ancestor segments. - pub has_trailing_crc: bool, -} - -pub fn split_selector_crc_and_region( - selector: Option<&str>, - crc: Option<&str>, - region: Option, -) -> Result { - let mut raw = selector - .map(str::trim) - .filter(|value| !matches!(*value, "" | "null" | "undefined")) - .unwrap_or_default() - .to_owned(); - if let Some(index) = chunk_read_path_separator_index(&raw) { - raw = raw[index + 1..].to_owned(); - } - - let (without_region, parsed_region) = if raw.is_empty() { - (raw.as_str(), None) - } else { - let (prefix, found, parsed_region) = split_region_suffix(raw.as_str()); - if found { - (prefix, parsed_region) - } else if let Some((_, suffix)) = raw.rsplit_once('@') { - return Err(format!( - "Unknown chunk region \"{}\". Valid regions: ^, ~ (or omit for the full chunk).", - suffix.trim() - )); - } else { - (raw.as_str(), None) - } - }; - - let mut selector_part = without_region.trim().to_owned(); - let mut collected_crcs: Vec = Vec::new(); - let mut has_trailing_crc = false; - - // Whole-selector bare-CRC forms ("#XXXX" or "XXXX") take precedence so - // the rest of the logic can treat the selector as a dotted name path. - if let Some(suffix) = selector_part.strip_prefix('#') - && is_checksum_token(suffix.trim()) - { - if let Some(c) = sanitize_crc(Some(suffix)) { - collected_crcs.push(c); - } - selector_part.clear(); - has_trailing_crc = true; - } else if is_checksum_token(selector_part.as_str()) { - if let Some(c) = sanitize_crc(Some(selector_part.as_str())) { - collected_crcs.push(c); - } - selector_part.clear(); - has_trailing_crc = true; - } else if !selector_part.is_empty() { - // Strip per-segment `#XXXX` checksums while traversing. Record whether - // the trailing (last) segment carried its own CRC so callers can keep - // legacy strict-matching behavior for `name#CRC` forms. - let segments: Vec<&str> = selector_part.split('.').collect(); - let last_idx = segments.len().saturating_sub(1); - let cleaned_segments: Vec = segments - .iter() - .enumerate() - .map(|(idx, seg)| { - let seg = seg.trim(); - if let Some((prefix, suffix)) = seg.rsplit_once('#') - && is_checksum_token(suffix.trim()) - { - if let Some(c) = sanitize_crc(Some(suffix)) { - collected_crcs.push(c); - } - if idx == last_idx { - has_trailing_crc = true; - } - prefix.trim_end().to_owned() - } else { - seg.to_owned() - } - }) - .collect(); - selector_part = cleaned_segments.join("."); - } - - let cleaned_selector = if selector_part.is_empty() { - None - } else { - Some(selector_part) - }; - let explicit_crc = sanitize_crc(crc); - if let Some(explicit) = explicit_crc.as_deref() { - if !collected_crcs.iter().any(|c| c == explicit) { - collected_crcs.push(explicit.to_owned()); - } - // An explicit `crc` parameter always targets the resolved chunk (the - // leaf), so it counts as a trailing CRC for legacy matching purposes. - has_trailing_crc = true; - } - let cleaned_crc = explicit_crc.or_else(|| { - if has_trailing_crc { - collected_crcs.last().cloned() - } else { - None - } - }); - let region = region.or(parsed_region); - - if let Some(cleaned_selector) = cleaned_selector.as_deref() - && cleaned_crc.is_some() - && looks_like_file_target(cleaned_selector) - { - return Ok(ParsedSelector { - selector: None, - crc: cleaned_crc, - region, - all_crcs: collected_crcs, - has_trailing_crc, - }); - } - - Ok(ParsedSelector { - selector: cleaned_selector, - crc: cleaned_crc, - region, - all_crcs: collected_crcs, - has_trailing_crc, - }) -} - -pub fn sanitize_chunk_selector(selector: Option<&str>) -> Option { - split_selector_crc_and_region(selector, None, None) - .ok() - .and_then(|parsed| parsed.selector) -} - -pub fn sanitize_crc(crc: Option<&str>) -> Option { - let value = crc?.trim(); - if matches!(value, "" | "null" | "undefined") { - None - } else { - Some(value.to_ascii_lowercase()) - } -} - -pub fn resolve_chunk_selector<'a>( - state: &'a ChunkStateInner, - selector: Option<&str>, - warnings: &mut Vec, -) -> Result<&'a ChunkNode, String> { - let ParsedSelector { crc: cleaned_crc, .. } = - split_selector_crc_and_region(selector, None, None)?; - let cleaned_selector = sanitize_chunk_selector(selector); - resolve_chunk_selector_impl(state, cleaned_selector.as_deref(), cleaned_crc.as_deref(), warnings) -} - -pub fn resolve_chunk_selector_with_crc_filter<'a>( - state: &'a ChunkStateInner, - selector: Option<&str>, - crc: Option<&str>, - warnings: &mut Vec, -) -> Result<&'a ChunkNode, String> { - let ParsedSelector { selector: cleaned_selector, .. } = - split_selector_crc_and_region(selector, None, None)?; - let cleaned_crc = sanitize_crc(crc); - if cleaned_selector.is_none() - && let Some(cleaned_crc) = cleaned_crc.as_deref() - { - return resolve_chunk_by_checksum(state, cleaned_crc); - } - resolve_chunk_selector_impl(state, cleaned_selector.as_deref(), cleaned_crc.as_deref(), warnings) -} - -pub fn resolve_chunk_with_crc<'a>( - state: &'a ChunkStateInner, - selector: Option<&str>, - crc: Option<&str>, - warnings: &mut Vec, -) -> Result, String> { - let ParsedSelector { selector: cleaned_selector, crc: cleaned_crc, .. } = - split_selector_crc_and_region(selector, crc, None)?; - - if cleaned_selector.is_none() - && let Some(cleaned_crc) = cleaned_crc.clone() - { - let chunk = resolve_chunk_by_checksum(state, &cleaned_crc)?; - return Ok(ResolvedChunk { chunk, crc: Some(cleaned_crc) }); - } - - if let (Some(cleaned_selector), Some(cleaned_crc)) = - (cleaned_selector.as_deref(), cleaned_crc.as_deref()) - && state.chunk(cleaned_selector).is_none() - && let Some(chunk) = - resolve_same_parent_crc_fallback(state, cleaned_selector, cleaned_crc, warnings)? - { - return Ok(ResolvedChunk { chunk, crc: Some(cleaned_crc.to_owned()) }); - } - - let chunk = resolve_chunk_selector_impl(state, cleaned_selector.as_deref(), None, warnings)?; - Ok(ResolvedChunk { chunk, crc: cleaned_crc }) -} - -/// Return `Ok(Some(matching_crc))` when at least one of `provided_crcs` -/// equals the checksum of `chunk` or any of its ancestors (including the -/// root). Return `Ok(None)` when `provided_crcs` is empty (nothing to -/// verify). Return an error listing the fresh ancestor CRCs when -/// `provided_crcs` is non-empty but none match. -pub fn verify_any_ancestor_crc_match( - state: &ChunkStateInner, - chunk: &ChunkNode, - provided_crcs: &[String], -) -> Result, String> { - if provided_crcs.is_empty() { - return Ok(None); - } - let mut ancestors: Vec<&ChunkNode> = Vec::new(); - let mut cursor: Option<&ChunkNode> = Some(chunk); - while let Some(node) = cursor { - ancestors.push(node); - cursor = match node.parent_path.as_deref() { - Some(parent_path) => state.chunk(parent_path), - None => None, - }; - } - for ancestor in &ancestors { - for crc in provided_crcs { - if ancestor.checksum.as_str() == crc.as_str() { - return Ok(Some(crc.clone())); - } - } - } - let fresh = ancestors - .iter() - .map(|node| format_node_ref(node)) - .collect::>() - .join(", "); - Err(format!( - "None of the provided checksums [{}] match any ancestor of \"{}\". Fresh chain: {fresh}. \ - Re-read the file to get current checksums.", - provided_crcs.join(", "), - if chunk.path.is_empty() { - "" - } else { - chunk.path.as_str() - }, - )) -} - -fn resolve_same_parent_crc_fallback<'a>( - state: &'a ChunkStateInner, - selector: &str, - crc: &str, - warnings: &mut Vec, -) -> Result, String> { - let (parent_path, requested_leaf) = split_parent_selector(selector); - let child_paths = match parent_path { - Some(parent_path) => { - let Some(parent_chunk) = state.chunk(parent_path) else { - return Ok(None); - }; - parent_chunk.children.as_slice() - }, - None => state.tree().root_children.as_slice(), - }; - let matches = collect_unique_matches( - child_paths - .iter() - .filter_map(|child_path| state.chunk(child_path)) - .filter(|chunk| chunk.checksum == crc), - ); - if matches.is_empty() { - return Ok(None); - } - - let resolved = if matches.len() == 1 { - matches[0] - } else if let Some(candidate) = choose_named_crc_match(matches.as_slice(), requested_leaf) { - candidate - } else { - let scope = parent_path.unwrap_or(""); - return Err(format!( - "Ambiguous stale selector \"{selector}#{crc}\" under \"{scope}\" matches {} siblings: \ - {}. Re-read the file to get the current selector.", - matches.len(), - matches - .iter() - .map(|chunk| format_node_ref(chunk)) - .collect::>() - .join(", "), - )); - }; - - warnings.push(format!( - "Auto-resolved stale selector \"{selector}#{crc}\" to sibling \"{}\". Use the fresh \ - selector from read output.", - format_node_ref(resolved) - )); - Ok(Some(resolved)) -} - -fn split_parent_selector(selector: &str) -> (Option<&str>, &str) { - match selector.rsplit_once('.') { - Some((parent_path, leaf)) if !parent_path.is_empty() => (Some(parent_path), leaf), - _ => (None, selector), - } -} - -fn choose_named_crc_match<'a>( - matches: &[&'a ChunkNode], - requested_leaf: &str, -) -> Option<&'a ChunkNode> { - let mut best_match = None; - let mut best_score = 0; - let mut tied = false; - - for candidate in matches { - let candidate_leaf = candidate - .path - .rsplit('.') - .next() - .unwrap_or(candidate.path.as_str()); - let score = leaf_name_similarity(requested_leaf, candidate_leaf); - if score > best_score { - best_match = Some(*candidate); - best_score = score; - tied = false; - } else if score > 0 && score == best_score { - tied = true; - } - } - - if tied || best_score == 0 { - None - } else { - best_match - } -} - -fn leaf_name_similarity(requested_leaf: &str, candidate_leaf: &str) -> usize { - if candidate_leaf == requested_leaf { - return 4; - } - if normalize_leaf_name(candidate_leaf) == normalize_leaf_name(requested_leaf) { - return 3; - } - if candidate_leaf.contains(requested_leaf) || requested_leaf.contains(candidate_leaf) { - return 2; - } - if chunk_path_similarity(requested_leaf, candidate_leaf) > 0.5 { - return 1; - } - 0 -} - -fn normalize_leaf_name(name: &str) -> &str { - match name.rsplit_once('_') { - Some((prefix, suffix)) - if !prefix.is_empty() && suffix.chars().all(|ch| ch.is_ascii_digit()) => - { - prefix - }, - _ => name, - } -} - -pub fn resolve_chunk_by_checksum<'a>( - state: &'a ChunkStateInner, - crc: &str, -) -> Result<&'a ChunkNode, String> { - let cleaned_crc = sanitize_crc(Some(crc)).ok_or_else(|| "Checksum is required".to_owned())?; - let matches = state.chunks_by_checksum(&cleaned_crc); - match matches.len() { - 0 => Err(format!( - "Checksum \"{cleaned_crc}\" did not match any chunk. Re-read the file to get current \ - checksums." - )), - 1 => Ok(matches[0]), - _ => Err(format!( - "Ambiguous checksum \"{cleaned_crc}\" matches {} chunks: {}. Provide sel to disambiguate.", - matches.len(), - matches - .iter() - .map(|chunk| format_node_ref(chunk)) - .collect::>() - .join(", "), - )), - } -} - -fn root_chunk(state: &ChunkStateInner) -> Result<&ChunkNode, String> { - state - .chunk("") - .ok_or_else(|| "Chunk tree is missing the root chunk".to_owned()) -} - -pub fn chunk_region_range(chunk: &ChunkNode, region: ChunkRegion) -> (usize, usize) { - let start = chunk.start_byte as usize; - let end = chunk.end_byte as usize; - let pro_end = chunk - .prologue_end_byte - .map_or(start, |b| (b as usize).clamp(start, end)); - let epi_start = chunk - .epilogue_start_byte - .map_or(end, |b| (b as usize).clamp(pro_end, end)); - match region { - ChunkRegion::Head => (start, pro_end), - ChunkRegion::Body => (pro_end, epi_start), - } -} - -pub fn format_region_ref(chunk: &ChunkNode, region: Option) -> String { - let suffix = region.map_or(String::new(), |r| r.as_str().to_owned()); - if chunk.path.is_empty() { - format!("#{}{suffix}", chunk.checksum) - } else { - format!("{}#{}{suffix}", chunk.path, chunk.checksum) - } -} - -fn resolve_chunk_selector_impl<'a>( - state: &'a ChunkStateInner, - selector: Option<&str>, - crc: Option<&str>, - warnings: &mut Vec, -) -> Result<&'a ChunkNode, String> { - let Some(cleaned) = selector else { - return root_chunk(state); - }; - - if is_line_number_selector(cleaned) { - if let Some(line) = parse_line_number(cleaned) - && let Some(chunk_path) = crate::chunk::line_to_chunk_path(state.tree(), line) - { - warnings - .push(format!("Auto-resolved line target \"{cleaned}\" to chunk \"{chunk_path}\".")); - return resolve_chunk_selector_impl(state, Some(&chunk_path), crc, warnings); - } - return Err(format!( - "Line target \"{cleaned}\" does not fall inside any chunk. Use chunk paths like \ - fn_foo#thth instead, or run read(sel=\"?\") to list available chunks." - )); - } - - if let Some(chunk) = state.chunk(cleaned) { - return match_crc_filter(cleaned, vec![chunk], crc); - } - - if is_checksum_token(cleaned) { - let matches = state.chunks_by_checksum(cleaned); - if !matches.is_empty() { - return resolve_matches( - matches, - cleaned, - crc, - warnings, - "checksum selector", - "Auto-resolved checksum selector", - ); - } - } - - let suffix_matches = state.chunks_by_suffix(cleaned); - if !suffix_matches.is_empty() { - return resolve_matches( - suffix_matches, - cleaned, - crc, - warnings, - "chunk selector", - "Auto-resolved chunk selector", - ); - } - - if !cleaned.contains('.') { - let prefixed_matches = collect_unique_matches(state.tree.chunks.iter().filter(|chunk| { - if chunk.path.is_empty() { - return false; - } - let leaf = chunk.path.rsplit('.').next().unwrap_or(chunk.path.as_str()); - let expected = chunk.kind.path_segment(Some(cleaned)); - path_segment_matches_requested(leaf, &expected) - || path_segment_matches_requested(leaf, cleaned) - })); - if !prefixed_matches.is_empty() { - return resolve_matches( - prefixed_matches, - cleaned, - crc, - warnings, - "chunk selector", - "Auto-resolved chunk selector", - ); - } - } - - let kind_segments = cleaned.split('.').collect::>(); - let kind_candidates = collect_unique_matches( - state - .chunks_by_leaf(kind_segments.last().copied().unwrap_or(cleaned)) - .into_iter() - .filter(|candidate| kind_path_matches(candidate, &kind_segments)), - ); - let full_path_candidates = if kind_candidates.is_empty() { - collect_unique_matches( - state - .tree - .chunks - .iter() - .filter(|candidate| kind_path_matches(candidate, &kind_segments)), - ) - } else { - Vec::new() - }; - if !kind_candidates.is_empty() { - return resolve_matches( - kind_candidates, - cleaned, - crc, - warnings, - "kind selector", - "Auto-resolved kind selector", - ); - } - if !full_path_candidates.is_empty() { - return resolve_matches( - full_path_candidates, - cleaned, - crc, - warnings, - "full-path selector", - "Auto-resolved full-path selector", - ); - } - - Err(build_not_found_error(state.tree(), cleaned)) -} - -fn match_crc_filter<'a>( - cleaned: &str, - matches: Vec<&'a ChunkNode>, - crc: Option<&str>, -) -> Result<&'a ChunkNode, String> { - let Some(cleaned_crc) = crc else { - return Ok(matches[0]); - }; - let filtered = filter_by_crc(&matches, cleaned_crc); - match filtered.len() { - 1 => Ok(filtered[0]), - 0 => { - let actual = matches - .iter() - .map(|chunk| format_node_ref(chunk)) - .collect::>() - .join(", "); - Err(format!("Stale checksum \"{cleaned_crc}\" for \"{cleaned}\". Current: {actual}.")) - }, - _ => Err(format!( - "Ambiguous chunk selector \"{cleaned}\" with checksum \"{cleaned_crc}\" matches {} \ - chunks: {}. Use the full path from read output.", - filtered.len(), - filtered - .iter() - .map(|chunk| format_node_ref(chunk)) - .collect::>() - .join(", "), - )), - } -} - -fn resolve_matches<'a>( - matches: Vec<&'a ChunkNode>, - cleaned: &str, - crc: Option<&str>, - warnings: &mut Vec, - selector_label: &str, - warning_label: &str, -) -> Result<&'a ChunkNode, String> { - let matches = if let Some(cleaned_crc) = crc { - let filtered = filter_by_crc(&matches, cleaned_crc); - if filtered.is_empty() { - let actual = matches - .iter() - .map(|chunk| format_node_ref(chunk)) - .collect::>() - .join(", "); - return Err(format!( - "Stale checksum \"{cleaned_crc}\" for {selector_label} \"{cleaned}\". Current: \ - {actual}." - )); - } - filtered - } else { - matches - }; - let outermost = retain_outermost_matches(matches); - resolve_unique_chunks(outermost, cleaned, warnings, selector_label, warning_label)?.ok_or_else( - || { - format!( - "{selector_label} \"{cleaned}\" did not match any chunk. Re-read the file to see \ - available chunk paths." - ) - }, - ) -} - -fn filter_by_crc<'a>(matches: &[&'a ChunkNode], crc: &str) -> Vec<&'a ChunkNode> { - matches - .iter() - .filter(|chunk| chunk.checksum == crc) - .copied() - .collect() -} - -fn collect_unique_matches<'a>( - matches: impl IntoIterator, -) -> Vec<&'a ChunkNode> { - let mut seen = BTreeSet::new(); - let mut out = Vec::new(); - for chunk in matches { - if chunk.path.is_empty() || !seen.insert(chunk.path.as_str()) { - continue; - } - out.push(chunk); - } - out -} - -fn retain_outermost_matches(matches: Vec<&ChunkNode>) -> Vec<&ChunkNode> { - let Some(min_depth) = matches - .iter() - .map(|chunk| chunk.path.split('.').count()) - .min() - else { - return matches; - }; - matches - .into_iter() - .filter(|chunk| chunk.path.split('.').count() == min_depth) - .collect() -} - -fn resolve_unique_chunks<'a>( - matches: Vec<&'a ChunkNode>, - cleaned: &str, - warnings: &mut Vec, - selector_label: &str, - warning_label: &str, -) -> Result, String> { - match matches.len() { - 0 => Ok(None), - 1 => { - warnings.push(format!( - "{warning_label} \"{cleaned}\" to \"{}\". Use the full path from read output.", - format_node_ref(matches[0]) - )); - Ok(Some(matches[0])) - }, - _ => Err(format!( - "Ambiguous {selector_label} \"{cleaned}\" matches {} chunks: {}. Use the full path from \ - read output.", - matches.len(), - matches - .iter() - .map(|chunk| format_node_ref(chunk)) - .collect::>() - .join(", "), - )), - } -} - -fn kind_path_matches(candidate: &ChunkNode, kind_segments: &[&str]) -> bool { - if candidate.path.is_empty() { - return false; - } - let path_segments = candidate.path.split('.').collect::>(); - path_segments.len() == kind_segments.len() - && kind_segments - .iter() - .zip(path_segments) - .all(|(requested, candidate_segment)| { - path_segment_matches_requested(candidate_segment, requested) - }) -} - -fn path_segment_matches_requested(candidate_segment: &str, requested_segment: &str) -> bool { - if candidate_segment == requested_segment { - return true; - } - - let normalized_candidate = normalize_leaf_name(candidate_segment); - if normalized_candidate == requested_segment { - return true; - } - - let (candidate_kind, candidate_identifier) = split_path_segment(normalized_candidate); - let (requested_kind, requested_identifier) = split_path_segment(requested_segment); - match (candidate_identifier, requested_identifier) { - (Some(candidate_identifier), Some(requested_identifier)) => { - candidate_kind == requested_kind && requested_identifier.starts_with(candidate_identifier) - }, - (Some(candidate_identifier), None) => { - requested_segment == candidate_identifier - || requested_segment.starts_with(candidate_identifier) - }, - (None, Some(_)) => false, - (None, None) => false, - } -} - -fn split_path_segment(segment: &str) -> (&str, Option<&str>) { - match segment.split_once('_') { - Some((kind, identifier)) if !identifier.is_empty() => (kind, Some(identifier)), - _ => (segment, None), - } -} - -/// Format a `ChunkNode` as `path#CRC`. -fn format_node_ref(chunk: &ChunkNode) -> String { - if chunk.path.is_empty() { - format!("#{}", chunk.checksum) - } else { - format!("{}#{}", chunk.path, chunk.checksum) - } -} - -pub fn format_selector_tree( - tree: &ChunkTree, - children: &[String], - leading_dot_on_first_level: bool, -) -> Vec { - fn emit_children( - tree: &ChunkTree, - children: &[String], - prefix: &str, - depth: usize, - leading_dot_on_first_level: bool, - lines: &mut Vec, - ) { - let count = children.len(); - for (index, child_path) in children.iter().enumerate() { - let Some(child) = find_chunk_by_path(tree, child_path) else { - continue; - }; - let is_last = index + 1 == count; - let connector = if is_last { "└── " } else { "├── " }; - let leaf = child.path.rsplit('.').next().unwrap_or(child.path.as_str()); - let dot = if depth > 0 || leading_dot_on_first_level { - "." - } else { - "" - }; - lines.push(format!( - "{prefix}{connector}{dot}{leaf}#{} L{}-L{}", - child.checksum, child.start_line, child.end_line, - )); - let continuation = if is_last { " " } else { "│ " }; - if let Some(signature) = child.signature.as_deref() { - lines.push(format!("{prefix}{continuation} {signature}")); - } - if !child.children.is_empty() { - let next_prefix = format!("{prefix}{continuation}"); - emit_children(tree, &child.children, next_prefix.as_str(), depth + 1, false, lines); - } - } - } - - let mut lines = Vec::new(); - emit_children(tree, children, "", 0, leading_dot_on_first_level, &mut lines); - lines -} - -fn build_not_found_error(tree: &ChunkTree, cleaned: &str) -> String { - let (direct_children_parent, direct_children, matched_empty_prefix) = - matching_prefix_context(tree, cleaned); - let similarity = suggest_chunk_paths(tree, cleaned, 8); - - let hint = if let Some(parent) = direct_children_parent { - let tree_lines = format_selector_tree(tree, &direct_children, true); - if tree_lines.is_empty() { - format!(" Direct children of \"{parent}\": none.") - } else { - format!(" Direct children of \"{parent}\":\n{}", tree_lines.join("\n")) - } - } else if let Some(prefix) = matched_empty_prefix { - if similarity.is_empty() { - format!(" The prefix \"{prefix}\" exists but has no child chunks.") - } else { - format!( - " The prefix \"{prefix}\" exists but has no child chunks. Similar paths: {}.", - similarity.join(", ") - ) - } - } else if !similarity.is_empty() { - format!(" Similar paths: {}.", similarity.join(", ")) - } else { - let tree_lines = format_selector_tree(tree, &tree.root_children, false); - if tree_lines.is_empty() { - " Use sel=\"?\" to see available chunk paths.".to_owned() - } else { - format!(" Available top-level chunks:\n{}", tree_lines.join("\n")) - } - }; - - if hint.contains('\n') { - format!( - "Chunk path not found: \"{cleaned}\".{hint}\nUse sel=\"?\" if you need the full chunk \ - tree with paths and checksums." - ) - } else { - format!( - "Chunk path not found: \"{cleaned}\".{hint} Use sel=\"?\" if you need the full chunk \ - tree with paths and checksums." - ) - } -} - -fn matching_prefix_context( - tree: &ChunkTree, - cleaned: &str, -) -> (Option, Vec, Option) { - let mut direct_children = None; - let mut direct_children_parent = None; - let mut matched_empty_prefix = None; - - if cleaned.contains('.') { - let parts = cleaned.split('.').collect::>(); - for index in (1..parts.len()).rev() { - let prefix = parts[..index].join("."); - let Some(parent) = find_chunk_by_path(tree, &prefix) else { - continue; - }; - if !parent.children.is_empty() { - let mut children = parent.children.clone(); - children.sort(); - direct_children_parent = Some(prefix); - direct_children = Some(children); - break; - } - if matched_empty_prefix.is_none() { - matched_empty_prefix = Some(prefix); - } - } - } - - (direct_children_parent, direct_children.unwrap_or_default(), matched_empty_prefix) -} - -fn suggest_chunk_paths(tree: &ChunkTree, query: &str, limit: usize) -> Vec { - let mut scored = tree - .chunks - .iter() - .filter(|chunk| !chunk.path.is_empty()) - .map(|chunk| (&chunk.path, &chunk.checksum, chunk_path_similarity(query, &chunk.path))) - .filter(|(_, _, score)| *score > 0.1) - .collect::>(); - scored.sort_by(|left, right| { - right - .2 - .partial_cmp(&left.2) - .unwrap_or(Ordering::Equal) - .then_with(|| left.0.cmp(right.0)) - }); - scored - .into_iter() - .take(limit) - .map(|(path, checksum, _)| format!("{path}#{checksum}")) - .collect() -} - -fn chunk_path_similarity(query: &str, candidate: &str) -> f64 { - if candidate.ends_with(query) || candidate.ends_with(&format!(".{query}")) { - return 0.9; - } - - let query_leaf = query.rsplit('.').next().unwrap_or(query); - let candidate_leaf = candidate.rsplit('.').next().unwrap_or(candidate); - if query_leaf == candidate_leaf { - return 0.85; - } - - if candidate.contains(query) || query.contains(candidate) { - return 0.6; - } - - let query_parts = query.split('.').collect::>(); - let overlap = candidate - .split('.') - .filter(|part| query_parts.contains(part)) - .count(); - if overlap > 0 { - 0.1f64.mul_add(overlap as f64, 0.3) - } else { - 0.0 - } -} - -fn looks_like_file_target(selector: &str) -> bool { - if selector.contains('/') || selector.contains('\\') { - return true; - } - - let Some((base, ext)) = selector.rsplit_once('.') else { - return false; - }; - !base.is_empty() && !ext.is_empty() && ext.chars().all(|ch| ch.is_ascii_alphanumeric()) -} - -fn is_line_number_selector(selector: &str) -> bool { - let Some(rest) = selector.strip_prefix('L') else { - return false; - }; - let Some((start, end)) = rest.split_once('-') else { - return !rest.is_empty() && rest.chars().all(|ch| ch.is_ascii_digit()); - }; - if start.is_empty() || !start.chars().all(|ch| ch.is_ascii_digit()) { - return false; - } - let end = end.strip_prefix('L').unwrap_or(end); - !end.is_empty() && end.chars().all(|ch| ch.is_ascii_digit()) -} - -/// Extract the start line number from a line selector like `L89` or `L24-L27`. -fn parse_line_number(selector: &str) -> Option { - let rest = selector.strip_prefix('L')?; - let digits = rest.split('-').next().unwrap_or(rest); - digits.parse::().ok() -} - -fn is_checksum_token(value: &str) -> bool { - if value.len() != 4 || !value.is_ascii() { - return false; - } - let lower = value.to_ascii_lowercase(); - CHUNK_BIGRAMS.contains(&&lower[..2]) && CHUNK_BIGRAMS.contains(&&lower[2..4]) -} - -fn find_chunk_by_path<'a>(tree: &'a ChunkTree, path: &str) -> Option<&'a ChunkNode> { - tree.chunks.iter().find(|chunk| chunk.path == path) -} - -/// Find the `:` separating a file path from a chunk selector in -/// `file.ts:chunk_path`. Skips Windows `C:\` / `C:/` drive prefixes. -fn chunk_read_path_separator_index(value: &str) -> Option { - let bytes = value.as_bytes(); - // Skip Windows drive prefix: `C:\` or `C:/` - let start = if bytes.len() >= 3 - && bytes[0].is_ascii_alphabetic() - && bytes[1] == b':' - && matches!(bytes[2], b'/' | b'\\') - { - value[2..].find(':').map(|i| i + 2)? - } else { - value.find(':')? - }; - Some(start) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::chunk::kind::ChunkKind; - - fn chunk( - path: &str, - checksum: &str, - parent_path: Option<&str>, - children: Vec<&str>, - ) -> ChunkNode { - let leaf = path.rsplit('.').next().unwrap_or(path); - let kind = match leaf.split_once('_').map_or(leaf, |(prefix, _)| prefix) { - "fn" => ChunkKind::Function, - "class" => ChunkKind::Class, - "try" => ChunkKind::Try, - _ => ChunkKind::Chunk, - }; - ChunkNode { - path: path.to_owned(), - identifier: leaf - .split_once('_') - .and_then(|(_, identifier)| (!identifier.is_empty()).then_some(identifier.to_owned())), - kind, - leaf: children.is_empty(), - virtual_content: None, - parent_path: parent_path.map(str::to_owned), - children: children.into_iter().map(str::to_owned).collect(), - signature: None, - start_line: 1, - end_line: 1, - line_count: 1, - start_byte: 0, - end_byte: 0, - checksum_start_byte: 0, - prologue_end_byte: None, - epilogue_start_byte: None, - checksum: checksum.to_owned(), - error: false, - indent: 0, - indent_char: " ".to_owned(), - group: false, - } - } - - fn state_for_resolution() -> ChunkStateInner { - ChunkStateInner::new(String::new(), "typescript".to_owned(), ChunkTree { - language: "typescript".to_owned(), - checksum: "ROOT".to_owned(), - line_count: 1, - parse_errors: 0, - parse_error_lines: Vec::new(), - fallback: false, - root_path: String::new(), - root_children: vec!["fn_han".to_owned()], - chunks: vec![ - chunk("", "ROOT", None, vec!["fn_han"]), - chunk("fn_han", "seas", Some(""), vec!["fn_han.try"]), - chunk("fn_han.try", "tete", Some("fn_han"), vec!["fn_han.try.if_2"]), - chunk("fn_han.try.if_2", "roro", Some("fn_han.try"), vec!["fn_han.try.if_2.loop"]), - chunk("fn_han.try.if_2.loop", "coco", Some("fn_han.try.if_2"), vec![ - "fn_han.try.if_2.loop.if_2", - ]), - chunk("fn_han.try.if_2.loop.if_2", "nene", Some("fn_han.try.if_2.loop"), vec![]), - ], - }) - } - - #[test] - fn resolves_requested_chunk_selector_forms() { - let state = state_for_resolution(); - let selectors = [ - "fn_han.try.if_2#roro", - "fn_han.try.if_2", - "handleTerraform.try.if_2", - "if_2", - "if_2#roro", - "#roro", - "roro", - ]; - - for selector in selectors { - let mut warnings = Vec::new(); - let resolved = resolve_chunk_with_crc(&state, Some(selector), None, &mut warnings) - .unwrap_or_else(|err| panic!("selector {selector} should resolve: {err}")); - assert_eq!(resolved.chunk.path, "fn_han.try.if_2"); - } - } - - #[test] - fn resolves_stale_selector_by_same_parent_checksum() { - let state = ChunkStateInner::new(String::new(), "typescript".to_owned(), ChunkTree { - language: "typescript".to_owned(), - checksum: "ROOT".to_owned(), - line_count: 1, - parse_errors: 0, - parse_error_lines: Vec::new(), - fallback: false, - root_path: String::new(), - root_children: vec!["fn_run".to_owned()], - chunks: vec![ - chunk("", "ROOT", None, vec!["fn_run"]), - chunk("fn_run", "riri", Some(""), vec!["fn_run.var_eff_1", "fn_run.var_eff_2"]), - chunk("fn_run.var_eff_1", "anan", Some("fn_run"), vec![]), - chunk("fn_run.var_eff_2", "enen", Some("fn_run"), vec![]), - ], - }); - let mut warnings = Vec::new(); - let resolved = - resolve_chunk_with_crc(&state, Some("fn_run.var_eff"), Some("enen"), &mut warnings) - .expect("stale selector should resolve to same-parent checksum match"); - assert_eq!(resolved.chunk.path, "fn_run.var_eff_2"); - assert!( - warnings - .iter() - .any(|warning| warning.contains("Auto-resolved stale selector")) - ); - } - - #[test] - fn stale_selector_prefers_best_leaf_name_when_crc_matches_multiple_siblings() { - let state = ChunkStateInner::new(String::new(), "typescript".to_owned(), ChunkTree { - language: "typescript".to_owned(), - checksum: "ROOT".to_owned(), - line_count: 1, - parse_errors: 0, - parse_error_lines: Vec::new(), - fallback: false, - root_path: String::new(), - root_children: vec!["fn_run".to_owned()], - chunks: vec![ - chunk("", "ROOT", None, vec!["fn_run"]), - chunk("fn_run", "riri", Some(""), vec!["fn_run.var_oth", "fn_run.var_eff_1"]), - chunk("fn_run.var_oth", "enen", Some("fn_run"), vec![]), - chunk("fn_run.var_eff_1", "enen", Some("fn_run"), vec![]), - ], - }); - let mut warnings = Vec::new(); - let resolved = - resolve_chunk_with_crc(&state, Some("fn_run.var_eff"), Some("enen"), &mut warnings) - .expect("best name match should disambiguate same-parent checksum siblings"); - assert_eq!(resolved.chunk.path, "fn_run.var_eff_1"); - } - - #[test] - fn stale_selector_fails_closed_when_same_parent_crc_matches_are_ambiguous() { - let state = ChunkStateInner::new(String::new(), "typescript".to_owned(), ChunkTree { - language: "typescript".to_owned(), - checksum: "ROOT".to_owned(), - line_count: 1, - parse_errors: 0, - parse_error_lines: Vec::new(), - fallback: false, - root_path: String::new(), - root_children: vec!["fn_run".to_owned()], - chunks: vec![ - chunk("", "ROOT", None, vec!["fn_run"]), - chunk("fn_run", "riri", Some(""), vec!["fn_run.var_eff_1", "fn_run.var_eff_2"]), - chunk("fn_run.var_eff_1", "enen", Some("fn_run"), vec![]), - chunk("fn_run.var_eff_2", "enen", Some("fn_run"), vec![]), - ], - }); - let mut warnings = Vec::new(); - let Err(err) = - resolve_chunk_with_crc(&state, Some("fn_run.var_eff"), Some("enen"), &mut warnings) - else { - panic!("ambiguous stale selector should fail closed"); - }; - assert!(err.contains("Ambiguous stale selector"), "{err}"); - } - - #[test] - fn resolves_full_untruncated_identifier_paths() { - let state = ChunkStateInner::new(String::new(), "typescript".to_owned(), ChunkTree { - language: "typescript".to_owned(), - checksum: "ROOT".to_owned(), - line_count: 1, - parse_errors: 0, - parse_error_lines: Vec::new(), - fallback: false, - root_path: String::new(), - root_children: vec!["cls_Ser".to_owned()], - chunks: vec![ - chunk("", "ROOT", None, vec!["cls_Ser"]), - chunk("cls_Ser", "lele", Some(""), vec!["cls_Ser.fn_han"]), - chunk("cls_Ser.fn_han", "eaea", Some("cls_Ser"), vec![]), - ], - }); - let mut warnings = Vec::new(); - let resolved = resolve_chunk_with_crc( - &state, - Some("cls_Server.fn_handleRequest"), - Some("eaea"), - &mut warnings, - ) - .expect("full untruncated selector should resolve to truncated chunk path"); - assert_eq!(resolved.chunk.path, "cls_Ser.fn_han"); - assert!( - warnings - .iter() - .any(|warning| warning.contains("Auto-resolved")) - ); - } - - #[test] - fn split_selector_strips_inline_per_segment_crcs() { - let parsed = - split_selector_crc_and_region(Some("fn_han#seas.try#tete.if_2"), None, None).unwrap(); - assert_eq!(parsed.selector.as_deref(), Some("fn_han.try.if_2")); - assert_eq!(parsed.all_crcs, vec!["seas".to_owned(), "tete".to_owned()]); - assert!(!parsed.has_trailing_crc); - // No trailing CRC → primary crc field stays empty so legacy strict - // matching doesn't fire on an ancestor CRC. - assert!(parsed.crc.is_none()); - } - - #[test] - fn split_selector_retains_trailing_crc_as_primary() { - let parsed = - split_selector_crc_and_region(Some("fn_han#seas.try.if_2#roro"), None, None).unwrap(); - assert_eq!(parsed.selector.as_deref(), Some("fn_han.try.if_2")); - assert_eq!(parsed.all_crcs, vec!["seas".to_owned(), "roro".to_owned()]); - assert_eq!(parsed.crc.as_deref(), Some("roro")); - assert!(parsed.has_trailing_crc); - } - - #[test] - fn verify_any_ancestor_crc_match_accepts_single_fresh() { - let state = state_for_resolution(); - let chunk = state.chunk("fn_han.try.if_2").unwrap(); - let stale_and_fresh = vec!["WRONG".to_owned(), "seas".to_owned()]; - let matched = verify_any_ancestor_crc_match(&state, chunk, &stale_and_fresh).unwrap(); - assert_eq!(matched.as_deref(), Some("seas")); - } - - #[test] - fn verify_any_ancestor_crc_match_rejects_all_stale() { - let state = state_for_resolution(); - let chunk = state.chunk("fn_han.try.if_2").unwrap(); - let all_stale = vec!["WRONG".to_owned(), "BADB".to_owned()]; - let err = verify_any_ancestor_crc_match(&state, chunk, &all_stale).unwrap_err(); - assert!(err.contains("None of the provided checksums"), "{err}"); - assert!(err.contains("seas") || err.contains("tete") || err.contains("roro"), "{err}"); - } - - #[test] - fn verify_any_ancestor_crc_match_no_crcs_is_noop() { - let state = state_for_resolution(); - let chunk = state.chunk("fn_han.try.if_2").unwrap(); - let matched = verify_any_ancestor_crc_match(&state, chunk, &[]).unwrap(); - assert!(matched.is_none()); - } -} diff --git a/crates/pi-natives/src/chunk/schema.rs b/crates/pi-natives/src/chunk/schema.rs deleted file mode 100644 index 65f609ef6..000000000 --- a/crates/pi-natives/src/chunk/schema.rs +++ /dev/null @@ -1,130 +0,0 @@ -use std::{cell::RefCell, collections::HashMap}; - -use serde::Deserialize; - -#[derive(Debug, Deserialize)] -struct GeneratedSchema { - languages: HashMap>, -} - -#[derive(Clone, Debug, Deserialize)] -pub struct NodeTypeSchema { - pub identifier_fields: Vec, - pub body_fields: Vec, - pub promotion_fields: Vec, - pub container_child_kinds: Vec, - pub is_supertype: bool, - pub has_structural_children: bool, -} - -impl NodeTypeSchema { - pub const fn is_structural(&self) -> bool { - self.is_supertype - || !self.identifier_fields.is_empty() - || !self.body_fields.is_empty() - || !self.container_child_kinds.is_empty() - || self.has_structural_children - } -} - -thread_local! { - static CURRENT_LANGUAGE: RefCell> = const { RefCell::new(None) }; -} - -static GENERATED_SCHEMA: std::sync::LazyLock>> = - std::sync::LazyLock::new(|| { - let raw = include_str!(concat!(env!("OUT_DIR"), "/chunk_schema.json")); - let generated: GeneratedSchema = - serde_json::from_str(raw).expect("generated chunk schema should parse"); - generated.languages - }); - -pub struct SchemaLanguageGuard { - previous: Option<&'static str>, -} - -impl Drop for SchemaLanguageGuard { - fn drop(&mut self) { - CURRENT_LANGUAGE.with(|current| { - *current.borrow_mut() = self.previous; - }); - } -} - -pub fn enter_language(language: &'static str) -> SchemaLanguageGuard { - let previous = CURRENT_LANGUAGE.with(|current| current.replace(Some(language))); - SchemaLanguageGuard { previous } -} - -pub fn current_language() -> Option<&'static str> { - CURRENT_LANGUAGE.with(|current| *current.borrow()) -} - -pub fn schema_for(language: &str, kind: &str) -> Option<&'static NodeTypeSchema> { - GENERATED_SCHEMA - .get(language) - .and_then(|schemas| schemas.get(kind)) -} - -pub fn schema_for_current(kind: &str) -> Option<&'static NodeTypeSchema> { - current_language().and_then(|language| schema_for(language, kind)) -} - -#[cfg(test)] -pub fn has_schema(language: &str) -> bool { - GENERATED_SCHEMA.contains_key(language) -} - -#[cfg(test)] -mod tests { - use super::{has_schema, schema_for}; - - #[test] - fn python_function_definition_schema_has_name_and_body() { - let schema = schema_for("python", "function_definition") - .expect("python function_definition schema should exist"); - assert_eq!(schema.identifier_fields, vec!["name".to_string()]); - assert_eq!(schema.body_fields, vec!["body".to_string()]); - assert!(schema.promotion_fields.is_empty()); - assert!(schema.is_structural()); - } - - #[test] - fn nix_let_expression_schema_exposes_binding_set_child() { - let schema = - schema_for("nix", "let_expression").expect("nix let_expression schema should exist"); - assert_eq!(schema.body_fields, vec!["body".to_string()]); - assert!( - schema - .container_child_kinds - .iter() - .any(|kind| kind == "binding_set"), - "let_expression should surface binding_set children" - ); - assert!(schema.has_structural_children); - } - - #[test] - fn wrapper_schemas_preserve_promotable_definition_fields() { - let python = schema_for("python", "decorated_definition") - .expect("python decorated_definition schema should exist"); - assert_eq!(python.promotion_fields, vec!["definition".to_string()]); - - let typescript = schema_for("typescript", "export_statement") - .expect("typescript export_statement schema should exist"); - assert!( - typescript - .promotion_fields - .iter() - .any(|field| field == "declaration"), - "export_statement should preserve declaration field for promotion" - ); - } - - #[test] - fn generated_schema_covers_expected_languages() { - for language in ["python", "nix", "toml", "typescript", "rust", "yaml", "handlebars"] { - assert!(has_schema(language), "{language} should have generated schema data"); - } - } -} diff --git a/crates/pi-natives/src/chunk/shape.rs b/crates/pi-natives/src/chunk/shape.rs deleted file mode 100644 index 8c97c2653..000000000 --- a/crates/pi-natives/src/chunk/shape.rs +++ /dev/null @@ -1,354 +0,0 @@ -use tree_sitter::Node; - -use super::{atom_list, schema}; - -const IDENTIFIER_FIELD_PRIORITY: &[&str] = &[ - "name", - "identifier", - "attrpath", - "key", - "label", - "alias", - "field", - "member", - "property", - "tag", - "target", - "variable", -]; - -const BODY_FIELD_PRIORITY: &[&str] = &["body", "value", "declaration_list", "block", "members"]; - -pub fn signature_end_byte(node: Node<'_>) -> Option { - recurse_target(node) - .filter(|child| child.start_byte() > node.start_byte()) - .map(|child| child.start_byte()) -} - -pub fn identifier_node(node: Node<'_>) -> Option> { - schema_identifier_node(node) - .or_else(|| probe_field_child(node, IDENTIFIER_FIELD_PRIORITY)) - .or_else(|| { - local_named_children(node) - .into_iter() - .find(|child| looks_like_identifier_kind(child.kind())) - }) -} - -pub fn recurse_target(node: Node<'_>) -> Option> { - schema_body_child(node) - .or_else(|| schema_container_child(node)) - .or_else(|| probe_field_child(node, BODY_FIELD_PRIORITY)) - .or_else(|| probe_structural_child(node)) - .and_then(descend_to_structural_target) -} - -pub fn value_container_target(node: Node<'_>) -> Option> { - for field in ["value", "body"] { - if let Some(child) = node.child_by_field_name(field) - && let Some(target) = descend_to_structural_target(child) - { - return Some(target); - } - } - - recurse_target(node) -} - -pub fn is_root_wrapper_node(node: Node<'_>) -> bool { - if is_atom_node(node) { - return false; - } - - if let Some(schema) = schema::schema_for_current(node.kind()) { - if schema.is_supertype { - return true; - } - if schema.is_structural() && !is_transparent_recurse_wrapper(node) { - return false; - } - } - - if identifier_node(node).is_some() { - return false; - } - - let non_trivia = local_named_children(node) - .into_iter() - .filter(|child| !is_generic_trivia(*child)) - .collect::>(); - non_trivia.len() == 1 && node_looks_structural(non_trivia[0]) -} - -pub fn is_generic_trivia(node: Node<'_>) -> bool { - node.is_extra() || kind_looks_like_comment(node.kind()) -} - -pub fn is_generic_absorbable_attr(kind: &str) -> bool { - matches!(kind, "attribute_item" | "inner_attribute_item") -} - -pub fn is_atom_node(node: Node<'_>) -> bool { - atom_list::is_atom_node_current(node.kind()) -} - -fn schema_identifier_node(node: Node<'_>) -> Option> { - let schema = schema::schema_for_current(node.kind())?; - for field in &schema.identifier_fields { - if let Some(child) = node.child_by_field_name(field) { - return Some(child); - } - } - None -} - -fn schema_body_child(node: Node<'_>) -> Option> { - let schema = schema::schema_for_current(node.kind())?; - for field in &schema.body_fields { - if let Some(child) = node.child_by_field_name(field) { - return Some(child); - } - } - None -} - -fn schema_container_child(node: Node<'_>) -> Option> { - let schema = schema::schema_for_current(node.kind())?; - local_named_children(node).into_iter().find(|child| { - schema - .container_child_kinds - .iter() - .any(|kind| child.kind() == kind) - }) -} - -fn probe_field_child<'tree>(node: Node<'tree>, fields: &[&str]) -> Option> { - fields - .iter() - .find_map(|field| node.child_by_field_name(field)) -} - -fn probe_structural_child(node: Node<'_>) -> Option> { - local_named_children(node).into_iter().find(|child| { - !is_generic_trivia(*child) - && !looks_like_identifier_kind(child.kind()) - && !is_atom_node(*child) - && node_looks_structural(*child) - }) -} - -fn descend_to_structural_target(node: Node<'_>) -> Option> { - if is_atom_node(node) { - return None; - } - - if is_transparent_recurse_wrapper(node) { - let child = local_named_children(node) - .into_iter() - .find(|child| !is_generic_trivia(*child) && node_looks_structural(*child))?; - return descend_to_structural_target(child).or(Some(child)); - } - - if node_looks_structural(node) { - return Some(node); - } - - probe_structural_child(node).and_then(descend_to_structural_target) -} - -fn is_transparent_recurse_wrapper(node: Node<'_>) -> bool { - if is_atom_node(node) || identifier_node(node).is_some() { - return false; - } - - schema::schema_for_current(node.kind()).is_some_and(|schema| schema.is_supertype) - || matches!(node.kind(), "document" | "stream") - || node.kind().ends_with("_node") -} - -fn node_looks_structural(node: Node<'_>) -> bool { - if is_atom_node(node) { - return false; - } - - if schema::schema_for_current(node.kind()).is_some_and(|schema| schema.is_structural()) { - return true; - } - - let non_trivia_children = local_named_children(node) - .into_iter() - .filter(|child| !is_generic_trivia(*child)) - .collect::>(); - !non_trivia_children.is_empty() - && non_trivia_children - .iter() - .any(|child| !looks_like_identifier_kind(child.kind()) || child.named_child_count() > 0) -} - -fn local_named_children(node: Node<'_>) -> Vec> { - let mut children = Vec::new(); - for index in 0..node.child_count() { - if let Some(child) = node.child(index) - && (child.is_named() || child.is_error() || child.kind() == "ERROR") - { - children.push(child); - } - } - children -} - -fn looks_like_identifier_kind(kind: &str) -> bool { - kind == "identifier" - || kind == "name" - || kind.ends_with("_identifier") - || kind.ends_with("_name") - || kind.ends_with("_label") -} - -fn kind_looks_like_comment(kind: &str) -> bool { - kind == "comment" || kind.contains("comment") -} - -/// Detect a call-with-trailing-callback pattern inside a node. -/// -/// Returns `(call_text_node, body_node)` where `call_text_node` is the function -/// being called (for name extraction) and `body_node` is the block body of the -/// trailing callback argument. -/// -/// This handles patterns like: -/// - JS/TS: `describe('x', () => { ... })`, `app.use(handler)` -/// - Go: `t.Run("x", func(t *testing.T) { ... })` -/// - Rust: `tokio::spawn(async { ... })` -/// - Ruby: `describe 'x' do ... end` (via `do_block`) -/// -/// Only matches when the last argument has a structural block body, so -/// expression-body arrows and simple value arguments are not promoted. -pub fn trailing_callback_body(node: Node<'_>) -> Option<(Node<'_>, Node<'_>)> { - // Look through the node's children for a call-like node. - let call = local_named_children(node) - .into_iter() - .find(|c| is_call_like(*c))?; - - // Find the arguments / parameter list. - let args_node = call - .child_by_field_name("arguments") - .or_else(|| call.child_by_field_name("args")) - .or_else(|| call.child_by_field_name("block"))?; - - // Get the last named child of the arguments list. - let last_arg = local_named_children(args_node).into_iter().last()?; - - // Try to find a structural body inside the last argument. - let body = recurse_target(last_arg)?; - - // Extract the call target node for name purposes. - let func_node = call - .child_by_field_name("function") - .or_else(|| call.child_by_field_name("method")) - .or_else(|| call.child_by_field_name("name")) - .unwrap_or(call); - - Some((func_node, body)) -} - -/// Returns `true` if the node looks like a function/method call. -fn is_call_like(node: Node<'_>) -> bool { - let kind = node.kind(); - // Explicit known call kinds. - if matches!( - kind, - "call_expression" - | "call" - | "function_call" - | "method_call" - | "method_call_expression" - | "invocation_expression" - ) { - return true; - } - // Heuristic: has an `arguments` field. - node.child_by_field_name("arguments").is_some() -} - -#[cfg(test)] -mod tests { - use ast_grep_core::tree_sitter::LanguageExt; - use tree_sitter::{Node, Parser}; - - use super::{ - identifier_node, is_root_wrapper_node, recurse_target, signature_end_byte, - trailing_callback_body, - }; - use crate::language::SupportLang; - - fn parse_tree_with_language( - source: &str, - language: SupportLang, - ) -> (crate::chunk::schema::SchemaLanguageGuard, tree_sitter::Tree) { - let schema_language = crate::chunk::schema::enter_language(language.canonical_name()); - let mut parser = Parser::new(); - parser - .set_language(&language.get_ts_language()) - .expect("language should parse"); - let tree = parser.parse(source, None).expect("tree should parse"); - (schema_language, tree) - } - - fn find_named_node<'tree>(node: Node<'tree>, kind: &str) -> Option> { - if node.kind() == kind { - return Some(node); - } - for index in 0..node.child_count() { - let Some(child) = node.child(index) else { - continue; - }; - if let Some(found) = find_named_node(child, kind) { - return Some(found); - } - } - None - } - - #[test] - fn python_function_definition_resolves_identifier_and_body() { - let (_schema_language, tree) = - parse_tree_with_language("def greet(name):\n return name\n", SupportLang::Python); - let node = - find_named_node(tree.root_node(), "function_definition").expect("function_definition"); - assert_eq!(identifier_node(node).expect("name").kind(), "identifier"); - assert_eq!(recurse_target(node).expect("body").kind(), "block"); - assert!(signature_end_byte(node).is_some()); - } - - #[test] - fn typescript_class_declaration_resolves_body() { - let (_schema_language, tree) = - parse_tree_with_language("class Greeter {\n hello() {}\n}\n", SupportLang::TypeScript); - let node = find_named_node(tree.root_node(), "class_declaration").expect("class_declaration"); - assert_eq!(recurse_target(node).expect("body").kind(), "class_body"); - } - - #[test] - fn yaml_wrapper_nodes_are_structural_without_shared_kind_lists() { - let (_schema_language, tree) = - parse_tree_with_language("root:\n child: 1\n", SupportLang::Yaml); - let node = find_named_node(tree.root_node(), "block_node").expect("block_node"); - assert!(is_root_wrapper_node(node), "block_node should be treated as a structural wrapper"); - } - - #[test] - fn js_expression_statement_with_callback_detects_body() { - let source = "describe(\"suite\", () => {\n\tit(\"a\", () => {});\n});\n"; - let (_schema_language, tree) = parse_tree_with_language(source, SupportLang::TypeScript); - let expr_stmt = - find_named_node(tree.root_node(), "expression_statement").expect("expression_statement"); - let (func_node, body) = - trailing_callback_body(expr_stmt).expect("should detect trailing callback body"); - assert_eq!(body.kind(), "statement_block"); - assert!( - matches!(func_node.kind(), "identifier" | "member_expression"), - "func_node should be an identifier or member_expression, got {}", - func_node.kind() - ); - } -} diff --git a/crates/pi-natives/src/chunk/state.rs b/crates/pi-natives/src/chunk/state.rs deleted file mode 100644 index 54d5d7036..000000000 --- a/crates/pi-natives/src/chunk/state.rs +++ /dev/null @@ -1,689 +0,0 @@ -use std::{ - collections::HashMap, - sync::{Arc, LazyLock}, -}; - -use napi::{Error, Result}; -use napi_derive::napi; -use regex::Regex; - -use super::{ - build_chunk_tree, - indent::{detect_file_indent_char, detect_file_indent_step, normalize_to_tabs}, - resolve::{ - ParsedSelector, chunk_region_range, format_region_ref, format_selector_tree, - resolve_chunk_selector, resolve_chunk_with_crc, split_selector_crc_and_region, - }, -}; -use crate::chunk::types::{ - ChunkInfo, ChunkNode, ChunkReadStatus, ChunkReadTarget, ChunkRegion, ChunkTree, EditParams, - EditResult, ReadRenderParams, ReadResult, RenderParams, VisibleLineRange, -}; - -const LINE_RANGE_SELECTOR_RE: &str = r"^L(\d+)(?:-L?(\d+))?$"; -const TLAPLUS_BEGIN_TRANSLATION_RE: &str = r"^\s*\\\*\s*BEGIN TRANSLATION\s*$"; -const TLAPLUS_END_TRANSLATION_RE: &str = r"^\s*\\\*\s*END TRANSLATION\s*$"; - -static LINE_RANGE_SELECTOR_REGEX: LazyLock = - LazyLock::new(|| Regex::new(LINE_RANGE_SELECTOR_RE).expect("line range regex must compile")); -static TLAPLUS_BEGIN_TRANSLATION_REGEX: LazyLock = LazyLock::new(|| { - Regex::new(TLAPLUS_BEGIN_TRANSLATION_RE).expect("tlaplus begin regex must compile") -}); -static TLAPLUS_END_TRANSLATION_REGEX: LazyLock = LazyLock::new(|| { - Regex::new(TLAPLUS_END_TRANSLATION_RE).expect("tlaplus end regex must compile") -}); - -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct ConflictMeta { - pub theirs_content: String, - pub ours_label: String, - pub theirs_label: String, - pub base_content: Option, - pub base_label: Option, - pub ours_start_byte: usize, - pub ours_end_byte: usize, -} - -#[derive(Clone)] -pub struct ChunkStateInner { - pub(crate) source: String, - pub(crate) language: String, - pub(crate) tree: ChunkTree, - pub(crate) notebook: Option, - pub(crate) conflict_meta: HashMap, - lookup: HashMap, - checksum_lookup: HashMap>, - leaf_lookup: HashMap>, - suffix_lookup: HashMap>, -} - -impl ChunkStateInner { - pub(crate) fn parse(source: String, language: String) -> Result { - let normalized_language = normalize_language(language.as_str()); - if normalized_language == "ipynb" { - let parsed = - crate::chunk::ast_ipynb::parse_notebook(&source).map_err(napi::Error::from_reason)?; - let kernel_lang = parsed.context.kernel_language.clone(); - let tree = crate::chunk::ast_ipynb::build_notebook_tree_from_virtual( - parsed.virtual_source.as_str(), - kernel_lang.as_str(), - ) - .map_err(napi::Error::from_reason)?; - let ctx = std::sync::Arc::new(parsed.context); - let mut inner = Self::new(parsed.virtual_source, normalized_language, tree); - inner.notebook = Some(ctx); - return Ok(inner); - } - if crate::chunk::conflict::has_conflict_markers(source.as_str()) { - let conflicts = crate::chunk::conflict::detect_conflicts(source.as_str()); - if !conflicts.is_empty() { - let clean_result = crate::chunk::conflict::accept_ours(source.as_str(), &conflicts); - let mut tree = - build_chunk_tree(clean_result.source.as_str(), normalized_language.as_str())?; - let conflict_meta = crate::chunk::conflict::inject_conflict_chunks( - &mut tree, - clean_result.source.as_str(), - &clean_result, - ); - let mut inner = Self::new(clean_result.source, normalized_language, tree); - inner.conflict_meta = conflict_meta; - return Ok(inner); - } - } - let tree = build_chunk_tree(source.as_str(), normalized_language.as_str())?; - Ok(Self::new(source, normalized_language, tree)) - } - - pub(crate) fn new(source: String, language: String, tree: ChunkTree) -> Self { - let mut lookup = HashMap::new(); - let mut checksum_lookup = HashMap::new(); - let mut leaf_lookup = HashMap::new(); - let mut suffix_lookup = HashMap::new(); - for (index, chunk) in tree.chunks.iter().enumerate() { - lookup.insert(chunk.path.clone(), index); - checksum_lookup - .entry(chunk.checksum.clone()) - .or_insert_with(Vec::new) - .push(index); - if chunk.path.is_empty() { - continue; - } - if let Some(leaf) = chunk.path.rsplit('.').next() { - leaf_lookup - .entry(leaf.to_string()) - .or_insert_with(Vec::new) - .push(index); - } - let segments = chunk.path.split('.').collect::>(); - for start in 1..segments.len() { - suffix_lookup - .entry(segments[start..].join(".")) - .or_insert_with(Vec::new) - .push(index); - } - } - Self { - source, - language, - tree, - notebook: None, - conflict_meta: HashMap::new(), - lookup, - checksum_lookup, - leaf_lookup, - suffix_lookup, - } - } - - pub(crate) const fn source(&self) -> &str { - self.source.as_str() - } - - pub(crate) const fn language(&self) -> &str { - self.language.as_str() - } - - pub(crate) const fn tree(&self) -> &ChunkTree { - &self.tree - } - - pub(crate) fn root(&self) -> Option<&ChunkNode> { - self.chunk("") - } - - pub(crate) fn chunk(&self, path: &str) -> Option<&ChunkNode> { - self - .lookup - .get(path) - .and_then(|index| self.tree.chunks.get(*index)) - } - - pub(crate) fn chunk_by_index(&self, index: usize) -> Option<&ChunkNode> { - self.tree.chunks.get(index) - } - - pub(crate) fn chunks_by_checksum(&self, checksum: &str) -> Vec<&ChunkNode> { - self - .checksum_lookup - .get(checksum) - .into_iter() - .flatten() - .filter_map(|index| self.chunk_by_index(*index)) - .collect() - } - - pub(crate) fn chunks_by_leaf(&self, leaf: &str) -> Vec<&ChunkNode> { - self - .leaf_lookup - .get(leaf) - .into_iter() - .flatten() - .filter_map(|index| self.chunk_by_index(*index)) - .collect() - } - - pub(crate) fn chunks_by_suffix(&self, suffix: &str) -> Vec<&ChunkNode> { - self - .suffix_lookup - .get(suffix) - .into_iter() - .flatten() - .filter_map(|index| self.chunk_by_index(*index)) - .collect() - } - - pub(crate) fn chunks(&self) -> impl Iterator { - self.tree.chunks.iter() - } - - pub(crate) fn child_chunks(&self, parent_path: &str) -> Vec<&ChunkNode> { - self - .chunk(parent_path) - .map(|parent| { - parent - .children - .iter() - .filter_map(|path| self.chunk(path.as_str())) - .collect() - }) - .unwrap_or_default() - } - - pub(crate) fn line_to_containing_chunk_path(&self, line: u32) -> Option { - crate::chunk::line_to_chunk_path(&self.tree, line) - } -} - -/// Parsed file as a chunk tree: query nodes, render views, format grep hits, -/// and apply edits. -#[napi] -#[derive(Clone)] -pub struct ChunkState { - inner: Arc, -} - -impl ChunkState { - pub(crate) fn from_inner(inner: ChunkStateInner) -> Self { - Self { inner: Arc::new(inner) } - } - - pub(crate) fn inner(&self) -> &ChunkStateInner { - self.inner.as_ref() - } -} - -#[napi] -impl ChunkState { - /// Build chunk state by parsing `source` with the given `language` id (e.g. - /// `typescript`). - #[napi(factory)] - pub fn parse(source: String, language: String) -> Result { - ChunkStateInner::parse(source, language).map(Self::from_inner) - } - - /// Normalized language identifier used for the tree-sitter parse. - #[napi(getter)] - pub fn language(&self) -> String { - self.inner.language().to_string() - } - - /// Full source text for this file. - #[napi(getter)] - pub fn source(&self) -> String { - self.inner.source().to_string() - } - - /// Stable checksum for the entire file contents. - #[napi(getter)] - pub fn checksum(&self) -> String { - self.inner.tree().checksum.clone() - } - - /// Line count of the source buffer. - #[napi(getter)] - pub fn line_count(&self) -> u32 { - self.inner.tree().line_count - } - - /// Count of tree-sitter error nodes seen while building the tree. - #[napi(getter)] - pub fn parse_errors(&self) -> u32 { - self.inner.tree().parse_errors - } - - /// True when a fallback classifier produced the tree. - #[napi(getter)] - pub fn fallback(&self) -> bool { - self.inner.tree().fallback - } - - /// Selector path string for the synthetic root (often empty). - #[napi(getter)] - pub fn root_path(&self) -> String { - self.inner.tree().root_path.clone() - } - - /// Top-level child chunk paths under the root. - #[napi(getter)] - pub fn root_children(&self) -> Vec { - self.inner.tree().root_children.clone() - } - - /// Total number of chunk nodes. - #[napi(getter)] - pub fn chunk_count(&self) -> u32 { - self.inner.tree().chunks.len() as u32 - } - - /// True when the parsed file contains unresolved merge conflicts. - #[napi] - pub fn has_conflicts(&self) -> bool { - !self.inner.conflict_meta.is_empty() - } - - /// Count of unresolved merge conflicts represented in the chunk tree. - #[napi] - pub fn conflict_count(&self) -> u32 { - self.inner.conflict_meta.len() as u32 - } - - /// Summary for the root chunk, if it exists. - #[napi] - pub fn root(&self) -> Option { - self.inner.root().map(chunk_info) - } - - /// Look up [`ChunkInfo`] for a chunk selector path. - #[napi] - pub fn chunk(&self, chunk_path: String) -> Option { - let mut warnings = Vec::new(); - resolve_chunk_selector(self.inner(), Some(chunk_path.as_str()), &mut warnings) - .ok() - .map(chunk_info) - } - - /// Every chunk node as a [`ChunkInfo`] list. - #[napi] - pub fn chunks(&self) -> Vec { - self.inner.chunks().map(chunk_info).collect() - } - - /// Direct children of `chunkPath` (use empty or omit for root); errors if - /// the path is missing. - #[napi] - pub fn children(&self, chunk_path: Option) -> Result> { - let parent = if let Some(chunk_path) = chunk_path { - let mut warnings = Vec::new(); - resolve_chunk_selector(self.inner(), Some(chunk_path.as_str()), &mut warnings) - .map_err(Error::from_reason)? - } else { - self - .inner - .root() - .ok_or_else(|| Error::from_reason("Chunk tree is missing the root chunk".to_string()))? - }; - Ok(self - .inner - .child_chunks(parent.path.as_str()) - .into_iter() - .map(chunk_info) - .collect()) - } - - /// Chunk selector path that contains 1-based source line `line`, if any. - #[napi] - pub fn line_to_containing_chunk_path(&self, line: u32) -> Option { - self.inner.line_to_containing_chunk_path(line) - } - - /// Render a chunk subtree or listing as UTF-8 text for tools. - #[napi] - pub fn render(&self, params: RenderParams) -> String { - crate::chunk::render::render_state(self.inner(), ¶ms) - } - - /// Parse `readPath` (selector, line scope, etc.) and return rendered text or - /// errors. - #[napi] - pub fn render_read(&self, params: ReadRenderParams) -> Result { - let ParsedChunkReadPath { selector, crc, region } = - match parse_chunk_read_path(params.read_path.as_str()) { - Ok(parsed) => parsed, - Err(err) => { - return Ok(ReadResult { - text: format!("{}\n\n{}", params.display_path, err), - chunk: Some(ChunkReadTarget { - status: ChunkReadStatus::UnsupportedRegion, - selector: params.read_path.clone(), - }), - }); - }, - }; - let visible_range = selector.as_deref().and_then(parse_visible_line_range); - let Some(root) = self.inner.root() else { - return Ok(ReadResult { - text: format!("{}\n\n[Chunk tree root missing]", params.display_path), - chunk: None, - }); - }; - - if let Some(visible_range) = visible_range { - if visible_range.start_line > self.inner.tree().line_count { - let suggestion = if self.inner.tree().line_count == 0 { - "The file is empty.".to_string() - } else { - format!( - "Use sel=L1 to read from the start, or sel=L{} to read the last line.", - self.inner.tree().line_count - ) - }; - return Ok(ReadResult { - text: format!( - "Line {} is beyond end of file ({} lines total). {suggestion}", - visible_range.start_line, - self.inner.tree().line_count, - ), - chunk: None, - }); - } - - let clamped_range = VisibleLineRange { - start_line: visible_range.start_line, - end_line: visible_range.end_line.min(self.inner.tree().line_count), - }; - let notice = format!( - "[Notice: chunk view scoped to requested lines L{}-L{}; clipped chunks keep head/tail \ - context and collapse non-overlapping children.]", - clamped_range.start_line, clamped_range.end_line - ); - let text = self.render(RenderParams { - chunk_path: Some(root.path.clone()), - title: params.display_path.clone(), - language_tag: params.language_tag.clone(), - visible_range: Some(clamped_range), - render_children_only: true, - omit_checksum: params.omit_checksum, - anchor_style: params.anchor_style, - show_leaf_preview: true, - tab_replacement: params.tab_replacement, - normalize_indent: params.normalize_indent, - focused_paths: None, - }); - return Ok(ReadResult { text: format!("{notice}\n\n{text}"), chunk: None }); - } - - if selector.as_deref().is_none_or(str::is_empty) && crc.is_none() && region.is_none() { - return Ok(ReadResult { - text: self.render(RenderParams { - chunk_path: Some(root.path.clone()), - title: params.display_path.clone(), - language_tag: params.language_tag.clone(), - visible_range: None, - render_children_only: true, - omit_checksum: params.omit_checksum, - anchor_style: params.anchor_style, - show_leaf_preview: true, - tab_replacement: params.tab_replacement, - normalize_indent: params.normalize_indent, - focused_paths: None, - }), - chunk: None, - }); - } - - if selector.as_deref() == Some("?") { - let mut lines = vec![format!("{} chunks (dot-joined paths):", params.display_path)]; - lines.extend(format_selector_tree( - self.inner.tree(), - &self.inner.tree().root_children, - false, - )); - return Ok(ReadResult { text: lines.join("\n"), chunk: None }); - } - - let mut warnings = Vec::new(); - let resolved = match resolve_chunk_with_crc( - self.inner(), - selector.as_deref(), - crc.as_deref(), - &mut warnings, - ) { - Ok(resolved) => resolved, - Err(err) => { - let sel = selector.unwrap_or_default(); - return Ok(ReadResult { - text: format!("{}:{}\n\n{}", params.display_path, sel, err), - chunk: Some(ChunkReadTarget { status: ChunkReadStatus::NotFound, selector: sel }), - }); - }, - }; - let chunk = resolved.chunk; - // Use the region from parse_chunk_read_path, NOT region, - // because resolve_chunk_with_crc re-parses the already-cleaned - // selector and loses the region suffix. - let selector_ref = format_region_ref(chunk, region); - - // Fall back to whole-chunk read when the chunk has no real region - // boundaries (e.g. markdown fenced blocks, leaf chunks without - // prologue/epilogue). Without this, `^` on such chunks returns - // "[Empty @^ region]" instead of the whole chunk. - let region = if region.is_some() - && (chunk.prologue_end_byte.is_none() || chunk.epilogue_start_byte.is_none()) - { - None - } else { - region - }; - - if let Some(absolute_line_range) = params.absolute_line_range { - let req_start = absolute_line_range.start_line; - let req_end = absolute_line_range.end_line; - let low = chunk.start_line.max(req_start.min(req_end)); - let high = chunk.end_line.min(req_start.max(req_end)); - if low > high { - let requested = if req_start == req_end { - format!("L{req_start}") - } else { - format!("L{req_start}-L{req_end}") - }; - return Ok(ReadResult { - text: format!( - "Requested range {requested} does not overlap {}:{} (lines {}-{}).", - params.display_path, chunk.path, chunk.start_line, chunk.end_line - ), - chunk: Some(ChunkReadTarget { - status: ChunkReadStatus::Ok, - selector: selector_ref, - }), - }); - } - } - - if let Some(target_region) = region { - let masked_source = mask_chunk_display_source(self.inner.source(), self.inner.language()); - let (start, end) = chunk_region_range(chunk, target_region); - let tab_replacement = params.tab_replacement.as_deref().unwrap_or(" "); - let normalize_indent = params.normalize_indent.unwrap_or(false).then(|| { - ( - detect_file_indent_char(self.inner.source(), self.inner.tree()), - detect_file_indent_step(self.inner.source(), self.inner.tree()) as usize, - ) - }); - // Extend the region start to the beginning of the line so that the - // leading indentation of the first line is included. Without this, - // regions whose start_byte is mid-line (e.g. a decorator `@property` - // inside a class) would show the first line without indentation, - // making the normalization inconsistent with subsequent lines. - let display_start = masked_source[..start].rfind('\n').map_or(0, |nl| nl + 1); - let region_text = masked_source - .get(display_start..end) - .unwrap_or_default() - .split('\n') - .map(|line| match normalize_indent { - Some((indent_char, indent_step)) => { - normalize_to_tabs(line, indent_char, indent_step) - }, - None => line.replace('\t', tab_replacement), - }) - .collect::>() - .join("\n"); - let text = if region_text.is_empty() { - format!("{selector_ref}\n\n[Empty @{} region]", target_region.as_str()) - } else { - format!("{selector_ref}\n\n{region_text}") - }; - return Ok(ReadResult { - text, - chunk: Some(ChunkReadTarget { status: ChunkReadStatus::Ok, selector: selector_ref }), - }); - } - - Ok(ReadResult { - text: self.render(RenderParams { - chunk_path: Some(chunk.path.clone()), - title: format!("{}:{}", params.display_path, chunk.path), - language_tag: params.language_tag.clone(), - visible_range: None, - render_children_only: false, - omit_checksum: params.omit_checksum, - anchor_style: params.anchor_style, - show_leaf_preview: true, - tab_replacement: params.tab_replacement, - normalize_indent: params.normalize_indent, - focused_paths: None, - }), - chunk: Some(ChunkReadTarget { status: ChunkReadStatus::Ok, selector: selector_ref }), - }) - } - - /// Prefix a grep line with `display_path` and the chunk path for - /// `line_number`, when known. - #[napi] - pub fn format_grep_line(&self, display_path: String, line_number: u32, line: String) -> String { - let chunk_path = self.inner.line_to_containing_chunk_path(line_number); - let location = chunk_path.map_or_else( - || display_path.clone(), - |chunk_path| { - if chunk_path.is_empty() { - display_path.clone() - } else { - format!("{display_path}:{chunk_path}") - } - }, - ); - format!("{location}>{line_number}|{line}") - } - - /// Apply batch edits, re-parse, write files, and return updated state and - /// messaging. - #[napi] - pub fn apply_edits(&self, params: EditParams) -> Result { - crate::chunk::edit::apply_edits(self, ¶ms).map_err(Error::from_reason) - } -} - -#[derive(Clone)] -struct ParsedChunkReadPath { - selector: Option, - crc: Option, - region: Option, -} - -fn normalize_language(language: &str) -> String { - language.trim().to_ascii_lowercase() -} - -fn chunk_info(chunk: &ChunkNode) -> ChunkInfo { - ChunkInfo { - path: chunk.path.clone(), - identifier: chunk.identifier.clone(), - checksum: chunk.checksum.clone(), - start_line: chunk.start_line, - end_line: chunk.end_line, - leaf: chunk.leaf, - } -} - -fn chunk_read_path_separator_index(read_path: &str) -> Option { - if read_path.len() >= 3 { - let bytes = read_path.as_bytes(); - if bytes[0].is_ascii_alphabetic() && bytes[1] == b':' && matches!(bytes[2], b'/' | b'\\') { - return read_path[2..].find(':').map(|index| index + 2); - } - } - read_path.find(':') -} - -fn parse_chunk_read_path(read_path: &str) -> std::result::Result { - let raw_selector = - chunk_read_path_separator_index(read_path).map(|index| &read_path[(index + 1)..]); - let ParsedSelector { selector, crc, region, .. } = - split_selector_crc_and_region(raw_selector, None, None)?; - Ok(ParsedChunkReadPath { selector, crc, region }) -} - -fn parse_visible_line_range(selector: &str) -> Option { - let captures = LINE_RANGE_SELECTOR_REGEX.captures(selector)?; - let start_line = captures.get(1)?.as_str().parse::().ok()?.max(1); - let end_line = captures - .get(2) - .and_then(|m| m.as_str().parse::().ok()) - .unwrap_or(start_line) - .max(start_line); - Some(VisibleLineRange { start_line, end_line }) -} - -pub fn mask_chunk_display_source(source: &str, language: &str) -> String { - if language != "tlaplus" { - return source.to_string(); - } - let lines = source.split('\n').collect::>(); - let mut masked = lines - .iter() - .map(|line| (*line).to_string()) - .collect::>(); - let mut index = 0usize; - while index < lines.len() { - if !TLAPLUS_BEGIN_TRANSLATION_REGEX.is_match(lines[index]) { - index += 1; - continue; - } - let begin_index = index; - let mut end_index = begin_index + 1; - while end_index < lines.len() && !TLAPLUS_END_TRANSLATION_REGEX.is_match(lines[end_index]) { - end_index += 1; - } - if begin_index + 1 < lines.len() { - masked[begin_index + 1] = "\\* [translation hidden]".to_string(); - for line in masked - .iter_mut() - .take(end_index.min(lines.len())) - .skip(begin_index + 2) - { - line.clear(); - } - } - index = end_index + 1; - } - masked.join("\n") -} diff --git a/crates/pi-natives/src/chunk/types.rs b/crates/pi-natives/src/chunk/types.rs deleted file mode 100644 index 17dbfd0ea..000000000 --- a/crates/pi-natives/src/chunk/types.rs +++ /dev/null @@ -1,403 +0,0 @@ -//! Shared types for the chunk-tree system. - -use napi_derive::napi; - -use crate::chunk::{kind::ChunkKind, state::ChunkState}; - -#[derive(Clone)] -pub struct ChunkNode { - pub path: String, - pub identifier: Option, - pub kind: ChunkKind, - pub leaf: bool, - /// For virtual chunks (for example `theirs` branches in conflicts), content - /// that is rendered instead of slicing `source`. - pub virtual_content: Option, - pub parent_path: Option, - pub children: Vec, - pub signature: Option, - pub start_line: u32, - pub end_line: u32, - pub line_count: u32, - pub start_byte: u32, - pub end_byte: u32, - /// Start byte of the semantic declaration used for checksums. This can be - /// later than `start_byte` when the chunk absorbs attached leading trivia - /// such as doc comments or attributes. - pub checksum_start_byte: u32, - - /// End byte of the prologue region. `None` means the chunk does not expose - /// regions. - pub prologue_end_byte: Option, - /// Start byte of the epilogue region. `None` means the chunk does not expose - /// regions. - pub epilogue_start_byte: Option, - - pub checksum: String, - pub error: bool, - pub indent: u32, - pub indent_char: String, - /// True for group-candidate chunks (e.g. `stmts`, `imports`, `decls`) that - /// represent an ordered list of similar items. Append/prepend is valid on - /// these even when they are leaf nodes. - pub group: bool, -} - -#[derive(Clone)] -pub struct ChunkTree { - pub language: String, - pub checksum: String, - pub line_count: u32, - pub parse_errors: u32, - /// 1-indexed line numbers of tree-sitter ERROR / MISSING nodes surfaced - /// during parsing. Used to focus error messages around failure locations - /// when an edit introduces a parse error far from the edit target. - pub parse_error_lines: Vec, - pub fallback: bool, - pub root_path: String, - pub root_children: Vec, - pub chunks: Vec, -} - -/// Summary of a single chunk node for tool output and navigation. -#[derive(Clone)] -#[napi(object)] -pub struct ChunkInfo { - /// Chunk selector path within the tree. - pub path: String, - /// Bare chunk identifier (without kind prefix), if available. - pub identifier: Option, - /// Stable checksum anchor for this chunk. - pub checksum: String, - /// 1-based start line in the source file (inclusive). - pub start_line: u32, - /// 1-based end line in the source file (inclusive). - pub end_line: u32, - /// Whether this node is a leaf (no child chunks). - pub leaf: bool, -} - -/// Result of resolving a chunk read request against the tree. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -#[napi(string_enum)] -pub enum ChunkReadStatus { - /// Selector matched a chunk and content was produced. - #[napi(value = "ok")] - Ok, - /// No chunk matched the requested selector. - #[napi(value = "not_found")] - NotFound, - /// Chunk matched but does not support the requested region. - #[napi(value = "unsupported_region")] - UnsupportedRegion, -} - -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -#[napi(string_enum)] -pub enum ChunkRegion { - #[napi(value = "^")] - Head, - #[napi(value = "~")] - Body, -} - -impl ChunkRegion { - pub const fn as_str(self) -> &'static str { - match self { - Self::Head => "^", - Self::Body => "~", - } - } -} - -/// Structural edit to apply relative to a chunk anchor. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -#[napi(string_enum)] -pub enum ChunkEditOp { - /// Put new content into the targeted region. - #[napi(value = "put")] - Put, - /// Find and replace a literal substring within the targeted region. - #[napi(value = "replace")] - Replace, - /// Remove the targeted region. - #[napi(value = "delete")] - Delete, - /// Insert `content` before the targeted region span. - #[napi(value = "before")] - Before, - /// Insert `content` after the targeted region span. - #[napi(value = "after")] - After, - /// Insert `content` at the start inside the targeted region. - #[napi(value = "prepend")] - Prepend, - /// Insert `content` at the end inside the targeted region. - #[napi(value = "append")] - Append, -} - -impl ChunkEditOp { - pub const fn as_str(self) -> &'static str { - match self { - Self::Put => "put", - Self::Replace => "replace", - Self::Delete => "delete", - Self::Before => "before", - Self::After => "after", - Self::Prepend => "prepend", - Self::Append => "append", - } - } -} - -/// Outcome of resolving which chunk was read for a `renderRead`-style request. -#[derive(Clone)] -#[napi(object)] -pub struct ChunkReadTarget { - /// Whether the selector matched. - pub status: ChunkReadStatus, - /// Sanitized selector string that was applied. - pub selector: String, -} - -/// Inclusive 1-based line range within a source file (used for scoped chunk -/// rendering). -#[derive(Clone)] -#[napi(object)] -pub struct VisibleLineRange { - /// First line to include. - pub start_line: u32, - /// Last line to include. - pub end_line: u32, -} - -/// How chunk anchors are formatted in rendered output (name and checksum -/// visibility). -#[derive(Clone, Copy, Default)] -#[napi(string_enum)] -pub enum ChunkAnchorStyle { - /// `[.name#crc]` style anchor. - #[default] - #[napi(value = "full")] - Full, - /// `[.kind#crc]` style anchor (kind is the name prefix before `_`). - #[napi(value = "kind")] - Kind, - /// `[#crc]` style anchor. - #[napi(value = "bare")] - Bare, - /// `[.name]` without checksum. - #[napi(value = "full-omit")] - FullOmit, - /// `[.kind]` without checksum. - #[napi(value = "kind-omit")] - KindOmit, - /// Minimal anchor without name or checksum. - #[napi(value = "none")] - None, -} - -impl ChunkAnchorStyle { - pub const fn with_omit_checksum(self, omit: bool) -> Self { - if !omit { - return self; - } - match self { - Self::Full => Self::FullOmit, - Self::Kind => Self::KindOmit, - Self::Bare => Self::None, - Self::FullOmit => Self::FullOmit, - Self::KindOmit => Self::KindOmit, - Self::None => Self::None, - } - } - - fn render_i( - &self, - marker: (&str, &str), - indent: &str, - name: &str, - crc: &str, - line_count_suffix: &str, - ) -> String { - fn extract_kind(name: &str) -> &str { - name.find('_').map_or_else(|| name, |index| &name[..index]) - } - let (open, close) = marker; - match self { - Self::Full => format!("{indent}{open}{name}#{crc}{close}{line_count_suffix}"), - Self::Kind => { - format!( - "{indent}{open}{kind}#{crc}{close}{line_count_suffix}", - kind = extract_kind(name) - ) - }, - Self::Bare => format!("{indent}{open}#{crc}{close}{line_count_suffix}"), - Self::FullOmit => format!("{indent}{open}{name}{close}{line_count_suffix}"), - Self::KindOmit => { - format!("{indent}{open}{kind}{close}{line_count_suffix}", kind = extract_kind(name)) - }, - Self::None => String::new(), - } - } - - /// Render an opening anchor: `{prefix}@{name}#{crc}`. - pub fn render(&self, prefix: &str, name: &str, crc: &str) -> String { - self.render_i(("@", ""), prefix, name, crc, "") - } -} - -/// How a chunk participates in a focus-scoped render pass. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -#[napi(string_enum)] -pub enum ChunkFocusMode { - /// Emit full content and recurse normally. - #[napi(value = "expanded")] - Expanded, - /// Emit just the opening anchor; do not recurse or emit body. - #[napi(value = "collapsed")] - Collapsed, - /// Emit opening + closing anchors; recurse into focused children only. - /// Interior gap lines between children are suppressed. - #[napi(value = "container")] - Container, -} - -/// Path + focus mode pair for the N-API boundary (`HashMap` doesn't cross FFI). -#[derive(Clone, Debug)] -#[napi(object)] -pub struct FocusedPath { - pub path: String, - pub mode: ChunkFocusMode, -} - -/// Options for `ChunkState.render`: which subtree to show and how anchors -/// appear. -#[derive(Clone)] -#[napi(object)] -pub struct RenderParams { - /// Path of the chunk to render; `None` uses the tree root. - pub chunk_path: Option, - /// Title line shown above the tree (often the file path). - pub title: String, - /// Optional language label for the header block. - pub language_tag: Option, - /// Restrict output to an inclusive line range of the file. - pub visible_range: Option, - /// When true, list only direct children instead of a full subtree. - pub render_children_only: bool, - /// Hide checksums in anchors when true. - pub omit_checksum: bool, - /// Anchor formatting style for chunk headers. - pub anchor_style: Option, - /// Include a one-line preview for leaf chunks. - pub show_leaf_preview: bool, - /// Replace tab characters in displayed previews (e.g. two spaces). - pub tab_replacement: Option, - /// When true, normalize displayed indentation to canonical tabs. - pub normalize_indent: Option, - - /// When set, restrict rendering to these chunks with their specified focus - /// modes. Everything not in this list is skipped. - pub focused_paths: Option>, -} - -/// Options for `ChunkState.renderRead`: selector path, display path, and -/// optional line scoping. -#[derive(Clone)] -#[napi(object)] -pub struct ReadRenderParams { - /// Read selector (`sel=...` path, line range, or empty for whole tree). - pub read_path: String, - /// Path shown in titles and error messages (often the file path). - pub display_path: String, - /// Optional language label for the rendered block. - pub language_tag: Option, - /// Hide checksums in rendered anchors. - pub omit_checksum: bool, - /// Anchor formatting style. - pub anchor_style: Option, - /// Optional absolute file line range to intersect with the resolved chunk. - pub absolute_line_range: Option, - /// Replace tabs in embedded previews. - pub tab_replacement: Option, - /// When true, normalize displayed indentation to canonical tabs. - pub normalize_indent: Option, -} - -/// Rendered chunk text plus optional resolution metadata for the read request. -#[derive(Clone)] -#[napi(object)] -pub struct ReadResult { - /// Rendered UTF-8 text (chunk tree, notice, or error message). - pub text: String, - /// When a selector was used, whether it matched and which selector applied. - pub chunk: Option, -} - -/// One edit in a batch; targets a chunk via `sel`/`crc` (with params-level -/// defaults). -#[derive(Clone)] -#[napi(object)] -pub struct EditOperation { - /// Edit kind (replace, delete, insert relative to anchor). - pub op: ChunkEditOp, - /// Chunk selector path; falls back to `EditParams.defaultSelector` when - /// omitted. - pub sel: Option, - /// Optional checksum anchor; falls back to `EditParams.defaultCrc` when - /// omitted. - pub crc: Option, - /// Region to target. When omitted, targets the full chunk. - pub region: Option, - /// Replacement or inserted text (meaning depends on `op`). - pub content: Option, - /// For `replace` op: literal substring to find inside the target chunk. - /// Must match exactly once. - pub find: Option, -} - -/// Arguments for applying a batch of chunk edits to a file. -#[derive(Clone)] -#[napi(object)] -pub struct EditParams { - /// Edits to apply in order. - pub operations: Vec, - /// When true, normalize indentation for response rendering and inserted - /// content. When false, preserve literal tabs/spaces. - pub normalize_indent: Option, - /// Default chunk selector when an `EditOperation` omits `sel`. - pub default_selector: Option, - /// Default checksum when an `EditOperation` omits `crc`. - pub default_crc: Option, - /// Anchor formatting for rendered response text. - pub anchor_style: Option, - /// Working directory used to resolve `filePath` and display paths. - pub cwd: String, - /// Path to the source file to edit (often relative to `cwd`). - pub file_path: String, -} - -/// Result of applying edits: new parse state plus before/after source and -/// messaging. -#[derive(Clone)] -#[napi(object, object_from_js = false)] -pub struct EditResult { - /// Chunk tree state after applying edits and re-parsing. - pub state: ChunkState, - /// Full file text before edits. - pub diff_before: String, - /// Full file text after edits. - pub diff_after: String, - /// Rendered summary for tooling (hunks, anchors), driven by `anchorStyle`. - pub response_text: String, - /// Whether the on-disk source changed. - pub changed: bool, - /// Whether the updated source re-parsed without fatal issues. - pub parse_valid: bool, - /// Absolute or normalized paths that were written or touched. - pub touched_paths: Vec, - /// Non-fatal issues (e.g. selector warnings) collected during apply. - pub warnings: Vec, -} diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index a39500520..d89dca487 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -27,7 +27,6 @@ static GLOBAL: MiMalloc = MiMalloc; pub mod appearance; pub mod ast; -pub mod chunk; pub mod clipboard; pub mod fd; pub mod fs_cache; diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 94f5cd3d4..8eb3344bc 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,15 @@ # Changelog ## [Unreleased] +### Removed + +- Removed the `chunk` edit mode, chunk-aware `read` selectors, chunk-aware `grep` rendering, and the `omp read` chunk CLI subcommand +- Removed the `read.prosechunks`, `read.explorechunks`, and `read.anchorstyle` settings +- Removed the underlying `chunk` native module and AST-based chunk schema generation from `pi-natives` + +### Fixed + +- Fixed `poll` wait duration parsing to fall back to `30s` when the provided value is an empty string ## [14.4.1] - 2026-04-26 ### Breaking Changes diff --git a/packages/coding-agent/src/cli.ts b/packages/coding-agent/src/cli.ts index 7bca23ab5..d04888e88 100755 --- a/packages/coding-agent/src/cli.ts +++ b/packages/coding-agent/src/cli.ts @@ -49,7 +49,6 @@ const commands: CommandEntry[] = [ { name: "config", load: () => import("./commands/config").then(m => m.default) }, { name: "grep", load: () => import("./commands/grep").then(m => m.default) }, { name: "grievances", load: () => import("./commands/grievances").then(m => m.default) }, - { name: "read", load: () => import("./commands/read").then(m => m.default) }, { name: "jupyter", load: () => import("./commands/jupyter").then(m => m.default) }, { name: "plugin", load: () => import("./commands/plugin").then(m => m.default) }, { name: "setup", load: () => import("./commands/setup").then(m => m.default) }, diff --git a/packages/coding-agent/src/cli/read-cli.ts b/packages/coding-agent/src/cli/read-cli.ts deleted file mode 100644 index 12e8aa9b2..000000000 --- a/packages/coding-agent/src/cli/read-cli.ts +++ /dev/null @@ -1,67 +0,0 @@ -/** - * Read CLI command handler. - * - * Handles `omp read` subcommand — emits chunk-mode read output for files, - * and delegates URL reads through the read tool pipeline. - */ -import * as path from "node:path"; -import chalk from "chalk"; -import { Settings } from "../config/settings"; -import { formatChunkedRead, resolveAnchorStyle } from "../edit/modes/chunk"; -import { getLanguageFromPath } from "../modes/theme/theme"; -import type { ToolSession } from "../tools"; -import { parseReadUrlTarget } from "../tools/fetch"; -import { ReadTool } from "../tools/read"; - -export interface ReadCommandArgs { - path: string; - sel?: string; -} - -function createCliReadSession(cwd: string, settings: Settings): ToolSession { - return { - cwd, - hasUI: false, - hasEditTool: true, - getSessionFile: () => null, - getSessionSpawns: () => null, - settings, - }; -} - -export async function runReadCommand(cmd: ReadCommandArgs): Promise { - const cwd = process.cwd(); - const parsedUrlTarget = parseReadUrlTarget(cmd.path, cmd.sel); - if (parsedUrlTarget) { - const settings = await Settings.init({ cwd }); - const tool = new ReadTool(createCliReadSession(cwd, settings)); - const result = await tool.execute("cli-read", { path: cmd.path, sel: cmd.sel }); - const text = result.content.find((content): content is { type: "text"; text: string } => content.type === "text"); - console.log(text?.text ?? ""); - return; - } - - const filePath = path.resolve(cmd.path); - const file = Bun.file(filePath); - if (!(await file.exists())) { - console.error(chalk.red(`Error: File not found: ${cmd.path}`)); - process.exit(1); - } - - const readPath = cmd.sel ? `${filePath}:${cmd.sel}` : filePath; - const language = getLanguageFromPath(filePath); - - try { - const result = await formatChunkedRead({ - filePath, - readPath, - cwd, - language, - anchorStyle: resolveAnchorStyle(), - }); - console.log(result.text); - } catch (err) { - console.error(chalk.red(`Error: ${err instanceof Error ? err.message : String(err)}`)); - process.exit(1); - } -} diff --git a/packages/coding-agent/src/commands/read.ts b/packages/coding-agent/src/commands/read.ts deleted file mode 100644 index 99cd5b7de..000000000 --- a/packages/coding-agent/src/commands/read.ts +++ /dev/null @@ -1,33 +0,0 @@ -/** - * Chunk-mode read tool. - */ -import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli"; -import { type ReadCommandArgs, runReadCommand } from "../cli/read-cli"; -import { initTheme } from "../modes/theme/theme"; - -export default class Read extends Command { - static description = "Read a file as a chunk tree"; - - static args = { - path: Args.string({ description: "File path to read", required: true }), - }; - - static flags = { - sel: Flags.string({ - char: "s", - description: "Chunk selector or line range (e.g. class_Foo.fn_bar, L10-L20)", - }), - }; - - async run(): Promise { - const { args, flags } = await this.parse(Read); - - const cmd: ReadCommandArgs = { - path: args.path ?? "", - sel: flags.sel, - }; - - await initTheme(); - await runReadCommand(cmd); - } -} diff --git a/packages/coding-agent/src/config/prompt-templates.ts b/packages/coding-agent/src/config/prompt-templates.ts index ab6ed2a73..e1549b2fc 100644 --- a/packages/coding-agent/src/config/prompt-templates.ts +++ b/packages/coding-agent/src/config/prompt-templates.ts @@ -1,6 +1,5 @@ import * as fs from "node:fs"; import * as path from "node:path"; -import { type ChunkAnchorStyle, formatAnchor } from "@oh-my-pi/pi-natives"; import { getProjectDir, getProjectPromptsDir, @@ -62,35 +61,6 @@ prompt.registerHelper("hline", (lineNum: unknown, content: unknown): string => { return `${ref}${HASHLINE_CONTENT_SEPARATOR}${text}`; }); -/** - * {{anchor name checksum}} — render a branch anchor tag using the current anchor style. - * Style is resolved from the template context (`anchorStyle`) or defaults to "full". - */ -prompt.registerHelper("anchor", function (this: prompt.TemplateContext, name: string, checksum: string): string { - const style = (this.anchorStyle as ChunkAnchorStyle) ?? "full"; - return formatAnchor(name, checksum, style); -}); - -/** - * {{sel "parent_Name.child_Name"}} — render a chunk path for `sel` fields in examples. - * In `full` style the path is returned as-is (`class_Server.fn_start`). - * In `kind` style each segment is trimmed to its kind prefix (`class.fn`). - * In `bare` style the path is omitted (the model uses only `crc` to identify chunks). - */ -prompt.registerHelper("sel", function (this: prompt.TemplateContext, chunkPath: string): string { - const style = (this.anchorStyle as ChunkAnchorStyle) ?? "full"; - if (style === "full") return chunkPath; - if (style === "bare") return ""; - // kind: trim each segment to its kind prefix (before the first `_`) - return chunkPath - .split(".") - .map(seg => { - const idx = seg.indexOf("_"); - return idx === -1 ? seg : seg.slice(0, idx); - }) - .join("."); -}); - const INLINE_ARG_SHELL_PATTERN = /\$(?:ARGUMENTS|@(?:\[\d+(?::\d*)?\])?|\d+)/; const INLINE_ARG_TEMPLATE_PATTERN = /\{\{[\s\S]*?(?:\b(?:arguments|ARGUMENTS|args)\b|\barg\s+[^}]+)[\s\S]*?\}\}/; diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index cd6556618..62a0e86e5 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -955,12 +955,12 @@ export const SETTINGS_SCHEMA = { // Edit tool "edit.mode": { type: "enum", - values: ["replace", "patch", "hashline", "chunk", "vim", "apply_patch", "atom"] as const, + values: ["replace", "patch", "hashline", "vim", "apply_patch", "atom"] as const, default: "hashline", ui: { tab: "editing", label: "Edit Mode", - description: "Select the edit tool variant (replace, patch, hashline, chunk, vim, or apply_patch)", + description: "Select the edit tool variant (replace, patch, hashline, vim, or apply_patch)", }, }, @@ -1046,38 +1046,6 @@ export const SETTINGS_SCHEMA = { }, }, - "read.prosechunks": { - type: "boolean", - default: false, - ui: { - tab: "editing", - label: "Prose Chunks", - description: "Enable chunk rendering for prose files in chunk edit mode", - }, - }, - - "read.explorechunks": { - type: "boolean", - default: false, - ui: { - tab: "editing", - label: "Explore Chunks", - description: "Show chunk tree without checksums for read-only agents like explore", - }, - }, - - "read.anchorstyle": { - type: "enum", - values: ["full", "kind", "bare"], - default: "full", - ui: { - tab: "editing", - label: "Anchor Style", - description: "Render chunk anchors with full names, kind prefixes, or checksum-only tags", - submenu: true, - }, - }, - // LSP "lsp.enabled": { type: "boolean", diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index b49b68e2a..4e4c39cf9 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -326,7 +326,7 @@ export class Settings { /** * Get the edit variant for a specific model. - * Returns "patch", "replace", "hashline", "chunk", "vim", "apply_patch", or null (use global default). + * Returns "patch", "replace", "hashline", "vim", "apply_patch", or null (use global default). */ getEditVariantForModel(model: string | undefined): EditMode | null { if (!model) return null; diff --git a/packages/coding-agent/src/edit/index.ts b/packages/coding-agent/src/edit/index.ts index 20df32e5d..a5499bd64 100644 --- a/packages/coding-agent/src/edit/index.ts +++ b/packages/coding-agent/src/edit/index.ts @@ -10,7 +10,6 @@ import { } from "../lsp"; import applyPatchDescription from "../prompts/tools/apply-patch.md" with { type: "text" }; import atomDescription from "../prompts/tools/atom.md" with { type: "text" }; -import chunkEditDescription from "../prompts/tools/chunk-edit.md" with { type: "text" }; import hashlineDescription from "../prompts/tools/hashline.md" with { type: "text" }; import patchDescription from "../prompts/tools/patch.md" with { type: "text" }; import replaceDescription from "../prompts/tools/replace.md" with { type: "text" }; @@ -27,15 +26,6 @@ import { executeAtomSingle, resolveAtomEntryPaths, } from "./modes/atom"; -import { - type ChunkParams, - type ChunkToolEdit, - chunkEditParamsSchema, - executeChunkSingle, - parseChunkEditPath, - resolveAnchorStyle, - resolveChunkAutoIndent, -} from "./modes/chunk"; import { executeHashlineSingle, HashlineMismatchError, @@ -53,7 +43,6 @@ export * from "./diff"; export * from "./line-hash"; export * from "./modes/apply-patch"; export * from "./modes/atom"; -export * from "./modes/chunk"; export * from "./modes/hashline"; export * from "./modes/patch"; export * from "./modes/replace"; @@ -66,7 +55,6 @@ type TInput = | typeof patchEditSchema | typeof hashlineEditParamsSchema | typeof atomEditParamsSchema - | typeof chunkEditParamsSchema | typeof vimSchema | typeof applyPatchSchema; @@ -76,7 +64,6 @@ type EditParams = | PatchParams | HashlineParams | AtomParams - | ChunkParams | VimParams | ApplyPatchParams; type EditToolResultDetails = EditToolDetails | VimToolDetails; @@ -325,39 +312,6 @@ export class EditTool implements AgentTool { #getModeDefinition(): EditModeDefinition { return { - chunk: { - description: (session: ToolSession) => - prompt.render(chunkEditDescription, { - anchorStyle: resolveAnchorStyle(session.settings), - chunkAutoIndent: resolveChunkAutoIndent(), - }), - parameters: chunkEditParamsSchema, - execute: ( - tool: EditTool, - params: EditParams, - signal: AbortSignal | undefined, - batchRequest: LspBatchRequest | undefined, - onUpdate?: (partialResult: AgentToolResult) => void, - ) => { - const { edits, path: topPath } = params as ChunkParams & { path?: string }; - const resolved = resolveEntryPaths(edits as ChunkToolEdit[], topPath); - const byFile = groupBy(resolved, (e: ChunkToolEdit) => parseChunkEditPath(e.path).filePath); - const entries = [...byFile.entries()].map(([filePath, fileEdits]) => ({ - path: filePath, - run: (br: LspBatchRequest | undefined) => - executeChunkSingle({ - session: tool.session, - path: filePath, - edits: fileEdits, - signal, - batchRequest: br, - writethrough: tool.#writethrough, - beginDeferredDiagnosticsForPath: p => tool.#beginDeferredDiagnosticsForPath(p), - }), - })); - return executePerFile(entries, batchRequest, onUpdate); - }, - }, patch: { description: () => prompt.render(patchDescription), parameters: patchEditSchema, diff --git a/packages/coding-agent/src/edit/line-hash.ts b/packages/coding-agent/src/edit/line-hash.ts index 6b378cf06..9f77711e9 100644 --- a/packages/coding-agent/src/edit/line-hash.ts +++ b/packages/coding-agent/src/edit/line-hash.ts @@ -667,59 +667,6 @@ export const HASHLINE_BIGRAMS = [ export const HASHLINE_BIGRAMS_COUNT = HASHLINE_BIGRAMS.length; -/** - * 40 common English BPE bigrams used by chunk checksums (`path#checksum`). - * Kept separate from {@link HASHLINE_BIGRAMS} because the chunk checksum - * format is `path#bigram1bigram2` (4 chars from a 1600-code namespace) and - * is independent of the line-anchor format. - * - * Order is stable forever — changing it invalidates every saved chunk path. - */ -export const CHUNK_BIGRAMS = [ - "th", - "he", - "in", - "er", - "an", - "re", - "on", - "at", - "en", - "nd", - "ti", - "es", - "or", - "te", - "of", - "ed", - "is", - "it", - "al", - "ar", - "st", - "to", - "nt", - "ng", - "se", - "ha", - "as", - "ou", - "io", - "le", - "ve", - "co", - "me", - "de", - "hi", - "ri", - "ro", - "ic", - "ne", - "ea", -] as const; - -export const CHUNK_BIGRAMS_COUNT = CHUNK_BIGRAMS.length; - /** * Regex source matching exactly one bigram from {@link HASHLINE_BIGRAMS}. * Used by hashline parsers — keep in sync with the alphabet array above. diff --git a/packages/coding-agent/src/edit/modes/chunk.ts b/packages/coding-agent/src/edit/modes/chunk.ts deleted file mode 100644 index 8ca1f599c..000000000 --- a/packages/coding-agent/src/edit/modes/chunk.ts +++ /dev/null @@ -1,832 +0,0 @@ -import * as fs from "node:fs/promises"; -import * as nodePath from "node:path"; -import type { AgentToolResult } from "@oh-my-pi/pi-agent-core"; -import { StringEnum } from "@oh-my-pi/pi-ai"; -import { - ChunkAnchorStyle, - ChunkEditOp, - type ChunkInfo, - ChunkReadStatus, - type ChunkReadTarget, - ChunkRegion, - ChunkState, - type EditOperation as NativeEditOperation, -} from "@oh-my-pi/pi-natives"; -import { $envpos } from "@oh-my-pi/pi-utils"; -import { type Static, Type } from "@sinclair/typebox"; -import type { BunFile } from "bun"; -import { LRUCache } from "lru-cache"; -import type { Settings } from "../../config/settings"; -import type { WritethroughCallback, WritethroughDeferredHandle } from "../../lsp"; -import { getLanguageFromPath } from "../../modes/theme/theme"; -import type { ToolSession } from "../../tools"; -import { assertEditableFileContent } from "../../tools/auto-generated-guard"; -import { invalidateFsScanAfterWrite } from "../../tools/fs-cache-invalidation"; -import { outputMeta } from "../../tools/output-meta"; -import { enforcePlanModeWrite, resolvePlanPath } from "../../tools/plan-mode-guard"; -import { generateUnifiedDiffString } from "../diff"; -import { HASHLINE_BIGRAMS } from "../line-hash"; -import { detectLineEnding, normalizeToLF, restoreLineEndings, stripBom } from "../normalize"; -import type { EditToolDetails, LspBatchRequest } from "../renderer"; - -export type { ChunkReadTarget }; - -export type ChunkEditOperation = - | { op: "put"; sel?: string; content: string } - | { op: "delete"; sel?: string } - | { op: "before"; sel?: string; content: string } - | { op: "after"; sel?: string; content: string } - | { op: "prepend"; sel?: string; content: string } - | { op: "append"; sel?: string; content: string }; - -type ChunkEditResult = { - diffSourceBefore: string; - diffSourceAfter: string; - responseText: string; - changed: boolean; - parseValid: boolean; - touchedPaths: string[]; - warnings: string[]; -}; - -export type ParsedChunkReadPath = { - filePath: string; - selector?: string; -}; - -type ChunkCacheEntry = { - mtimeMs: number; - size: number; - source: string; - state: ChunkState; -}; - -const validAnchorStyles: Record = { - full: ChunkAnchorStyle.Full, - kind: ChunkAnchorStyle.Kind, - bare: ChunkAnchorStyle.Bare, -}; - -export function resolveChunkAutoIndent(rawValue = Bun.env.PI_CHUNK_AUTOINDENT): boolean { - if (!rawValue) return true; - const normalized = rawValue.trim().toLowerCase(); - switch (normalized) { - case "1": - case "true": - case "yes": - case "on": - return true; - case "0": - case "false": - case "no": - case "off": - return false; - default: - throw new Error(`Invalid PI_CHUNK_AUTOINDENT: ${rawValue}`); - } -} - -function getChunkRenderIndentOptions(): { - normalizeIndent: boolean; - tabReplacement: string; -} { - return resolveChunkAutoIndent() - ? { normalizeIndent: true, tabReplacement: " " } - : { normalizeIndent: false, tabReplacement: "\t" }; -} - -export function resolveAnchorStyle(settings?: Settings): ChunkAnchorStyle { - const envStyle = Bun.env.PI_ANCHOR_STYLE; - return ( - (envStyle && validAnchorStyles[envStyle]) || - (settings?.get("read.anchorstyle") as ChunkAnchorStyle | undefined) || - ChunkAnchorStyle.Full - ); -} - -const chunkStateCache = new LRUCache({ - max: $envpos("PI_CHUNK_CACHE_MAX_ENTRIES", 200), -}); - -export function invalidateChunkCache(filePath: string): void { - chunkStateCache.delete(filePath); -} - -type ChunkSourceContext = { - resolvedPath: string; - sourceFile: BunFile; - sourceExists: boolean; - rawContent: string; - chunkLanguage: string | undefined; -}; - -type ChunkSourceIntent = "read" | "write"; - -function normalizeLanguage(language: string | undefined): string { - return language?.trim().toLowerCase() || ""; -} - -function normalizeChunkSource(text: string): string { - return normalizeToLF(stripBom(text).text); -} - -function displayPathForFile(filePath: string, cwd: string): string { - const relative = nodePath.relative(cwd, filePath).replace(/\\/g, "/"); - return relative && !relative.startsWith("..") ? relative : filePath.replace(/\\/g, "/"); -} - -function fileLanguageTag(filePath: string, language?: string): string | undefined { - const normalizedLanguage = normalizeLanguage(language); - if (normalizedLanguage.length > 0) return normalizedLanguage; - const ext = nodePath.extname(filePath).replace(/^\./, "").toLowerCase(); - return ext.length > 0 ? ext : undefined; -} - -async function resolveChunkSourceContext( - session: ToolSession, - path: string, - options?: { intent?: ChunkSourceIntent }, -): Promise { - const resolvedPath = resolvePlanPath(session, path); - const sourceFile = Bun.file(resolvedPath); - const sourceExists = await sourceFile.exists(); - if ((options?.intent ?? "write") === "write") { - enforcePlanModeWrite(session, path, { op: sourceExists ? "update" : "create" }); - } - - let rawContent = ""; - if (sourceExists) { - rawContent = await sourceFile.text(); - assertEditableFileContent(rawContent, path); - } - - return { - resolvedPath, - sourceFile, - sourceExists, - rawContent, - chunkLanguage: getLanguageFromPath(resolvedPath), - }; -} - -/** - * Preview-safe loader: read raw source without plan-mode enforcement or - * editable-file guards. Used by streaming diff previews that must not throw - * side-effecting errors while args are still being streamed. - */ -export async function loadChunkSource(params: { - cwd: string; - path: string; -}): Promise<{ resolvedPath: string; rawContent: string; language: string | undefined; exists: boolean }> { - const resolvedPath = nodePath.isAbsolute(params.path) ? params.path : nodePath.resolve(params.cwd, params.path); - const sourceFile = Bun.file(resolvedPath); - const exists = await sourceFile.exists(); - const rawContent = exists ? await sourceFile.text() : ""; - return { resolvedPath, rawContent, language: getLanguageFromPath(resolvedPath), exists }; -} - -/** - * Compute a unified diff preview for a chunk edit without applying it. - * Used for streaming previews while args are still arriving. Returns - * `{ error }` on any failure so callers can decide whether to surface it. - */ -export async function computeChunkDiff( - input: { path: string; edits: ChunkToolEdit[] }, - cwd: string, - options?: { anchorStyle?: ChunkAnchorStyle; signal?: AbortSignal }, -): Promise<{ diff: string; firstChangedLine: number | undefined } | { error: string }> { - try { - options?.signal?.throwIfAborted?.(); - const { filePath } = parseChunkEditPath(input.path); - if (!filePath) return { error: "chunk edit path is empty" }; - const { resolvedPath, rawContent, language } = await loadChunkSource({ cwd, path: filePath }); - options?.signal?.throwIfAborted?.(); - const { operations } = normalizeChunkEditOperations(input.edits); - const result = applyChunkEdits({ - source: rawContent, - language, - cwd, - filePath: resolvedPath, - operations, - anchorStyle: options?.anchorStyle, - }); - options?.signal?.throwIfAborted?.(); - if (!result.changed) { - return { diff: "", firstChangedLine: undefined }; - } - return generateUnifiedDiffString(result.diffSourceBefore, result.diffSourceAfter); - } catch (err) { - return { error: err instanceof Error ? err.message : String(err) }; - } -} - -function normalizeChunkRegionSyntax(text: string): string { - return text.replaceAll("@body", "~").replaceAll("@head", "^"); -} - -function buildChunkEditResult(result: { - diffBefore: string; - diffAfter: string; - responseText: string; - changed: boolean; - parseValid: boolean; - touchedPaths: string[]; - warnings: string[]; -}): ChunkEditResult { - return { - diffSourceBefore: result.diffBefore, - diffSourceAfter: result.diffAfter, - responseText: result.responseText, - changed: result.changed, - parseValid: result.parseValid, - touchedPaths: result.touchedPaths, - warnings: result.warnings.map(normalizeChunkRegionSyntax), - }; -} - -function chunkReadPathSeparatorIndex(readPath: string): number { - if (/^[a-zA-Z]:[/\\]/.test(readPath)) { - return readPath.indexOf(":", 2); - } - const urlMatch = readPath.match(/^([a-z][a-z0-9+.-]*):\/\//i); - if (urlMatch) { - const scheme = urlMatch[1].toLowerCase(); - const urlPrefixEnd = urlMatch[0].length; - if (scheme === "local") { - const index = readPath.lastIndexOf(":"); - return index >= urlPrefixEnd ? index : -1; - } - - const pathStart = readPath.indexOf("/", urlPrefixEnd); - if (pathStart === -1) return -1; - const index = readPath.lastIndexOf(":"); - return index >= pathStart ? index : -1; - } - return readPath.indexOf(":"); -} - -export function parseChunkSelector(selector: string | undefined): { selector?: string } { - if (!selector || selector.length === 0) { - return {}; - } - return { selector }; -} - -/** Split a combined `file:selector` path into file path and chunk selector. */ -export function parseChunkEditPath(editPath: string | undefined): { filePath: string; selector?: string } { - if (!editPath) return { filePath: "" }; - const colonIndex = chunkReadPathSeparatorIndex(editPath); - if (colonIndex === -1) { - return { filePath: editPath }; - } - const sel = editPath.slice(colonIndex + 1) || undefined; - return { filePath: editPath.slice(0, colonIndex), selector: sel }; -} - -export function parseChunkReadPath(readPath: string): ParsedChunkReadPath { - const colonIndex = chunkReadPathSeparatorIndex(readPath); - if (colonIndex === -1) { - return { filePath: readPath }; - } - const parsedSelector = parseChunkSelector(readPath.slice(colonIndex + 1) || undefined); - return { - filePath: readPath.slice(0, colonIndex), - selector: parsedSelector.selector, - }; -} - -export function isChunkReadablePath(readPath: string): boolean { - return parseChunkReadPath(readPath).selector !== undefined; -} - -export async function loadChunkStateForFile(filePath: string, language: string | undefined): Promise { - const file = Bun.file(filePath); - const stat = await file.stat(); - const cached = chunkStateCache.get(filePath); - if (cached && cached.mtimeMs === stat.mtimeMs && cached.size === stat.size) { - return cached; - } - - const source = normalizeChunkSource(await file.text()); - const state = ChunkState.parse(source, normalizeLanguage(language)); - const entry = { mtimeMs: stat.mtimeMs, size: stat.size, source, state }; - chunkStateCache.set(filePath, entry); - return entry; -} - -export async function formatChunkedRead(params: { - filePath: string; - readPath: string; - cwd: string; - language?: string; - omitChecksum?: boolean; - anchorStyle?: ChunkAnchorStyle; - absoluteLineRange?: { startLine: number; endLine?: number }; -}): Promise<{ text: string; resolvedPath?: string; chunk?: ChunkReadTarget }> { - const { filePath, readPath, cwd, language, omitChecksum = false, anchorStyle, absoluteLineRange } = params; - const normalizedLanguage = normalizeLanguage(language); - const { state } = await loadChunkStateForFile(filePath, normalizedLanguage); - const displayPath = displayPathForFile(filePath, cwd); - const renderIndentOptions = getChunkRenderIndentOptions(); - const result = state.renderRead({ - readPath, - displayPath, - languageTag: fileLanguageTag(filePath, normalizedLanguage), - omitChecksum, - anchorStyle, - absoluteLineRange: absoluteLineRange - ? { startLine: absoluteLineRange.startLine, endLine: absoluteLineRange.endLine ?? absoluteLineRange.startLine } - : undefined, - tabReplacement: renderIndentOptions.tabReplacement, - normalizeIndent: renderIndentOptions.normalizeIndent, - }); - return { text: result.text, resolvedPath: filePath, chunk: result.chunk }; -} - -export type ChunkedGrepMatch = { - displayPath: string; - fileLineCount: number; - chunkPath?: string; - chunkChecksum?: string; - lineNumber: number; - line: string; -}; - -export async function describeChunkedGrepMatch(params: { - filePath: string; - lineNumber: number; - line: string; - cwd: string; - language?: string; -}): Promise { - const { filePath, lineNumber, line, cwd, language } = params; - const { state } = await loadChunkStateForFile(filePath, language); - const chunkPath = state.lineToContainingChunkPath(lineNumber) || undefined; - const chunkInfo = chunkPath ? state.chunk(chunkPath) : null; - return { - displayPath: displayPathForFile(filePath, cwd), - fileLineCount: state.lineCount, - chunkPath, - chunkChecksum: chunkInfo?.checksum, - lineNumber, - line, - }; -} - -const CHUNK_CHECKSUM_BIGRAMS = new Set(HASHLINE_BIGRAMS); -type NativeChunkRegion = "head" | "body"; - -function isChunkChecksumToken(value: string): boolean { - if (value.length !== 4) return false; - const lower = value.toLowerCase(); - return CHUNK_CHECKSUM_BIGRAMS.has(lower.slice(0, 2)) && CHUNK_CHECKSUM_BIGRAMS.has(lower.slice(2, 4)); -} - -function parseChunkEditSelector(selector: string | undefined): { - selector?: string; - crc?: string; - region?: NativeChunkRegion; -} { - if (!selector) { - return {}; - } - - let trimmed = selector.trim(); - if (trimmed.length === 0) { - return {}; - } - - let region: NativeChunkRegion | undefined; - const suffix = trimmed.at(-1); - if (suffix === "~" || suffix === "^") { - region = suffix === "~" ? "body" : "head"; - trimmed = trimmed.slice(0, -1).trimEnd(); - } - - let selectorPart = trimmed; - let crc: string | undefined; - const hashIndex = selectorPart.lastIndexOf("#"); - if (hashIndex >= 0) { - const suffix = selectorPart.slice(hashIndex + 1).trim(); - if (isChunkChecksumToken(suffix)) { - crc = suffix.toLowerCase(); - selectorPart = selectorPart.slice(0, hashIndex).trimEnd(); - } - } else if (isChunkChecksumToken(selectorPart)) { - crc = selectorPart.toLowerCase(); - selectorPart = ""; - } - - return { selector: selectorPart || undefined, crc, region }; -} - -type NativeChunkRegionEncoding = "named" | "symbolic"; - -function toNativeEditRegion( - region: NativeChunkRegion | undefined, - encoding: NativeChunkRegionEncoding, -): NativeEditOperation["region"] | undefined { - if (!region) { - return undefined; - } - if (encoding === "symbolic") { - return region === "body" ? ChunkRegion.Body : ChunkRegion.Head; - } - return region as unknown as NativeEditOperation["region"] | undefined; -} - -function toNativeEditOperation( - operation: ChunkEditOperation, - defaultRegion: NativeChunkRegion | undefined, - encoding: NativeChunkRegionEncoding, -): NativeEditOperation { - const { selector, crc, region } = parseChunkEditSelector(operation.sel); - const nativeRegion = toNativeEditRegion(operation.sel === undefined ? (region ?? defaultRegion) : region, encoding); - switch (operation.op) { - case "put": - return { - op: ChunkEditOp.Put, - sel: selector, - crc, - region: nativeRegion, - content: operation.content, - }; - case "before": - return { op: ChunkEditOp.Before, sel: selector, crc, region: nativeRegion, content: operation.content }; - case "after": - return { op: ChunkEditOp.After, sel: selector, crc, region: nativeRegion, content: operation.content }; - case "prepend": - return { op: ChunkEditOp.Prepend, sel: selector, crc, region: nativeRegion, content: operation.content }; - case "append": - return { op: ChunkEditOp.Append, sel: selector, crc, region: nativeRegion, content: operation.content }; - case "delete": - return { op: ChunkEditOp.Delete, sel: selector, crc, region: nativeRegion }; - default: { - const exhaustive: never = operation; - return exhaustive; - } - } -} - -function buildNativeChunkEditRequest( - params: { defaultSelector?: string; defaultCrc?: string; operations: ChunkEditOperation[] }, - encoding: NativeChunkRegionEncoding, -): Pick[0], "operations" | "defaultSelector" | "defaultCrc"> { - const parsedDefaultSelector = parseChunkEditSelector(params.defaultSelector); - const operations = params.operations.map(operation => - toNativeEditOperation(operation, parsedDefaultSelector.region, encoding), - ); - return { - operations, - defaultSelector: parsedDefaultSelector.selector, - defaultCrc: params.defaultCrc ?? parsedDefaultSelector.crc, - }; -} - -function isChunkRegionEncodingError(error: unknown): error is Error { - return ( - error instanceof Error && - /value `"(body|head|~|\^)"` does not match any variant of enum `ChunkRegion`/.test(error.message) - ); -} - -export function applyChunkEdits(params: { - source: string; - language?: string; - cwd: string; - filePath: string; - operations: ChunkEditOperation[]; - defaultSelector?: string; - defaultCrc?: string; - anchorStyle?: ChunkAnchorStyle; -}): ChunkEditResult { - const normalizedSource = normalizeChunkSource(params.source); - const applyNativeEdits = (encoding: NativeChunkRegionEncoding): ChunkEditResult => { - const request = buildNativeChunkEditRequest(params, encoding); - const state = ChunkState.parse(normalizedSource, normalizeLanguage(params.language)); - return buildChunkEditResult( - state.applyEdits({ - operations: request.operations, - normalizeIndent: resolveChunkAutoIndent(), - defaultSelector: request.defaultSelector, - defaultCrc: request.defaultCrc, - anchorStyle: params.anchorStyle, - cwd: params.cwd, - filePath: params.filePath, - }), - ); - }; - - try { - return applyNativeEdits("named"); - } catch (error) { - if (isChunkRegionEncodingError(error)) { - try { - return applyNativeEdits("symbolic"); - } catch (fallbackError) { - if (fallbackError instanceof Error) { - throw new Error(normalizeChunkRegionSyntax(fallbackError.message)); - } - throw fallbackError; - } - } - if (error instanceof Error) { - throw new Error(normalizeChunkRegionSyntax(error.message)); - } - throw error; - } -} - -export async function getChunkInfoForFile( - filePath: string, - language: string | undefined, - chunkPath: string, -): Promise { - const { state } = await loadChunkStateForFile(filePath, language); - return state.chunk(chunkPath) ?? undefined; -} - -export function missingChunkReadTarget(selector: string): ChunkReadTarget { - return { status: ChunkReadStatus.NotFound, selector }; -} - -export const chunkToolEditSchema = Type.Object( - { - path: Type.Optional( - Type.String({ - description: "File path with chunk selector. Examples: 'src/app.ts:fn_foo#thth~', 'src/app.ts:class_Bar'.", - }), - ), - write: Type.Optional( - Type.Union([Type.String(), Type.Null()], { - description: - "Write complete new content to the targeted region. Null is rejected; use delete: true for deletion.", - }), - ), - delete: Type.Optional( - Type.Boolean({ - description: "Explicitly delete the targeted chunk. Must be true; include the current chunk ID.", - }), - ), - insert: Type.Optional( - Type.Object( - { - loc: StringEnum(["append", "prepend"] as const), - body: Type.String({ description: "Content to insert." }), - }, - { description: "Insert content relative to the chunk." }, - ), - ), - }, - { additionalProperties: false }, -); -export const chunkEditParamsSchema = Type.Object( - { - path: Type.Optional(Type.String({ description: "Default file path used when an edit omits its own `path`" })), - edits: Type.Array(chunkToolEditSchema, { - description: "Chunk edits", - minItems: 1, - }), - }, - { additionalProperties: false }, -); - -export type ChunkToolEdit = Static; -export type ChunkParams = Static; - -export interface ExecuteChunkSingleOptions { - session: ToolSession; - path: string; - edits: ChunkToolEdit[]; - signal?: AbortSignal; - batchRequest?: LspBatchRequest; - writethrough: WritethroughCallback; - beginDeferredDiagnosticsForPath: (path: string) => WritethroughDeferredHandle; -} - -/** Auto-correct indentation for content targeting a body region (`~`) when autoIndent is on. - * Handles two patterns: - * 1. Tab-based over-indentation: models include the function's base \t indent. - * 2. Space-based indentation: models use literal spaces instead of \t. - * Returns the corrected content and any warnings. */ -function autoCorrectBodyIndent(content: string, index: number): { content: string; warnings: string[] } { - const warnings: string[] = []; - if (!content || !resolveChunkAutoIndent()) return { content, warnings }; - const lines = content.split("\n"); - const nonEmpty = lines.filter(l => l.length > 0); - if (nonEmpty.length <= 1) return { content, warnings }; - - // 1. Tab-based over-indentation: strip common leading tabs. - const minTabs = Math.min(...nonEmpty.map(l => l.match(/^\t*/)?.[0].length ?? 0)); - if (minTabs >= 1) { - const fixed = lines.map(l => (l.length === 0 ? l : l.slice(minTabs))).join("\n"); - warnings.push( - `Edit ${index + 1}: auto-corrected body indentation \u2014 stripped ${minTabs} leading tab(s). When writing to \`~\`, write at column 0; the tool adds the function's base indent.`, - ); - return { content: fixed, warnings }; - } - - // 2. Space-based indentation: strip common leading spaces and convert to tabs. - const spaceIndents = nonEmpty.map(l => l.match(/^ */)?.[0].length ?? 0); - const minSpaces = Math.min(...spaceIndents); - if (minSpaces >= 2) { - const indentDiffs = spaceIndents.map(s => s - minSpaces).filter(d => d > 0); - const indentUnit = indentDiffs.length > 0 ? Math.min(...indentDiffs) : 4; - const unit = indentUnit >= 2 && indentUnit <= 8 ? indentUnit : 4; - const fixed = lines - .map(line => { - if (line.length === 0) return line; - const stripped = line.slice(minSpaces); - const leadingSpaces = stripped.match(/^ */)?.[0].length ?? 0; - const tabs = Math.floor(leadingSpaces / unit); - const rem = leadingSpaces % unit; - return "\t".repeat(tabs) + " ".repeat(rem) + stripped.slice(leadingSpaces); - }) - .join("\n"); - warnings.push( - `Edit ${index + 1}: auto-converted space indentation to tabs \u2014 stripped ${minSpaces} common leading spaces and converted ${unit}-space indent to tabs. When auto-indent is on, use \\t for indentation.`, - ); - return { content: fixed, warnings }; - } - - return { content, warnings }; -} - -function chunkEditOperationFields(edit: ChunkToolEdit): string[] { - const fields: string[] = []; - if (edit.write !== undefined) fields.push("write"); - if (edit.insert != null) fields.push("insert"); - if (edit.delete === true) fields.push("delete"); - return fields; -} - -function assertSingleChunkOperation(edit: ChunkToolEdit, index: number): string { - const fields = chunkEditOperationFields(edit); - if (fields.length === 0) { - throw new Error( - `Edit ${index + 1}: no operation specified. Use write:"..." to replace, insert:{loc,body} to insert, or delete:true to delete. Use the open tool to inspect chunks.`, - ); - } - if (fields.length > 1) { - throw new Error( - `Edit ${index + 1}: multiple operation fields set (${fields.join(", ")}). Each chunk edit entry must have exactly one operation.`, - ); - } - return fields[0]; -} - -function normalizeChunkEditOperations(edits: ChunkToolEdit[]): { - operations: ChunkEditOperation[]; - warnings: string[]; -} { - const warnings: string[] = []; - const operations = edits.map((edit, index): ChunkEditOperation => { - const { selector } = parseChunkEditPath(edit.path); - const operation = assertSingleChunkOperation(edit, index); - if (operation === "write") { - if (edit.write === null) { - throw new Error( - `Edit ${index + 1}: write:null no longer deletes chunks. Use delete:true to delete, or open the chunk to inspect its content without modifying the file.`, - ); - } - if (typeof edit.write !== "string") { - throw new Error(`Edit ${index + 1}: write must be a string.`); - } - if (edit.write.length === 0) { - throw new Error( - `Edit ${index + 1}: write:"" is a destructive empty replacement. Use delete:true to delete the chunk, or open the chunk to inspect its content without modifying the file.`, - ); - } - let writeContent = edit.write; - if (selector?.endsWith("~")) { - const corrected = autoCorrectBodyIndent(writeContent, index); - writeContent = corrected.content; - warnings.push(...corrected.warnings); - } - return { op: "put", sel: selector, content: writeContent }; - } - if (operation === "insert") { - if (edit.insert == null || typeof edit.insert.body !== "string" || edit.insert.body.length === 0) { - throw new Error(`Edit ${index + 1}: insert.body must be a non-empty string.`); - } - const op = edit.insert.loc === "prepend" ? "before" : "after"; - let insertContent = edit.insert.body; - if (selector?.endsWith("~")) { - const corrected = autoCorrectBodyIndent(insertContent, index); - insertContent = corrected.content; - warnings.push(...corrected.warnings); - } - return { op, sel: selector, content: insertContent }; - } - if (operation !== "delete") { - throw new Error(`Edit ${index + 1}: unsupported chunk edit operation "${operation}".`); - } - return { op: "delete", sel: selector }; - }); - return { operations, warnings }; -} - -async function writeChunkResult(params: { - result: ChunkEditResult; - resolvedPath: string; - sourceFile: BunFile; - sourceText: string; - sourceExists: boolean; - signal?: AbortSignal; - batchRequest?: LspBatchRequest; - writethrough: WritethroughCallback; - beginDeferredDiagnosticsForPath: (path: string) => WritethroughDeferredHandle; -}): Promise> { - const { - result, - resolvedPath, - sourceFile, - sourceText, - sourceExists, - signal, - batchRequest, - writethrough, - beginDeferredDiagnosticsForPath, - } = params; - - const { bom, text } = stripBom(sourceText); - const originalEnding = detectLineEnding(text); - const finalContent = bom + restoreLineEndings(result.diffSourceAfter, originalEnding); - const diagnostics = await writethrough(resolvedPath, finalContent, signal, sourceFile, batchRequest, dst => - dst === resolvedPath ? beginDeferredDiagnosticsForPath(resolvedPath) : undefined, - ); - invalidateFsScanAfterWrite(resolvedPath); - - const diffResult = generateUnifiedDiffString(result.diffSourceBefore, result.diffSourceAfter); - const warningsBlock = result.warnings.length > 0 ? `\n\nWarnings:\n${result.warnings.join("\n")}` : ""; - const meta = outputMeta() - .diagnostics(diagnostics?.summary ?? "", diagnostics?.messages ?? []) - .get(); - - return { - content: [{ type: "text", text: `${result.responseText}${warningsBlock}` }], - details: { - diff: diffResult.diff, - firstChangedLine: diffResult.firstChangedLine, - diagnostics, - op: sourceExists ? "update" : "create", - meta, - }, - }; -} - -export async function executeChunkSingle( - options: ExecuteChunkSingleOptions, -): Promise> { - const { session, path, edits, signal, batchRequest, writethrough, beginDeferredDiagnosticsForPath } = options; - const { resolvedPath, sourceFile, sourceExists, rawContent, chunkLanguage } = await resolveChunkSourceContext( - session, - path, - { intent: "write" }, - ); - const parentDir = nodePath.dirname(resolvedPath); - if (parentDir && parentDir !== ".") { - await fs.mkdir(parentDir, { recursive: true }); - } - const { operations: normalizedOperations, warnings: normWarnings } = normalizeChunkEditOperations(edits); - - if (!sourceExists && normalizedOperations.some(op => op.sel)) { - throw new Error( - `File does not exist: ${path}. Cannot resolve chunk selectors on a non-existent file. Use the write tool to create a new file, or check the path for typos.`, - ); - } - - const chunkResult = applyChunkEdits({ - source: rawContent, - language: chunkLanguage, - cwd: session.cwd, - filePath: resolvedPath, - operations: normalizedOperations, - anchorStyle: resolveAnchorStyle(session.settings), - }); - chunkResult.warnings.push(...normWarnings); - - if (!chunkResult.changed) { - const warningsBlock = chunkResult.warnings.length > 0 ? `\n\nWarnings:\n${chunkResult.warnings.join("\n")}` : ""; - return { - content: [{ type: "text", text: `[No changes needed — content already matches.]${warningsBlock}` }], - details: { - diff: "", - op: sourceExists ? "update" : "create", - meta: outputMeta().get(), - }, - }; - } - - return writeChunkResult({ - result: chunkResult, - resolvedPath, - sourceFile, - sourceText: rawContent, - sourceExists, - signal, - batchRequest, - writethrough, - beginDeferredDiagnosticsForPath, - }); -} diff --git a/packages/coding-agent/src/edit/renderer.ts b/packages/coding-agent/src/edit/renderer.ts index d64bc9b2e..2542a06ab 100644 --- a/packages/coding-agent/src/edit/renderer.ts +++ b/packages/coding-agent/src/edit/renderer.ts @@ -29,7 +29,6 @@ import type { VimToolDetails } from "../vim/types"; import type { DiffError, DiffResult } from "./diff"; import { expandApplyPatchToEntries, expandApplyPatchToPreviewEntries } from "./modes/apply-patch"; import type { Operation, PatchEditEntry } from "./modes/patch"; -import type { PerFileDiffPreview } from "./streaming"; // ═══════════════════════════════════════════════════════════════════════════ // LSP Batching @@ -94,7 +93,7 @@ interface EditRenderArgs { */ previewDiff?: string; __partialJson?: string; - // Hashline / chunk mode fields + // Hashline mode fields edits?: EditRenderEntry[]; } @@ -141,8 +140,6 @@ export interface EditRenderContext { editMode?: EditMode; /** Pre-computed diff preview (computed before tool executes) */ editDiffPreview?: DiffResult | DiffError; - /** Multi-file streaming diff preview (chunk edits spanning several files) */ - perFileDiffPreview?: PerFileDiffPreview[]; /** Function to render diff text with syntax highlighting */ renderDiff?: (diffText: string, options?: { filePath?: string }) => string; } @@ -151,11 +148,9 @@ const EDIT_STREAMING_PREVIEW_LINES = 12; const CALL_TEXT_PREVIEW_LINES = 6; const CALL_TEXT_PREVIEW_WIDTH = 80; -/** Extract file path from an edit entry's path (handles chunk's file:selector format). */ +/** Extract file path from an edit entry. */ function filePathFromEditEntry(p: string | undefined): string | undefined { - if (!p) return undefined; - const ci = /^[a-zA-Z]:[/\\]/.test(p) ? p.indexOf(":", 2) : p.indexOf(":"); - return ci === -1 ? p : p.slice(0, ci); + return p ?? undefined; } function decodePartialJsonStringFragment(fragment: string): string { @@ -280,32 +275,7 @@ function formatMetadataLine(lineCount: number | null, language: string | undefin return uiTheme.fg("dim", `${icon}`); } -function formatMultiFileStreamingDiff(previews: PerFileDiffPreview[], uiTheme: Theme): string { - const parts: string[] = []; - for (const preview of previews) { - if (!preview.diff && !preview.error) continue; - const header = uiTheme.fg("dim", `\n\n── ${shortenPath(preview.path)} ──`); - if (preview.error) { - parts.push(`${header}\n${uiTheme.fg("error", replaceTabs(preview.error))}`); - continue; - } - if (preview.diff) { - parts.push(`${header}${formatStreamingDiff(preview.diff, preview.path, uiTheme, "preview")}`); - } - } - return parts.join(""); -} - -function getCallPreview( - args: EditRenderArgs, - rawPath: string, - uiTheme: Theme, - renderContext: EditRenderContext | undefined, -): string { - const multi = renderContext?.perFileDiffPreview; - if (multi && multi.length > 0 && multi.some(p => p.diff || p.error)) { - return formatMultiFileStreamingDiff(multi, uiTheme); - } +function getCallPreview(args: EditRenderArgs, rawPath: string, uiTheme: Theme): string { if (args.previewDiff) { return formatStreamingDiff(args.previewDiff, rawPath, uiTheme, "preview"); } @@ -438,7 +408,7 @@ export const editToolRenderer = { if (fileCount > 1) { text += uiTheme.fg("dim", ` (+${fileCount - 1} more)`); } - text += getCallPreview(editArgs, rawPath, uiTheme, renderContext); + text += getCallPreview(editArgs, rawPath, uiTheme); if (applyPatchSummary?.error) { text += `\n\n${uiTheme.fg("error", truncateToWidth(replaceTabs(applyPatchSummary.error), CALL_TEXT_PREVIEW_WIDTH))}`; } diff --git a/packages/coding-agent/src/edit/streaming.ts b/packages/coding-agent/src/edit/streaming.ts index 1d5296590..a0ea59074 100644 --- a/packages/coding-agent/src/edit/streaming.ts +++ b/packages/coding-agent/src/edit/streaming.ts @@ -16,7 +16,6 @@ import type { Theme } from "../modes/theme/theme"; import { type EditMode, resolveEditMode } from "../utils/edit-mode"; import { computeEditDiff, type DiffError, type DiffResult } from "./diff"; import { expandApplyPatchToEntries, expandApplyPatchToPreviewEntries } from "./modes/apply-patch"; -import { type ChunkToolEdit, computeChunkDiff, parseChunkEditPath } from "./modes/chunk"; import { computeHashlineDiff, type HashlineToolEdit } from "./modes/hashline"; import { computePatchDiff, type PatchEditEntry } from "./modes/patch"; import type { ReplaceEditEntry } from "./modes/replace"; @@ -226,70 +225,6 @@ const hashlineStrategy: EditStreamingStrategy = { }, }; -interface ChunkArgs { - path?: string; - edits?: ChunkToolEdit[]; - __partialJson?: string; -} - -const chunkStrategy: EditStreamingStrategy = { - extractCompleteEdits(args, partialJson) { - if (!args?.edits) return args; - let edits = dropIncompleteLastEdit(args.edits, partialJson, "edits"); - // Extra guard: if partial JSON still contains `":nu` / `":nul` (partial - // `null` literals), `partial-json` may have already surfaced the last - // entry with `write === null`. When that entry's `}` hasn't closed - // yet, it has already been dropped above. But if dropping was not - // triggered (e.g. list still open and no new `{` after), also drop the - // trailing null-write entry so the preview does not flicker with an - // error for an incomplete string/null literal. - if (partialJson && edits.length > 0) { - const last = edits[edits.length - 1] as Partial | undefined; - const endsInPartialNull = /:\s*nu?l?\s*$/.test(partialJson.trimEnd()); - if (last && endsInPartialNull && last.write === null) { - edits = edits.slice(0, -1); - } - } - return { ...args, edits }; - }, - async computeDiffPreview(args, ctx) { - const edits = args.edits ?? []; - if (edits.length === 0) return null; - // Group edits by file path - const groups = new Map(); - const fileOrder: string[] = []; - for (const edit of edits) { - if (!edit) continue; - const editPath = edit.path ?? args.path; - if (!editPath) continue; - const { filePath } = parseChunkEditPath(editPath); - if (!filePath) continue; - let bucket = groups.get(filePath); - if (!bucket) { - bucket = []; - groups.set(filePath, bucket); - fileOrder.push(filePath); - } - bucket.push({ ...edit, path: editPath }); - } - if (fileOrder.length === 0) return null; - - const MAX_FILES = 5; - const selected = fileOrder.slice(0, MAX_FILES); - const previews: PerFileDiffPreview[] = []; - for (const filePath of selected) { - ctx.signal.throwIfAborted(); - const fileEdits = groups.get(filePath) ?? []; - const result = await computeChunkDiff({ path: filePath, edits: fileEdits }, ctx.cwd, { signal: ctx.signal }); - previews.push(toPerFilePreview(filePath, result)); - } - return previews; - }, - renderStreamingFallback() { - return ""; - }, -}; - interface ApplyPatchArgs { input?: string; } @@ -365,7 +300,6 @@ export const EDIT_MODE_STRATEGIES: Record, patch: patchStrategy as EditStreamingStrategy, hashline: hashlineStrategy as EditStreamingStrategy, - chunk: chunkStrategy as EditStreamingStrategy, apply_patch: applyPatchStrategy as EditStreamingStrategy, vim: vimStrategy, atom: atomStrategy as EditStreamingStrategy, diff --git a/packages/coding-agent/src/modes/components/settings-defs.ts b/packages/coding-agent/src/modes/components/settings-defs.ts index f92155a05..f8cafb7f0 100644 --- a/packages/coding-agent/src/modes/components/settings-defs.ts +++ b/packages/coding-agent/src/modes/components/settings-defs.ts @@ -298,11 +298,6 @@ const OPTION_PROVIDERS: Partial> = { { value: "1000", label: "1000 lines" }, { value: "5000", label: "5000 lines" }, ], - "read.anchorstyle": [ - { value: "full", label: "Full", description: "Show the kind prefix and identifier" }, - { value: "kind", label: "Kind", description: "Show only the kind prefix plus checksum" }, - { value: "bare", label: "Bare", description: "Show only the checksum" }, - ], // Todo auto-clear delay "tasks.todoClearDelay": [ { value: "0", label: "Instant" }, diff --git a/packages/coding-agent/src/modes/components/tool-execution.ts b/packages/coding-agent/src/modes/components/tool-execution.ts index b6ef5a29d..61134db6b 100644 --- a/packages/coding-agent/src/modes/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/components/tool-execution.ts @@ -109,7 +109,7 @@ export class ToolExecutionComponent extends Container { isError?: boolean; details?: any; }; - // Edit preview state (single-file for legacy modes, multi-file for chunk) + // Edit preview state #editMode?: EditMode; #editDiffPreview?: PerFileDiffPreview[]; #editDiffScheduleTimer?: NodeJS.Timeout; @@ -639,10 +639,7 @@ export class ToolExecutionComponent extends Container { return this.#args; } // Single-file previews feed the existing `previewDiff` channel consumed - // by `formatStreamingDiff` in the renderer. Multi-file previews are - // piped via `renderContext.perFileDiffPreview`, so the args we hand to - // `renderCall` only need the first file's diff to preserve prior - // single-file behavior. + // by `formatStreamingDiff` in the renderer. const first = previews[0]; if (!first?.diff) { return this.#args; @@ -687,9 +684,6 @@ export class ToolExecutionComponent extends Container { ? { error: first.error } : { diff: first.diff ?? "", firstChangedLine: first.firstChangedLine }; } - if (previews.length > 1) { - context.perFileDiffPreview = previews; - } } context.renderDiff = renderDiff; } diff --git a/packages/coding-agent/src/prompts/tools/chunk-edit.md b/packages/coding-agent/src/prompts/tools/chunk-edit.md deleted file mode 100644 index 5ca211a2a..000000000 --- a/packages/coding-agent/src/prompts/tools/chunk-edit.md +++ /dev/null @@ -1,158 +0,0 @@ -Edits files via syntax-aware chunks. Use `read(path="file.ts")` to read and discover chunks before editing. -- `read` is the canonical read path for chunk source and `sel="?"` tree listings. -- `write` rewrites the entire targeted region — best for most edits. -- `insert` adds content before/after a chunk. -- `delete` deletes a targeted chunk and must be explicit. - -Call format: `{"edits": [{"path": "file:chunk#ID~", "write": "new body"}, …]}` - - -- **MUST** inspect first with `read`. Never invent chunk paths or IDs. Copy them from the latest `read` output or edit response. -- `path` format: `file:selector` — e.g. `src/app.ts:fn_foo#thth~`. Append `~` for body, `^` for head, or nothing for the whole chunk. Include `#ID` for `write`/`delete`. -- If the exact chunk path is unclear, run `read(path="file", sel="?")` and copy a selector from that listing. -{{#if chunkAutoIndent}} -- Use `\t` for indentation in `content`. Write content at indent-level 0 — the tool re-indents it to match the chunk's position in the file. For example, to replace `~` of a method, write the body starting at column 0: - ``` - content: "if (x) {\n\treturn true;\n}" - ``` - The tool adds the correct base indent automatically. Never manually pad with the chunk's own indentation. - Multiple sibling body lines at the same level all start at column 0: `"print(a)\nprint(b)\nprint(c)\n"`. Only use `\t` when nesting deeper (e.g. `"if cond:\n\tinner\nouter\n"`). - Before applying the target's base indent, the tool strips any common leading whitespace shared by all non-empty `write` lines as a safety net. Do not rely on that cleanup for mixed indentation; write `~` bodies at column 0 and use one `\t` per relative nesting level. - Multi-line replacements use the same relative-indentation model: the replacement text is dedented, then re-indented to the matched source line. Do not include the chunk's base indentation in replacement text. - **Common mistake** when replacing `~` of a function body: do NOT include the function's own indentation. - Wrong: `"if b == 0:\n\t\treturn None\n\treturn a / b\n"` — adds the function's base `\t` to every line. - Correct: `"if b == 0:\n\treturn None\nreturn a / b\n"` — `if` and `return a / b` at column 0, only `return None` gets `\t` for nesting. -{{else}} -- Match the file's literal tabs/spaces in `content`. Do not convert indentation to canonical `\t`. -- Write content at indent-level 0 relative to the target region. For example, to replace `~` of a method, write: - ``` - content: "if (x) {\n return true;\n}" - ``` - The tool adds the correct base indent automatically, then preserves the tabs/spaces you used inside the snippet. Never manually pad with the chunk's own indentation. - Before applying the target's base indent, the tool strips any common leading whitespace shared by all non-empty `write` lines as a safety net. Do not rely on that cleanup for mixed indentation; write `~` bodies at column 0. - Multi-line replacements use the same relative-indentation model: the replacement text is dedented, then re-indented to the matched source line. Do not include the chunk's base indentation in replacement text. -{{/if}} -- Region suffixes only apply to chunks with a real head/body boundary (classes, functions, impl blocks, and similar containers). On code leaf chunks (enum variants, fields, single statements, and compound statements like `if`/`for`/`while`/`match`/`try`), `~` and `^` are rejected. Use the unsuffixed selector and supply the complete replacement content, or edit the parent container's `~` body. -- Unsuffixed `write` on a leaf chunk uses your content verbatim after normal replacement; it is not a body-region rewrite. Include the exact indentation and punctuation the leaf needs in the file. -- `^` head writes and `~` body writes use the same base-indent model: write content at column 0 relative to the target region, and the tool applies the chunk's file indentation. -- `write` and `delete` require the current ID. `prepend`/`append` do not. -- **IDs change after every edit.** The edit response always carries the new IDs — use those for the next call or run `read(path="file", sel="?")` to refresh. Never reuse an ID from before the latest edit. -- Same-file edit batches are transactional: if any operation in that file fails, no changes from that file's batch are saved. Multi-file edit calls run per file, so a later file error does not roll back earlier files that already succeeded. - - - -You **MUST** use the narrowest region that covers your change. Putting without a region overwrites the **entire chunk including leading comments, decorators, and attributes** — omitting them from `content` deletes them. - -**`put` is total, not surgical.** The `content` you supply becomes the *complete* new content for the targeted region. Everything in the original region that you omit from `content` is deleted. Before using `put` on any chunk's `~`, verify the chunk does not contain children you intend to keep. If a chunk spans hundreds of lines and your change touches only a few, target a specific child chunk — not the parent. - -**Group chunks (`stmts_*`, `imports_*`, `decls_*`) are containers.** They hold many sibling items (test functions, import statements, declarations). `put` on a group chunk's `~` overwrites **all** of its children. To edit one item inside a group, target that item's own chunk path. If no child chunk exists, use the specific child's chunk selector from `read` output — do not `put` the parent group. - - - -In `read` output, lines marked `^` between the line number and `|` are **head** lines (doc comments, attributes/decorators, signature). Lines without `^` are **body** lines. Use this to decide which region to target: -- `fn_foo#ID~` — **body only (the default choice for most edits).** Head lines (`^`) are preserved automatically — doc comments, attributes, and signature stay untouched. On code leaf chunks, this is rejected because there is no safe body boundary. -- `fn_foo#ID^` — head only (decorators, attributes, doc comments, signature, opening delimiter). Body stays untouched. -- `fn_foo#ID` — entire chunk including leading trivia. **You must include doc comments and attributes in `content`; omitting them deletes them.** -- `chunk~` + `append`/`prepend` inserts *inside* the container. `chunk` + `append`/`prepend` inserts *outside*. Appending to a container without `~` emits a warning because it lands after the closing delimiter, not before it. - -**Note on leading trivia:** whether a decorator/doc comment belongs to `^` depends on the parser. In Rust and Python, attributes and decorators are attached to the function chunk, so `^` covers them. In TypeScript/JavaScript, a `@decorator` + `/** jsdoc */` block immediately above a method often surfaces as a **separate sibling chunk** (shown as `chunk#ID` in the `?` listing) rather than as part of the function's `^`. JSDoc directly above a plain function is more likely to be absorbed into that function's `^`. If you need to rewrite a decorated member, run `read(path="file", sel="?")` and check for a sibling `chunk#ID` directly above your target. - -**Python notes:** Python docstrings are body lines, not head lines. A `~` body write on a function that has a docstring deletes the docstring unless you include the docstring in `content`. Python enum members and nested functions/closures are often opaque inside their parent chunk and may not appear as addressable child chunks; rewrite the parent container body. Python decorated class/function `^` writes and Python `^` deletes are rejected because indentation-sensitive bodies can become attached to the wrong block while still parsing. - -**Note on non-code formats:** for prose and data formats (markdown, YAML, JSON, frontmatter), unsupported `^` and `~` suffixes warn and fall back to whole-chunk editing. Always replace the entire chunk and include any delimiter syntax (fence backticks, `---` frontmatter markers, list markers, table rows, headings) in your `content` — omitting them deletes them. For markdown sections (`sect_*`), prefer unsuffixed whole-chunk replace because `^`/`~` on prose sections can replace the heading and child content too; if you only need the heading, target the heading child chunk shown in `sel="?"`. Fenced code blocks with a declared language are parsed again and can expose inner chunks such as `code_py#ID.fn_gre#ID`; target those inner chunks when available. Markdown root writes preserve fenced code indentation verbatim. Recognized pipe tables expose `row_N` children for row-level edits; table cells and list items are not independently addressable, so rewrite the whole list/table chunk for those structural changes. Appending a table-row-shaped string (`| value |`) to a table chunk inserts it before the trailing blank-line separator so it remains part of the table. Otherwise read with `raw` first and preserve the exact whitespace inside fences. To insert content after a markdown section heading, use `after` on the heading chunk (`sect_*.chunk` or `sect_*.chunk_1`) — not `before`/`prepend` on the section itself, which lands physically before the heading and gets absorbed by the preceding section on reparse. - - - -Each edit entry has `path` (`file:selector`) plus **exactly one** operation field — `write`, `insert`, or `delete`. Never set more than one on the same entry. `write:null`, `write:""`, and bare `{path}` entries are rejected; they do not delete. - -|fields|path (selector part)|effect| -|---|---|---| -|`write: "content"`|`file:chunk#ID`, `file:chunk#ID~`, or `file:chunk#ID^`|write complete new content to the region| -|`delete: true`|`file:chunk#ID`|delete the chunk explicitly| -|`insert: {loc, body}`|`file:chunk` or `file:chunk~`|insert before/after the chunk (`loc`: `"prepend"` or `"append"`)| - - - -Given this `read` output for `counter.rs`: -``` - | counter.rs·62L·rust·#anth - | -@imp#erhe - 1 |use std::fmt; - | -@struct_Counte#onat - 3^|/// A simple counter that tracks a value and its history. - 4^|#[derive(Debug, Clone)] - 5^|pub struct Counter { --@struct_Counte.field_value#enth - 6 | /// The current value. - 7 | value: i32, --@struct_Counte.field_max#seti - 8 | /// Maximum allowed value. - 9 | max: i32, -10 |} - | -@impl_Counte#reha -12^|impl Counter { --@impl_Counte.fn_new#ndas -13^| /// Creates a new counter starting at zero. -14^| pub fn new(max: i32) -> Self { -15 | Self { value: 0, max } -16 | } -17 | --@impl_Counte.fn_increm#ouer -18^| /// Increments the counter by one, clamping at max. -19^| pub fn increment(&mut self) { -20 | if self.value < self.max { -21 | self.value += 1; -22 | } -23 | } -24 | --@impl_Counte.fn_decrem#arve -25^| /// Decrements the counter by one, clamping at zero. -26^| pub fn decrement(&mut self) { -27 | if self.value > 0 { -28 | self.value -= 1; -29 | } -30 | } -31 | --@impl_Counte.fn_get#arco -32^| /// Returns the current value. -33^| pub fn get(&self) -> i32 { -34 | self.value -35 | } -36 |} - | -@impl_Displa#meha -38^|impl fmt::Display for Counter { --@impl_Displa.fn_fmt#deri -39^| fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { -40 | write!(f, "Counter({}/{})", self.value, self.max) -41 | } -42 |} -``` -Lines marked `^` between the line number and `|` are **head** lines (doc comments, attributes, signature). Lines without `^` are **body** lines. `~` replaces body lines only; `^` replaces head lines only. - -# Put body (`~` — the common case) -`{ "path": "counter.rs:impl_Counte.fn_increm#ouer~", "write": "self.value = (self.value + 1).min(self.max);\n" }` -Only body changes; doc comment, signature, and closing `}` are preserved. -# Write whole chunk (rewrite signature + doc + body) -`{ "path": "counter.rs:impl_Counte.fn_increm#ouer", "write": "/// Increments by the given step, clamping at max.\npub fn increment(&mut self, step: i32) {\n\tself.value = (self.value + step).min(self.max);\n}\n" }` -Everything is rewritten. Omitting the doc comment or signature deletes them. -# Write head (`^` — attributes, doc comments, signature) -`{ "path": "counter.rs:impl_Counte.fn_get#arco^", "write": "/// Returns the current counter value.\n#[inline]\npub fn get(&self) → i32 {\n" }` -Head changes (all `^` lines + opening brace); body untouched. -# Insert before a chunk (`prepend`) -`{ "path": "counter.rs:impl_Counte.fn_get", "insert": { "loc": "prepend", "body": "/// Resets the counter to zero.\npub fn reset(&mut self) {\n\tself.value = 0;\n}\n\n" } }` -# Insert after a chunk (`append`) -`{ "path": "counter.rs:struct_Counte", "insert": { "loc": "append", "body": "\nimpl Default for Counter {\n\tfn default() → Self {\n\t\tSelf { value: 0, max: 100 }\n\t}\n}\n" } }` -# Insert at start of container body (`~` + `prepend`) -`{ "path": "counter.rs:impl_Counte~", "insert": { "loc": "prepend", "body": "/// Creates a counter starting at the given value.\npub fn with_value(value: i32, max: i32) → Self {\n\tSelf { value: value.min(max), max }\n}\n\n" } }` -Lands at the top of the impl body, before existing methods. -# Insert at end of container body (`~` + `append`) -`{ "path": "counter.rs:impl_Counte~", "insert": { "loc": "append", "body": "\n/// Returns true if the counter is at its maximum.\npub fn is_maxed(&self) → bool {\n\tself.value ≥ self.max\n}\n" } }` -Lands at the end of the impl body, before the closing `}`. -# Delete a chunk -`{ "path": "counter.rs:impl_Counte.fn_decrem#arve", "delete": true }` -Removes the method including its doc comment and signature. - diff --git a/packages/coding-agent/src/prompts/tools/grep.md b/packages/coding-agent/src/prompts/tools/grep.md index 05b072504..76f6d10c4 100644 --- a/packages/coding-agent/src/prompts/tools/grep.md +++ b/packages/coding-agent/src/prompts/tools/grep.md @@ -14,14 +14,11 @@ Searches files using powerful regex matching. - Text output is line-number-prefixed {{/if}} {{/if}} -{{#if IS_CHUNK_MODE}} -- Text output is chunk-path-prefixed: `path:sel>123|content` -{{/if}} - You **MUST** use the built-in Grep tool for any content search. Do **NOT** shell out to `grep`, `rg`, `ripgrep`, `ag`, `ack`, `git grep`, `awk`, `sed`-for-search, or any other CLI search via Bash — even for a single match, even "just to check quickly", even piped through other commands. -- Bash `grep`/`rg` returns raw text without chunk paths, loses `.gitignore` semantics, bypasses result limits, and wastes tokens. The Grep tool is faster, structured, and already wired into the workspace — there is no scenario where Bash search is preferable. +- Bash `grep`/`rg` loses `.gitignore` semantics, bypasses result limits, and wastes tokens. The Grep tool is faster, structured, and already wired into the workspace — there is no scenario where Bash search is preferable. - If you catch yourself typing `grep`, `rg`, or `| grep` in a Bash command, stop and re-issue the search through the Grep tool instead. - If the search is open-ended, requiring multiple rounds, you **MUST** use the Task tool with the explore subagent instead of chaining Grep calls yourself. diff --git a/packages/coding-agent/src/prompts/tools/poll.md b/packages/coding-agent/src/prompts/tools/poll.md index a852ddc2a..56bda283c 100644 --- a/packages/coding-agent/src/prompts/tools/poll.md +++ b/packages/coding-agent/src/prompts/tools/poll.md @@ -4,4 +4,4 @@ You **MUST** use the `poll` tool (in a loop, if necessary) instead of manually r If the timeout elapses before any job changes state, it returns the current snapshot (still-running jobs and any already-completed deliveries) without erroring — call `poll` again to keep waiting. -You **MUST NOT** poll the same job repeatedly without evidence of progress. Between calls, inspect `read jobs://` to confirm new output or activity. If a job is stalled, has hung, or is producing nothing useful, cancel it via `cancel_job` and try a different approach instead of waiting indefinitely. \ No newline at end of file +You **MUST NOT** poll the same job repeatedly without evidence of progress. Between calls, inspect `read jobs://` to confirm new output or activity. If a job is stalled, has hung, or is producing nothing useful, cancel it via `cancel_job` and try a different approach instead of waiting indefinitely. diff --git a/packages/coding-agent/src/prompts/tools/read-chunk.md b/packages/coding-agent/src/prompts/tools/read-chunk.md deleted file mode 100644 index 0488fbd7c..000000000 --- a/packages/coding-agent/src/prompts/tools/read-chunk.md +++ /dev/null @@ -1,73 +0,0 @@ -Reads files using syntax-aware chunks. Also inspects directories, archives, SQLite databases, images, documents (PDF/DOCX/PPTX/XLSX/RTF/EPUB/ipynb), **and URLs**. - - -The chunk-aware `read` variant returns AST-scoped chunks with current checksum IDs for structural editing, and otherwise behaves like `open` for non-code content. -- You **MUST** parallelize calls when exploring related files -- For URLs, `read` fetches the page and returns clean extracted text/markdown by default (reader-mode). It handles HTML pages, GitHub issues/PRs, Stack Overflow, Wikipedia, Reddit, NPM, arXiv, RSS/Atom, JSON endpoints, PDFs, etc. You **SHOULD** reach for `read` — not a browser/puppeteer tool — for fetching and inspecting web content. - -## Parameters -- `path` — file path or URL; may include `:selector` suffix (required) -- `sel` — optional selector for chunks, line ranges, listing, or raw mode -- `timeout` — seconds, for URLs only - -## Selectors - -|`sel` value|Behavior| -|---|---| -|*(omitted)*|Read full file as chunks (up to {{DEFAULT_LIMIT}} lines)| -|`class_Foo`|Read a specific chunk| -|`class_Foo.fn_bar#thth~`|Read a chunk region (body `~` / head `^`) by ID| -|`?`|List all chunk paths with IDs| -|`L50`|Read from line 50 onward (shorthand for L50 to EOF)| -|`L50-L120`|Read lines 50 through 120| -|`L20-L20`|Read exactly one line| -|`raw`|Raw content without transformations (for URLs: untouched HTML)| - -Max {{DEFAULT_MAX_LINES}} lines per call. - -# Chunks -Each anchor `@full.chunk.path#thth` (with `-` prefixes for nesting depth) in the output identifies a chunk. Use `full.chunk.path#thth` as-is to read truncated chunks. -If you need a canonical target list, run `read(path="file", sel="?")`. That listing shows chunk paths with IDs and is the safest structural discovery mode. Summary lines in this listing are orientation hints; follow a selector with `read(path="file", sel="chunk#ID")` or use `raw` when you need exact source. -Line numbers in the gutter are absolute file line numbers. - -{{#if chunkAutoIndent}} -Chunk reads normalize leading indentation so copied content round-trips cleanly into chunk edits. -{{else}} -Chunk reads preserve literal leading tabs/spaces from the file. When editing, keep the same whitespace characters you see here. -{{/if}} -`raw` shows the file's literal whitespace. Structured chunk views may normalize or display indentation for edit round-tripping, so use `raw` when exact tabs/spaces matter, especially inside markdown fenced code blocks. - -IDs change after every edit. Use the new IDs from the edit response or refresh with `sel="?"` before the next `write`/`delete`. `insert` selectors may omit IDs, but still prefer fresh paths after structural edits. - -Parser boundaries vary by language: TypeScript/JavaScript decorators and JSDoc above decorated methods may appear as sibling `chunk#ID` entries, Python decorators are part of the function/class head, Python docstrings are body lines, and Python enum members or nested closures may remain opaque inside their parent chunk. Decorated Python `^` writes and Python `^` deletes are rejected for safety. -Markdown sections, lists, and tables are structural chunks. Recognized pipe tables expose `row_N` children for row-level edits; list items and table cells are not independently addressable. Fenced code blocks with a declared language are parsed again when possible, so functions inside a markdown fence can appear as addressable nested chunks. - -Chunk trees: JS, TS, TSX, Python, Rust, Go. Others use blank-line fallback. -# Inspection -Extracts text from PDF, Word, PowerPoint, Excel, RTF, EPUB, and Jupyter notebook files. Can inspect images. - -# Directories & Archives -Directories and archive roots return a list of entries. Supports `.tar`, `.tar.gz`, `.tgz`, `.zip`. Use `archive.ext:path/inside/archive` to read contents. - -# SQLite Databases -When used against a SQLite database (`.sqlite`, `.sqlite3`, `.db`, `.db3`), returns structured database content. -- `file.db` — list tables with row counts -- `file.db:table` — table schema + sample rows -- `file.db:table:key` — single row by primary key -- `file.db:table?limit=50&offset=100` — paginated rows -- `file.db:table?where=status='active'&order=created:desc` — filtered rows -- `file.db?q=SELECT …` — read-only SELECT query - -# URLs -Extracts content from web pages, GitHub issues/PRs, Stack Overflow, Wikipedia, Reddit, NPM, arXiv, RSS/Atom feeds, JSON endpoints, PDFs at URLs, and similar text-based resources. Returns clean reader-mode text/markdown — no browser required. Use `sel="raw"` for untouched HTML; `timeout` to override the default request timeout. You **SHOULD** prefer `read` over a browser/puppeteer tool for fetching URL content; only use a browser when the page requires JS execution, authentication, or interactive actions (clicks, forms, scrolling). - - - -- You **MUST** `read` before editing — never invent chunk names or IDs. - - Chunk names are truncated (e.g., `handleRequest` becomes `fn_handleRequ`). Always copy chunk paths from `read` or `?` output — never construct them from source identifiers. -- You **MUST** use `read` (never bash `cat`/`head`/`tail`/`less`/`more`/`ls`/`tar`/`unzip`/`curl`/`wget`) for all file, directory, archive, and URL reads. -- You **MUST NOT** reach for a browser/puppeteer tool to fetch static web content — `read` handles HTML, PDFs, JSON, feeds, and docs directly. Reserve browser tools for JS-heavy pages or interactive flows. -- You **MUST** always include the `path` parameter; never call with `{}`. -- For specific line ranges, use `sel`: `read(path="file", sel="L50-L150")` — not `cat -n file | sed`. -- You **MAY** use `sel` with URL reads; the tool paginates cached fetched output. - diff --git a/packages/coding-agent/src/prompts/tools/read.md b/packages/coding-agent/src/prompts/tools/read.md index 9d77a552a..7fb6510fc 100644 --- a/packages/coding-agent/src/prompts/tools/read.md +++ b/packages/coding-agent/src/prompts/tools/read.md @@ -18,7 +18,7 @@ The `read` tool is multi-purpose and more capable than it looks — inspects fil |`L50`|Read from line 50 onward (shorthand for L50 to EOF)| |`L50-L120`|Read lines 50 through 120| |`L20-L20`|Read exactly one line| -|`raw`|Skip line-numbering / hashline / chunking; return file content as plain text. For URLs: untouched HTML.| +|`raw`|Skip line-numbering / hashline; return file content as plain text. For URLs: untouched HTML.| Max {{DEFAULT_MAX_LINES}} lines per call. diff --git a/packages/coding-agent/src/prompts/tools/todo-write.md b/packages/coding-agent/src/prompts/tools/todo-write.md index 0ae5c263c..56f2a4328 100644 --- a/packages/coding-agent/src/prompts/tools/todo-write.md +++ b/packages/coding-agent/src/prompts/tools/todo-write.md @@ -45,7 +45,7 @@ If `done`, `rm`, or `drop` omits both `task` and `phase`, it applies to all task ## Phase Anatomy - `name`: Short, human-readable noun phrase (1-3 words). Capitalize naturally. -- Always prefix with a roman-numeral ordinal (`I.`, `II.`, `III.`, `IV.`, ...) to convey ordering — e.g. `I. Foundation`, `II. Auth`, `III. Routing`. Single-phase plans use `I.` too. +- Always prefix with a roman-numeral ordinal (`I.`, `II.`, `III.`, `IV.`, …) to convey ordering — e.g. `I. Foundation`, `II. Auth`, `III. Routing`. Single-phase plans use `I.` too. - You **MUST NOT** use snake_case, `Phase1_*`, arabic numerals (`1.`), or letter prefixes (`A.`) — they render as ugly identifiers. ## Rules @@ -66,7 +66,7 @@ Create a todo list when: # Initial setup (multi-phase) `{"ops":[{"op":"replace","phases":[{"name":"I. Foundation","tasks":[{"content":"Scaffold crate"},{"content":"Wire workspace"}]},{"name":"II. Auth","tasks":[{"content":"Port credential store"},{"content":"Wire OAuth providers"}]},{"name":"III. Verification","tasks":[{"content":"Run cargo test"}]}]}]}` -# Initial setup (single phase " still prefixed) +# Initial setup (single phase — still prefixed) `{"ops":[{"op":"replace","phases":[{"name":"I. Implementation","tasks":[{"content":"Apply fix"},{"content":"Run tests"}]}]}]}` # Complete one task `{"ops":[{"op":"done","task":"task-2"}]}` diff --git a/packages/coding-agent/src/tools/fs-cache-invalidation.ts b/packages/coding-agent/src/tools/fs-cache-invalidation.ts index 13f541e48..91fd88d3a 100644 --- a/packages/coding-agent/src/tools/fs-cache-invalidation.ts +++ b/packages/coding-agent/src/tools/fs-cache-invalidation.ts @@ -1,12 +1,10 @@ import { invalidateFsScanCache } from "@oh-my-pi/pi-natives"; -import { invalidateChunkCache } from "../edit/modes/chunk"; /** * Invalidate shared filesystem scan caches after a content write/update. */ export function invalidateFsScanAfterWrite(path: string): void { invalidateFsScanCache(path); - invalidateChunkCache(path); } /** @@ -14,7 +12,6 @@ export function invalidateFsScanAfterWrite(path: string): void { */ export function invalidateFsScanAfterDelete(path: string): void { invalidateFsScanCache(path); - invalidateChunkCache(path); } /** @@ -25,9 +22,7 @@ export function invalidateFsScanAfterDelete(path: string): void { */ export function invalidateFsScanAfterRename(oldPath: string, newPath: string): void { invalidateFsScanCache(oldPath); - invalidateChunkCache(oldPath); if (newPath !== oldPath) { invalidateFsScanCache(newPath); - invalidateChunkCache(newPath); } } diff --git a/packages/coding-agent/src/tools/grep.ts b/packages/coding-agent/src/tools/grep.ts index a0919e720..74e825034 100644 --- a/packages/coding-agent/src/tools/grep.ts +++ b/packages/coding-agent/src/tools/grep.ts @@ -6,9 +6,8 @@ import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; import { prompt, untilAborted } from "@oh-my-pi/pi-utils"; import { type Static, Type } from "@sinclair/typebox"; -import { type ChunkedGrepMatch, describeChunkedGrepMatch } from "../edit/modes/chunk"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; -import { getLanguageFromPath, type Theme } from "../modes/theme/theme"; +import { type Theme } from "../modes/theme/theme"; import grepDescription from "../prompts/tools/grep.md" with { type: "text" }; import { DEFAULT_MAX_COLUMN, type TruncationResult, truncateHead } from "../session/streaming-output"; import { Ellipsis, Hasher, type RenderCache, renderStatusLine, renderTreeList, truncateToWidth } from "../tui"; @@ -83,7 +82,6 @@ export class GrepTool implements AgentTool { this.description = prompt.render(grepDescription, { IS_HASHLINE_MODE: displayMode.hashLines, IS_LINE_NUMBER_MODE: !displayMode.hashLines && displayMode.lineNumbers, - IS_CHUNK_MODE: displayMode.chunked, }); } @@ -98,7 +96,6 @@ export class GrepTool implements AgentTool { return untilAborted(signal, async () => { const normalizedPattern = pattern.trim(); - const chunkMode = resolveEditMode(this.session) === "chunk"; if (!normalizedPattern) { throw new ToolError("Pattern must not be empty"); } @@ -297,124 +294,6 @@ export class GrepTool implements AgentTool { } matchesByFile.get(relativePath)!.push(match); } - if (chunkMode) { - const annotatedMatches = await Promise.all( - selectedMatches.map(match => { - const relativePath = match.path.startsWith("/") ? match.path.slice(1) : match.path; - const absoluteFilePath = isDirectory ? path.join(searchPath, relativePath) : searchPath; - return describeChunkedGrepMatch({ - filePath: absoluteFilePath, - lineNumber: match.lineNumber, - line: match.line, - cwd: this.session.cwd, - language: getLanguageFromPath(absoluteFilePath), - }); - }), - ); - const chunkMatchesByFile = new Map(); - for (const match of annotatedMatches) { - recordFile(match.displayPath); - if (!chunkMatchesByFile.has(match.displayPath)) { - chunkMatchesByFile.set(match.displayPath, []); - } - chunkMatchesByFile.get(match.displayPath)!.push(match); - } - const renderChunkedMatchesForFile = (relativePath: string): string[] => { - const renderedLines: string[] = []; - const fileMatches = chunkMatchesByFile.get(relativePath) ?? []; - if (fileMatches.length === 0) { - return renderedLines; - } - const matchesByChunk = new Map(); - for (const match of fileMatches) { - const chunkKey = match.chunkPath ?? ""; - if (!matchesByChunk.has(chunkKey)) { - matchesByChunk.set(chunkKey, []); - } - matchesByChunk.get(chunkKey)!.push(match); - } - for (const [chunkPath, chunkMatches] of matchesByChunk) { - if (chunkPath) { - const chunkChecksum = chunkMatches[0]?.chunkChecksum; - const dashes = "-".repeat(chunkPath.split(".").length - 1); - const anchor = chunkChecksum - ? `${dashes}@${chunkPath}#${chunkChecksum}` - : `${dashes}@${chunkPath}`; - renderedLines.push(anchor); - } - for (const match of chunkMatches) { - renderedLines.push(` ${match.lineNumber}|${match.line}`); - fileMatchCounts.set(relativePath, (fileMatchCounts.get(relativePath) ?? 0) + 1); - } - } - return renderedLines; - }; - if (isDirectory) { - const filesByDirectory = new Map(); - for (const relativePath of fileList) { - const directory = path.dirname(relativePath).replace(/\\/g, "/"); - if (!filesByDirectory.has(directory)) { - filesByDirectory.set(directory, []); - } - filesByDirectory.get(directory)!.push(relativePath); - } - for (const [directory, directoryFiles] of filesByDirectory) { - if (directory === ".") { - for (const relativePath of directoryFiles) { - const renderedLines = renderChunkedMatchesForFile(relativePath); - if (renderedLines.length === 0) continue; - if (outputLines.length > 0) { - outputLines.push(""); - } - outputLines.push(`# ${path.basename(relativePath)}`); - outputLines.push(...renderedLines); - } - continue; - } - const renderedFiles = directoryFiles - .map(relativePath => ({ relativePath, lines: renderChunkedMatchesForFile(relativePath) })) - .filter(file => file.lines.length > 0); - if (renderedFiles.length === 0) continue; - if (outputLines.length > 0) { - outputLines.push(""); - } - outputLines.push(`# ${directory}`); - for (const { relativePath, lines } of renderedFiles) { - outputLines.push(`## └─ ${path.basename(relativePath)}`); - outputLines.push(...lines); - } - } - } else { - for (const relativePath of fileList) { - outputLines.push(...renderChunkedMatchesForFile(relativePath)); - } - } - if (matchLimitReached || result.limitReached) { - outputLines.push("", limitMessage); - } - const rawOutput = outputLines.join("\n"); - const truncation = truncateHead(rawOutput, { maxLines: Number.MAX_SAFE_INTEGER }); - const truncated = Boolean(matchLimitReached || result.limitReached || truncation.truncated); - const details: GrepToolDetails = { - scopePath, - matchCount: selectedMatches.length, - fileCount: fileList.length, - files: fileList, - fileMatches: fileList.map(path => ({ - path, - count: fileMatchCounts.get(path) ?? 0, - })), - truncated, - matchLimitReached: matchLimitReached ? effectiveLimit : undefined, - resultLimitReached: result.limitReached ? internalLimit : undefined, - }; - if (truncation.truncated) details.truncation = truncation; - const resultBuilder = toolResult(details).text(truncation.content); - if (truncation.truncated) { - resultBuilder.truncation(truncation, { direction: "head" }); - } - return resultBuilder.done(); - } const displayLines: string[] = []; const renderMatchesForFile = (relativePath: string): { model: string[]; display: string[] } => { const modelOut: string[] = []; diff --git a/packages/coding-agent/src/tools/poll-tool.ts b/packages/coding-agent/src/tools/poll-tool.ts index 009459567..8b27d79f2 100644 --- a/packages/coding-agent/src/tools/poll-tool.ts +++ b/packages/coding-agent/src/tools/poll-tool.ts @@ -25,7 +25,7 @@ const WAIT_DURATION_MS: Record = { }; function parseWaitDurationMs(value: string | undefined): number { - return (value && WAIT_DURATION_MS[value]) ?? WAIT_DURATION_MS["30s"]; + return (value ? WAIT_DURATION_MS[value] : undefined) ?? WAIT_DURATION_MS["30s"]; } interface PollResult { diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index a5fde7e89..a19dfbc38 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -9,20 +9,11 @@ import { Text } from "@oh-my-pi/pi-tui"; import { getRemoteDir, prompt, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils"; import { type Static, Type } from "@sinclair/typebox"; import { formatHashLines } from "../edit/line-hash"; -import { - type ChunkReadTarget, - formatChunkedRead, - parseChunkReadPath, - parseChunkSelector, - resolveAnchorStyle, - resolveChunkAutoIndent, -} from "../edit/modes/chunk"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import { parseInternalUrl } from "../internal-urls/parse"; import type { InternalUrl } from "../internal-urls/types"; import { getLanguageFromPath, type Theme } from "../modes/theme/theme"; import readDescription from "../prompts/tools/read.md" with { type: "text" }; -import readChunkDescription from "../prompts/tools/read-chunk.md" with { type: "text" }; import type { ToolSession } from "../sdk"; import { DEFAULT_MAX_BYTES, @@ -71,12 +62,6 @@ import { import { ToolAbortError, ToolError, throwIfAborted } from "./tool-errors"; import { toolResult } from "./tool-result"; -const PROSE_LANGUAGES = new Set(["markdown", "text", "log", "asciidoc", "restructuredtext"]); - -function isProseLanguage(language: string | undefined): boolean { - return language !== undefined && PROSE_LANGUAGES.has(language); -} - // Document types converted to markdown via markit. const CONVERTIBLE_EXTENSIONS = new Set([".pdf", ".doc", ".docx", ".ppt", ".pptx", ".xls", ".xlsx", ".rtf", ".epub"]); @@ -370,7 +355,6 @@ export interface ReadToolDetails { isDirectory?: boolean; resolvedPath?: string; suffixResolution?: { from: string; to: string }; - chunk?: ChunkReadTarget; url?: string; finalUrl?: string; contentType?: string; @@ -389,16 +373,14 @@ type ReadParams = ReadToolInput; type ParsedSelector = | { kind: "none" } | { kind: "raw" } - | { kind: "lines"; startLine: number; endLine: number | undefined } - | { kind: "chunk"; selector: string }; + | { kind: "lines"; startLine: number; endLine: number | undefined }; const LINE_RANGE_RE = /^L(\d+)(?:-L?(\d+))?$/i; function parseSel(sel: string | undefined): ParsedSelector { if (!sel || sel.length === 0) return { kind: "none" }; - const normalizedSelector = parseChunkSelector(sel).selector ?? sel; - if (normalizedSelector === "raw") return { kind: "raw" }; - const lineMatch = LINE_RANGE_RE.exec(normalizedSelector); + if (sel === "raw") return { kind: "raw" }; + const lineMatch = LINE_RANGE_RE.exec(sel); if (lineMatch) { const rawStart = Number.parseInt(lineMatch[1]!, 10); if (rawStart < 1) { @@ -410,7 +392,7 @@ function parseSel(sel: string | undefined): ParsedSelector { } return { kind: "lines", startLine: rawStart, endLine: rawEnd }; } - return { kind: "chunk", selector: normalizedSelector }; + throw new ToolError(`Invalid sel '${sel}'. Use 'raw' or a line range like 'L50' or 'L50-L120'.`); } /** Convert a line-range selector to the offset/limit pair used by internal pagination. */ @@ -477,18 +459,12 @@ export class ReadTool implements AgentTool { Math.min(session.settings.get("read.defaultLimit") ?? DEFAULT_MAX_LINES, DEFAULT_MAX_LINES), ); this.#inspectImageEnabled = session.settings.get("inspect_image.enabled"); - this.description = - resolveEditMode(session) === "chunk" - ? prompt.render(readChunkDescription, { - anchorStyle: resolveAnchorStyle(session.settings), - chunkAutoIndent: resolveChunkAutoIndent(), - }) - : prompt.render(readDescription, { - DEFAULT_LIMIT: String(this.#defaultLimit), - DEFAULT_MAX_LINES: String(DEFAULT_MAX_LINES), - IS_HASHLINE_MODE: displayMode.hashLines, - IS_LINE_NUMBER_MODE: !displayMode.hashLines && displayMode.lineNumbers, - }); + this.description = prompt.render(readDescription, { + DEFAULT_LIMIT: String(this.#defaultLimit), + DEFAULT_MAX_LINES: String(DEFAULT_MAX_LINES), + IS_HASHLINE_MODE: displayMode.hashLines, + IS_LINE_NUMBER_MODE: !displayMode.hashLines && displayMode.lineNumbers, + }); } async #resolveArchiveReadPath(readPath: string, signal?: AbortSignal): Promise { @@ -930,7 +906,6 @@ export class ReadTool implements AgentTool { readPath = expandPath(readPath); } const displayMode = resolveFileDisplayMode(this.session); - const chunkMode = resolveEditMode(this.session) === "chunk"; // Handle internal URLs (agent://, artifact://, memory://, skill://, rule://, local://, mcp://) const internalRouter = this.session.internalRouter; @@ -964,13 +939,8 @@ export class ReadTool implements AgentTool { return executeReadUrl(this.session, { path: parsedUrlTarget.path, timeout, raw: parsedUrlTarget.raw }, signal); } - const parsedReadPath = chunkMode ? parseChunkReadPath(readPath) : { filePath: readPath }; - const localReadPath = parsedReadPath.filePath; - const pathSelectorParsed = chunkMode ? parseSel(parsedReadPath.selector) : { kind: "none" as const }; - const pathChunkSelector = pathSelectorParsed.kind === "chunk" ? pathSelectorParsed.selector : undefined; - const selectorInput = sel ?? parsedReadPath.selector; - const rawSelectorInput = sel ?? parsedReadPath.selector; - const parsed = parseSel(selectorInput); + const localReadPath = readPath; + const parsed = parseSel(sel); const archivePath = await this.#resolveArchiveReadPath(localReadPath, signal); if (archivePath) { @@ -1032,52 +1002,8 @@ export class ReadTool implements AgentTool { const ext = path.extname(absolutePath).toLowerCase(); const hasEditTool = this.session.hasEditTool ?? true; const language = getLanguageFromPath(absolutePath); - const skipChunksForExplore = !hasEditTool && !this.session.settings.get("read.explorechunks"); - const skipChunksForProse = isProseLanguage(language) && !this.session.settings.get("read.prosechunks"); const shouldConvertWithMarkit = - CONVERTIBLE_EXTENSIONS.has(ext) || (ext === ".ipynb" && (parsed.kind === "raw" || !chunkMode)); - - if (chunkMode && parsed.kind !== "raw" && !skipChunksForExplore && !skipChunksForProse) { - const absoluteLineRange = - pathChunkSelector && parsed.kind === "lines" - ? { startLine: parsed.startLine, endLine: parsed.endLine } - : undefined; - // sel= wins over path:chunk when both are provided (explicit param > embedded path). - const effectiveSelector = sel ? selectorInput : (pathChunkSelector ?? selectorInput); - const rawEffectiveSelector = sel ? selectorInput : (rawSelectorInput ?? effectiveSelector); - const chunkReadPath = - parsed.kind === "chunk" || (pathChunkSelector && !sel) - ? rawEffectiveSelector - ? `${localReadPath}:${rawEffectiveSelector}` - : localReadPath - : parsed.kind === "lines" - ? parsed.endLine !== undefined - ? `${localReadPath}:L${parsed.startLine}-L${parsed.endLine}` - : `${localReadPath}:L${parsed.startLine}` - : localReadPath; - const chunkResult = await formatChunkedRead({ - filePath: absolutePath, - readPath: chunkReadPath, - cwd: this.session.cwd, - language, - omitChecksum: !hasEditTool, - anchorStyle: resolveAnchorStyle(this.session.settings), - absoluteLineRange, - }); - let text = chunkResult.text; - if (suffixResolution) { - text = prependSuffixResolutionNotice(text, suffixResolution); - } - return toolResult({ - resolvedPath: absolutePath, - suffixResolution, - chunk: chunkResult.chunk, - }) - .text(text) - .sourcePath(absolutePath) - .done(); - } - + CONVERTIBLE_EXTENSIONS.has(ext) || (ext === ".ipynb" && parsed.kind === "raw"); // Read the file based on type let content: Array; let details: ReadToolDetails = {}; @@ -1160,31 +1086,6 @@ export class ReadTool implements AgentTool { content = [{ type: "text", text: `[Cannot read ${ext} file: conversion failed]` }]; } } else { - // Chunk mode: dispatch to chunk tree unless raw or line range requested - if (chunkMode && parsed.kind !== "raw" && parsed.kind !== "lines") { - const chunkSel = parsed.kind === "chunk" ? parsed.selector : undefined; - const chunkResult = await formatChunkedRead({ - filePath: absolutePath, - readPath: chunkSel ? `${localReadPath}:${chunkSel}` : localReadPath, - cwd: this.session.cwd, - language: getLanguageFromPath(absolutePath), - omitChecksum: !(this.session.hasEditTool ?? true), - anchorStyle: resolveAnchorStyle(this.session.settings), - }); - let text = chunkResult.text; - if (suffixResolution) { - text = prependSuffixResolutionNotice(text, suffixResolution); - } - return toolResult({ - resolvedPath: absolutePath, - suffixResolution, - chunk: chunkResult.chunk, - }) - .text(text) - .sourcePath(absolutePath) - .done(); - } - // Raw text or line-range mode const { offset, limit } = selToOffsetLimit(parsed); const startLine = offset ? Math.max(0, offset - 1) : 0; diff --git a/packages/coding-agent/src/utils/edit-mode.ts b/packages/coding-agent/src/utils/edit-mode.ts index 8e2b57aa1..515a47d1a 100644 --- a/packages/coding-agent/src/utils/edit-mode.ts +++ b/packages/coding-agent/src/utils/edit-mode.ts @@ -1,13 +1,12 @@ import { $env, $flag } from "@oh-my-pi/pi-utils"; -export type EditMode = "replace" | "patch" | "hashline" | "chunk" | "vim" | "apply_patch" | "atom"; +export type EditMode = "replace" | "patch" | "hashline" | "vim" | "apply_patch" | "atom"; export const DEFAULT_EDIT_MODE: EditMode = "hashline"; const EDIT_MODE_IDS = { apply_patch: "apply_patch", atom: "atom", - chunk: "chunk", hashline: "hashline", patch: "patch", replace: "replace", diff --git a/packages/coding-agent/src/utils/file-display-mode.ts b/packages/coding-agent/src/utils/file-display-mode.ts index 688a51540..e9006a4be 100644 --- a/packages/coding-agent/src/utils/file-display-mode.ts +++ b/packages/coding-agent/src/utils/file-display-mode.ts @@ -7,7 +7,6 @@ import { resolveEditMode } from "./edit-mode"; export interface FileDisplayMode { lineNumbers: boolean; hashLines: boolean; - chunked: boolean; } /** Session-like object providing settings and tool availability for display mode resolution. */ @@ -33,10 +32,8 @@ export function resolveFileDisplayMode(session: FileDisplayModeSession, options? const usesHashLineAnchors = editMode === "hashline" || editMode === "atom"; const raw = options?.raw === true; const hashLines = !raw && hasEditTool && usesHashLineAnchors && settings.get("readHashLines") !== false; - const chunked = !raw && hasEditTool && editMode === "chunk"; return { hashLines, lineNumbers: !raw && (hashLines || settings.get("readLineNumbers") === true), - chunked, }; } diff --git a/packages/coding-agent/test/cli/read-cli.test.ts b/packages/coding-agent/test/cli/read-cli.test.ts deleted file mode 100644 index 34ec5fb46..000000000 --- a/packages/coding-agent/test/cli/read-cli.test.ts +++ /dev/null @@ -1,33 +0,0 @@ -import { afterEach, describe, expect, it, vi } from "bun:test"; -import * as os from "node:os"; -import * as path from "node:path"; -import { runReadCommand } from "@oh-my-pi/pi-coding-agent/cli/read-cli"; -import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types"; - -describe("runReadCommand URL handling", () => { - afterEach(() => { - vi.restoreAllMocks(); - }); - - it("delegates URL inputs through the read tool pipeline", async () => { - const cwd = path.join(os.tmpdir(), "read-cli-url-test"); - const settings = Settings.isolated({ "fetch.enabled": true }); - const pageUrl = "https://example.com/cli-read"; - const consoleLogSpy = vi.spyOn(console, "log").mockImplementation(() => {}); - vi.spyOn(Settings, "init").mockResolvedValue(settings); - vi.spyOn(scrapers, "loadPage").mockResolvedValue({ - ok: true, - status: 200, - contentType: "text/plain", - finalUrl: pageUrl, - content: "CLI URL content", - }); - const cwdSpy = vi.spyOn(process, "cwd").mockReturnValue(cwd); - - await runReadCommand({ path: pageUrl }); - - expect(cwdSpy).toHaveBeenCalled(); - expect(consoleLogSpy).toHaveBeenCalledWith(expect.stringContaining("CLI URL content")); - }); -}); diff --git a/packages/coding-agent/test/edit-diff.test.ts b/packages/coding-agent/test/edit-diff.test.ts index b5e00551b..e3c3015a1 100644 --- a/packages/coding-agent/test/edit-diff.test.ts +++ b/packages/coding-agent/test/edit-diff.test.ts @@ -4,14 +4,10 @@ import * as os from "node:os"; import * as path from "node:path"; import { adjustIndentation, - computeChunkDiff, computeEditDiff, computeHashlineDiff, DEFAULT_FUZZY_THRESHOLD, findMatch, - loadChunkSource, - parseChunkEditPath, - parseChunkReadPath, } from "@oh-my-pi/pi-coding-agent/edit"; describe("findMatch", () => { @@ -162,127 +158,6 @@ describe("findMatch", () => { }); }); -describe("computeChunkDiff", () => { - let tmpDir: string; - beforeEach(async () => { - tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "compute-chunk-")); - }); - afterEach(async () => { - await fs.rm(tmpDir, { recursive: true, force: true }); - }); - - test("returns { error } when chunk selector cannot resolve", async () => { - const file = path.join(tmpDir, "c.ts"); - await fs.writeFile(file, "export const x = 1;\n"); - const result = await computeChunkDiff( - { - path: "c.ts:fn_does_not_exist#ABCD", - edits: [ - { - path: "c.ts:fn_does_not_exist#ABCD", - write: "console.log('replaced')\n", - }, - ], - }, - tmpDir, - ); - expect("error" in result).toBe(true); - }); - - test("returns { error } when path is empty", async () => { - const result = await computeChunkDiff({ path: "", edits: [{ path: "", write: "x\n" }] }, tmpDir); - expect("error" in result).toBe(true); - }); - - test("rejects write:null instead of previewing a delete", async () => { - const file = path.join(tmpDir, "null-delete.ts"); - await fs.writeFile(file, "export const x = 1;\n"); - const result = await computeChunkDiff( - { - path: "null-delete.ts", - edits: [{ path: "null-delete.ts", write: null }], - }, - tmpDir, - ); - expect("error" in result).toBe(true); - if ("error" in result) { - expect(result.error).toContain("write:null no longer deletes chunks"); - } - }); - - test("rejects bare chunk edit entries instead of treating them as deletes", async () => { - const file = path.join(tmpDir, "bare.ts"); - await fs.writeFile(file, "export const x = 1;\n"); - const result = await computeChunkDiff( - { - path: "bare.ts", - edits: [{ path: "bare.ts" }], - }, - tmpDir, - ); - expect("error" in result).toBe(true); - if ("error" in result) { - expect(result.error).toContain("no operation specified"); - } - }); - - test("rejects write empty string instead of previewing a destructive empty replacement", async () => { - const file = path.join(tmpDir, "empty-write.ts"); - await fs.writeFile(file, "export const x = 1;\n"); - const result = await computeChunkDiff( - { - path: "empty-write.ts", - edits: [{ path: "empty-write.ts", write: "" }], - }, - tmpDir, - ); - expect("error" in result).toBe(true); - if ("error" in result) { - expect(result.error).toContain('write:"" is a destructive empty replacement'); - } - }); - - test("aborts when signal fires before compute completes", async () => { - const controller = new AbortController(); - controller.abort(); - const result = await computeChunkDiff( - { - path: "d.ts", - edits: [{ path: "d.ts", write: "foo\n" }], - }, - tmpDir, - { signal: controller.signal }, - ); - expect("error" in result).toBe(true); - }); - - test("computes diff for a root chunk replacement with valid checksum", async () => { - const file = path.join(tmpDir, "e.ts"); - await fs.writeFile(file, "export const x = 1;\n"); - // Read the file once via loadChunkSource so the test does not depend on - // knowing the internal chunk checksum scheme. - const loaded = await loadChunkSource({ cwd: tmpDir, path: "e.ts" }); - expect(loaded.exists).toBe(true); - expect(loaded.rawContent).toContain("export const x"); - }); -}); - -describe("chunk path parsing", () => { - test("splits local plan URLs with chunk selectors after the URL path", () => { - expect(parseChunkEditPath("local://PLAN.md:sct_0_T#SRJJ")).toEqual({ - filePath: "local://PLAN.md", - selector: "sct_0_T#SRJJ", - }); - expect(parseChunkReadPath("local://PLAN.md:sct_6_R.sct_6_u#MZKS")).toEqual({ - filePath: "local://PLAN.md", - selector: "sct_6_R.sct_6_u#MZKS", - }); - }); - - test("does not treat the local URL scheme colon as a chunk selector separator", () => { - expect(parseChunkEditPath("local://PLAN.md")).toEqual({ filePath: "local://PLAN.md" }); - }); -}); describe("adjustIndentation", () => { test("adds indentation when actualText is more indented than oldText", () => { diff --git a/packages/coding-agent/test/edit-streaming-preview.test.ts b/packages/coding-agent/test/edit-streaming-preview.test.ts index b559db165..7f0bfc646 100644 --- a/packages/coding-agent/test/edit-streaming-preview.test.ts +++ b/packages/coding-agent/test/edit-streaming-preview.test.ts @@ -30,45 +30,6 @@ describe("dropIncompleteLastEdit", () => { }); }); -describe("chunk extractCompleteEdits", () => { - const strategy = EDIT_MODE_STRATEGIES.chunk; - - test("passes through a single complete entry", () => { - const args = { - edits: [{ path: "a.ts", write: "foo" }], - __partialJson: '{"edits":[{"path":"a.ts","write":"foo"}]}', - }; - const out = strategy.extractCompleteEdits(args, args.__partialJson) as typeof args; - expect(out.edits).toHaveLength(1); - }); - - test("drops trailing entry when partial JSON has open-brace after last close", () => { - const args = { - edits: [{ path: "a.ts", write: "foo" }, { path: "b.ts" }], - __partialJson: '{"edits":[{"path":"a.ts","write":"foo"},{"path":"b.ts"', - }; - const out = strategy.extractCompleteEdits(args, args.__partialJson) as typeof args; - expect(out.edits).toHaveLength(1); - expect(out.edits[0].path).toBe("a.ts"); - }); - - test("drops trailing entry when partial JSON ends in ':nu' (write: null guard)", () => { - const args = { - edits: [ - { path: "a.ts", write: "foo" }, - { path: "b.ts", write: null }, - ], - // simulates partial-json coercing the in-flight `nu` to `null` - __partialJson: '{"edits":[{"path":"a.ts","write":"foo"},{"path":"b.ts","write":nu', - }; - const out = strategy.extractCompleteEdits(args, args.__partialJson) as typeof args; - // Last entry should be dropped because its `}` hasn't arrived yet, so - // incomplete null-write errors are suppressed while streaming. - expect(out.edits).toHaveLength(1); - expect(out.edits[0].path).toBe("a.ts"); - }); -}); - describe("apply_patch extractCompleteEdits", () => { const strategy = EDIT_MODE_STRATEGIES.apply_patch; diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index a260f357b..971d2256f 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Removed + +- Removed the `chunk` napi module (`ChunkState`, chunk schema, chunk rendering, chunk edit) and dropped `generate_chunk_schema()` from the build script + ## [14.3.0] - 2026-04-25 ### Added diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index ebfbfd40c..3dca5c6c4 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -1,69 +1,5 @@ /* auto-generated by NAPI-RS */ /* eslint-disable */ -/** - * Parsed file as a chunk tree: query nodes, render views, format grep hits, - * and apply edits. - */ -export declare class ChunkState { - /** - * Build chunk state by parsing `source` with the given `language` id (e.g. - * `typescript`). - */ - static parse(source: string, language: string): ChunkState - /** Normalized language identifier used for the tree-sitter parse. */ - get language(): string - /** Full source text for this file. */ - get source(): string - /** Stable checksum for the entire file contents. */ - get checksum(): string - /** Line count of the source buffer. */ - get lineCount(): number - /** Count of tree-sitter error nodes seen while building the tree. */ - get parseErrors(): number - /** True when a fallback classifier produced the tree. */ - get fallback(): boolean - /** Selector path string for the synthetic root (often empty). */ - get rootPath(): string - /** Top-level child chunk paths under the root. */ - get rootChildren(): Array - /** Total number of chunk nodes. */ - get chunkCount(): number - /** True when the parsed file contains unresolved merge conflicts. */ - hasConflicts(): boolean - /** Count of unresolved merge conflicts represented in the chunk tree. */ - conflictCount(): number - /** Summary for the root chunk, if it exists. */ - root(): ChunkInfo | null - /** Look up [`ChunkInfo`] for a chunk selector path. */ - chunk(chunkPath: string): ChunkInfo | null - /** Every chunk node as a [`ChunkInfo`] list. */ - chunks(): Array - /** - * Direct children of `chunkPath` (use empty or omit for root); errors if - * the path is missing. - */ - children(chunkPath?: string | undefined | null): Array - /** Chunk selector path that contains 1-based source line `line`, if any. */ - lineToContainingChunkPath(line: number): string | null - /** Render a chunk subtree or listing as UTF-8 text for tools. */ - render(params: RenderParams): string - /** - * Parse `readPath` (selector, line scope, etc.) and return rendered text or - * errors. - */ - renderRead(params: ReadRenderParams): ReadResult - /** - * Prefix a grep line with `display_path` and the chunk path for - * `line_number`, when known. - */ - formatGrepLine(displayPath: string, lineNumber: number, line: string): string - /** - * Apply batch edits, re-parse, write files, and return updated state and - * messaging. - */ - applyEdits(params: EditParams): EditResult -} - /** * Long-lived macOS appearance observer. * @@ -347,95 +283,6 @@ export interface AstReplaceResult { parseErrors?: Array } -/** - * How chunk anchors are formatted in rendered output (name and checksum - * visibility). - */ -export declare enum ChunkAnchorStyle { - /** `[.name#crc]` style anchor. */ - Full = 'full', - /** `[.kind#crc]` style anchor (kind is the name prefix before `_`). */ - Kind = 'kind', - /** `[#crc]` style anchor. */ - Bare = 'bare', - /** `[.name]` without checksum. */ - FullOmit = 'full-omit', - /** `[.kind]` without checksum. */ - KindOmit = 'kind-omit', - /** Minimal anchor without name or checksum. */ - None = 'none' -} - -/** Structural edit to apply relative to a chunk anchor. */ -export declare enum ChunkEditOp { - /** Put new content into the targeted region. */ - Put = 'put', - /** Find and replace a literal substring within the targeted region. */ - Replace = 'replace', - /** Remove the targeted region. */ - Delete = 'delete', - /** Insert `content` before the targeted region span. */ - Before = 'before', - /** Insert `content` after the targeted region span. */ - After = 'after', - /** Insert `content` at the start inside the targeted region. */ - Prepend = 'prepend', - /** Insert `content` at the end inside the targeted region. */ - Append = 'append' -} - -/** How a chunk participates in a focus-scoped render pass. */ -export declare enum ChunkFocusMode { - /** Emit full content and recurse normally. */ - Expanded = 'expanded', - /** Emit just the opening anchor; do not recurse or emit body. */ - Collapsed = 'collapsed', - /** - * Emit opening + closing anchors; recurse into focused children only. - * Interior gap lines between children are suppressed. - */ - Container = 'container' -} - -/** Summary of a single chunk node for tool output and navigation. */ -export interface ChunkInfo { - /** Chunk selector path within the tree. */ - path: string - /** Bare chunk identifier (without kind prefix), if available. */ - identifier?: string - /** Stable checksum anchor for this chunk. */ - checksum: string - /** 1-based start line in the source file (inclusive). */ - startLine: number - /** 1-based end line in the source file (inclusive). */ - endLine: number - /** Whether this node is a leaf (no child chunks). */ - leaf: boolean -} - -/** Result of resolving a chunk read request against the tree. */ -export declare enum ChunkReadStatus { - /** Selector matched a chunk and content was produced. */ - Ok = 'ok', - /** No chunk matched the requested selector. */ - NotFound = 'not_found', - /** Chunk matched but does not support the requested region. */ - UnsupportedRegion = 'unsupported_region' -} - -/** Outcome of resolving which chunk was read for a `renderRead`-style request. */ -export interface ChunkReadTarget { - /** Whether the selector matched. */ - status: ChunkReadStatus - /** Sanitized selector string that was applied. */ - selector: string -} - -export declare enum ChunkRegion { - Head = '^', - Body = '~' -} - /** Clipboard image payload encoded as PNG bytes. */ export interface ClipboardImage { /** PNG-encoded image bytes. */ @@ -469,78 +316,6 @@ export declare function copyToClipboard(text: string): void */ export declare function detectMacOSAppearance(): MacOSAppearance | null -/** - * One edit in a batch; targets a chunk via `sel`/`crc` (with params-level - * defaults). - */ -export interface EditOperation { - /** Edit kind (replace, delete, insert relative to anchor). */ - op: ChunkEditOp - /** - * Chunk selector path; falls back to `EditParams.defaultSelector` when - * omitted. - */ - sel?: string - /** - * Optional checksum anchor; falls back to `EditParams.defaultCrc` when - * omitted. - */ - crc?: string - /** Region to target. When omitted, targets the full chunk. */ - region?: ChunkRegion - /** Replacement or inserted text (meaning depends on `op`). */ - content?: string - /** - * For `replace` op: literal substring to find inside the target chunk. - * Must match exactly once. - */ - find?: string -} - -/** Arguments for applying a batch of chunk edits to a file. */ -export interface EditParams { - /** Edits to apply in order. */ - operations: Array - /** - * When true, normalize indentation for response rendering and inserted - * content. When false, preserve literal tabs/spaces. - */ - normalizeIndent?: boolean - /** Default chunk selector when an `EditOperation` omits `sel`. */ - defaultSelector?: string - /** Default checksum when an `EditOperation` omits `crc`. */ - defaultCrc?: string - /** Anchor formatting for rendered response text. */ - anchorStyle?: ChunkAnchorStyle - /** Working directory used to resolve `filePath` and display paths. */ - cwd: string - /** Path to the source file to edit (often relative to `cwd`). */ - filePath: string -} - -/** - * Result of applying edits: new parse state plus before/after source and - * messaging. - */ -export interface EditResult { - /** Chunk tree state after applying edits and re-parsing. */ - state: ChunkState - /** Full file text before edits. */ - diffBefore: string - /** Full file text after edits. */ - diffAfter: string - /** Rendered summary for tooling (hunks, anchors), driven by `anchorStyle`. */ - responseText: string - /** Whether the on-disk source changed. */ - changed: boolean - /** Whether the updated source re-parsed without fatal issues. */ - parseValid: boolean - /** Absolute or normalized paths that were written or touched. */ - touchedPaths: Array - /** Non-fatal issues (e.g. selector warnings) collected during apply. */ - warnings: Array -} - /** Ellipsis strategy for [`truncate_to_width`]. */ export declare enum Ellipsis { /** Use a single Unicode ellipsis character ("…"). */ @@ -601,18 +376,6 @@ export declare enum FileType { Symlink = 3 } -/** Path + focus mode pair for the N-API boundary (`HashMap` doesn't cross FFI). */ -export interface FocusedPath { - path: string - mode: ChunkFocusMode -} - -/** - * Format one chunk anchor string for a node at `depth` using `style` and - * optional checksum omission. - */ -export declare function formatAnchor(name: string, checksum: string, style: ChunkAnchorStyle, omitChecksum?: boolean | undefined | null): string - /** Fuzzy file path search for autocomplete. */ export declare function fuzzyFind(options: FuzzyFindOptions): Promise @@ -1157,69 +920,6 @@ export interface PtyStartOptions { */ export declare function readImageFromClipboard(): Promise -/** - * Options for `ChunkState.renderRead`: selector path, display path, and - * optional line scoping. - */ -export interface ReadRenderParams { - /** Read selector (`sel=...` path, line range, or empty for whole tree). */ - readPath: string - /** Path shown in titles and error messages (often the file path). */ - displayPath: string - /** Optional language label for the rendered block. */ - languageTag?: string - /** Hide checksums in rendered anchors. */ - omitChecksum: boolean - /** Anchor formatting style. */ - anchorStyle?: ChunkAnchorStyle - /** Optional absolute file line range to intersect with the resolved chunk. */ - absoluteLineRange?: VisibleLineRange - /** Replace tabs in embedded previews. */ - tabReplacement?: string - /** When true, normalize displayed indentation to canonical tabs. */ - normalizeIndent?: boolean -} - -/** Rendered chunk text plus optional resolution metadata for the read request. */ -export interface ReadResult { - /** Rendered UTF-8 text (chunk tree, notice, or error message). */ - text: string - /** When a selector was used, whether it matched and which selector applied. */ - chunk?: ChunkReadTarget -} - -/** - * Options for `ChunkState.render`: which subtree to show and how anchors - * appear. - */ -export interface RenderParams { - /** Path of the chunk to render; `None` uses the tree root. */ - chunkPath?: string - /** Title line shown above the tree (often the file path). */ - title: string - /** Optional language label for the header block. */ - languageTag?: string - /** Restrict output to an inclusive line range of the file. */ - visibleRange?: VisibleLineRange - /** When true, list only direct children instead of a full subtree. */ - renderChildrenOnly: boolean - /** Hide checksums in anchors when true. */ - omitChecksum: boolean - /** Anchor formatting style for chunk headers. */ - anchorStyle?: ChunkAnchorStyle - /** Include a one-line preview for leaf chunks. */ - showLeafPreview: boolean - /** Replace tab characters in displayed previews (e.g. two spaces). */ - tabReplacement?: string - /** When true, normalize displayed indentation to canonical tabs. */ - normalizeIndent?: boolean - /** - * When set, restrict rendering to these chunks with their specified focus - * modes. Everything not in this list is skipped. - */ - focusedPaths?: Array -} - /** Sampling filter for resize operations. */ export declare enum SamplingFilter { /** Nearest-neighbor sampling (fast, low quality). */ @@ -1395,17 +1095,6 @@ export declare function supportsLanguage(lang: string): boolean */ export declare function truncateToWidth(text: string, maxWidth: number, ellipsisKind: Ellipsis | undefined | null, pad: boolean | undefined | null, tabWidth: number): string -/** - * Inclusive 1-based line range within a source file (used for scoped chunk - * rendering). - */ -export interface VisibleLineRange { - /** First line to include. */ - startLine: number - /** Last line to include. */ - endLine: number -} - /** * Calculate visible width of text, excluding ANSI escape sequences. * diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index 7e3e6ec00..ef7b00235 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -233,37 +233,6 @@ module.exports.AstMatchStrictness = { Signature: 'signature', Template: 'template', }; -module.exports.ChunkAnchorStyle = { - Full: 'full', - Kind: 'kind', - Bare: 'bare', - FullOmit: 'full-omit', - KindOmit: 'kind-omit', - None: 'none', -}; -module.exports.ChunkEditOp = { - Put: 'put', - Replace: 'replace', - Delete: 'delete', - Before: 'before', - After: 'after', - Prepend: 'prepend', - Append: 'append', -}; -module.exports.ChunkFocusMode = { - Expanded: 'expanded', - Collapsed: 'collapsed', - Container: 'container', -}; -module.exports.ChunkReadStatus = { - Ok: 'ok', - NotFound: 'not_found', - UnsupportedRegion: 'unsupported_region', -}; -module.exports.ChunkRegion = { - Head: '^', - Body: '~', -}; module.exports.Ellipsis = { Unicode: 0, Ascii: 1, diff --git a/packages/typescript-edit-benchmark/src/index.ts b/packages/typescript-edit-benchmark/src/index.ts index 15e9bfdb1..7ab279b5a 100644 --- a/packages/typescript-edit-benchmark/src/index.ts +++ b/packages/typescript-edit-benchmark/src/index.ts @@ -98,7 +98,7 @@ Options: --tasks Comma-separated task IDs to run (default: all) --max-tasks Max tasks to sample (default: 80, 0 = all) --fixtures Fixtures directory or .tar.gz archive (default: built-in) - --edit-variant Edit variant: any string (e.g. replace, patch, hashline, chunk, vim, atom, apply_patch), or auto (default: auto) + --edit-variant Edit variant: any string (e.g. replace, patch, hashline, vim, atom, apply_patch), or auto (default: auto) --edit-fuzzy Fuzzy matching: true, false, auto (default: auto) --edit-fuzzy-threshold Fuzzy threshold 0-1 or auto (default: auto) --auto-format Auto-format output files after verify (debug only) diff --git a/packages/typescript-edit-benchmark/src/report.ts b/packages/typescript-edit-benchmark/src/report.ts index 580417c86..e519c8b09 100644 --- a/packages/typescript-edit-benchmark/src/report.ts +++ b/packages/typescript-edit-benchmark/src/report.ts @@ -136,7 +136,7 @@ export function generateReport(result: BenchmarkResult): string { if (typeof summary.mutationIntentMatchRate === "number") { lines.push(`| Mutation Intent Match Rate | ${formatPercent(summary.mutationIntentMatchRate)} |`); } - if (config.editVariant === "patch" || config.editVariant === "hashline" || config.editVariant === "chunk") { + if (config.editVariant === "patch" || config.editVariant === "hashline") { lines.push(`| Patch Failure Rate | ${formatRate(totalEditFailures, totalEditAttempts)} |`); } lines.push(`| Tasks All Passing | ${summary.tasksWithAllPassing} |`); @@ -188,24 +188,6 @@ export function generateReport(result: BenchmarkResult): string { } } - if (summary.chunkEditSubtypes) { - const order = ["append", "prepend", "replace", "delete"] as const; - const total = order.reduce((sum, key) => sum + (summary.chunkEditSubtypes?.[key] ?? 0), 0); - if (total > 0) { - lines.push("### Chunk Edit Subtypes"); - lines.push(""); - lines.push("| Operation | Count | % |"); - lines.push("|-----------|-------|---|"); - for (const key of order) { - const count = summary.chunkEditSubtypes[key] ?? 0; - const pct = formatPercent(count / total); - lines.push(`| ${key} | ${count} | ${pct} |`); - } - lines.push(`| **Total** | **${total}** | 100% |`); - lines.push(""); - } - } - lines.push("## Task Results"); lines.push(""); lines.push("| Task | File | Success | Edit Hit | R/E/W | Tokens (In/Out) | Time | Indent |"); diff --git a/packages/typescript-edit-benchmark/src/runner.ts b/packages/typescript-edit-benchmark/src/runner.ts index 3f791665f..17e102255 100644 --- a/packages/typescript-edit-benchmark/src/runner.ts +++ b/packages/typescript-edit-benchmark/src/runner.ts @@ -192,8 +192,6 @@ function getEditPathFromArgs(args: unknown): string | null { const HASHLINE_SUBTYPES = ["set", "set_range", "insert"] as const; -const CHUNK_OP_SUBTYPES = ["append", "prepend", "replace", "delete"] as const; - const BENCHMARK_TOOL_NAMES = ["read", "edit", "vim", "write", "apply_patch"] as const; const EDIT_TOOL_NAMES = ["edit", "vim", "apply_patch"] as const; @@ -205,21 +203,6 @@ function isMutationTool(toolName: unknown): boolean { return isEditTool(toolName) || toolName === "write"; } -function countChunkEditSubtypes(args: unknown): Record { - const counts: Record = Object.fromEntries(CHUNK_OP_SUBTYPES.map(k => [k, 0])); - if (!args || typeof args !== "object") return counts; - const operations = (args as { operations?: unknown[] }).operations; - if (!Array.isArray(operations)) return counts; - for (const operation of operations) { - if (!operation || typeof operation !== "object") continue; - const op = (operation as { op?: string }).op; - if (typeof op === "string" && op in counts) { - counts[op]++; - } - } - return counts; -} - function countHashlineEditSubtypes(args: unknown): Record { const counts: Record = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0])); if (!args || typeof args !== "object") return counts; @@ -771,8 +754,6 @@ export interface TaskRunResult { editAutocorrectCount: number; /** Hashline edit subtype counts (replaceLine, replaceLines, etc.) — only when editVariant is hashline */ hashlineEditSubtypes?: Record; - /** Chunk edit subtype counts — only when editVariant is chunk */ - chunkEditSubtypes?: Record; mutationIntentMatched?: boolean; mutationIntentReason?: string; timeoutTelemetry?: PromptAttemptTelemetry; @@ -838,8 +819,6 @@ export interface BenchmarkSummary { mutationIntentMatchRate?: number; /** Hashline edit subtype totals — only when editVariant is hashline */ hashlineEditSubtypes?: Record; - /** Chunk edit subtype totals — only when editVariant is chunk */ - chunkEditSubtypes?: Record; } export interface BenchmarkResult { @@ -931,7 +910,6 @@ async function runSingleTask( totalInputChars: 0, }; const hashlineSubtypes: Record = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0])); - const chunkSubtypes: Record = Object.fromEntries(CHUNK_OP_SUBTYPES.map(k => [k, 0])); const logFile = path.join(TMP, `run-${task.id}-${runIndex}.jsonl`); const logEvent = async (event: unknown) => { @@ -1154,12 +1132,6 @@ async function runSingleTask( hashlineSubtypes[key] += counts[key]; } } - if (config.editVariant === "chunk" && args) { - const counts = countChunkEditSubtypes(args); - for (const key of CHUNK_OP_SUBTYPES) { - chunkSubtypes[key] += counts[key]; - } - } if (e.isError) { toolStats.editFailures++; const error = await appendNoChangeMutationHint( @@ -1286,7 +1258,6 @@ async function runSingleTask( editWarnings, editAutocorrectCount, hashlineEditSubtypes: config.editVariant === "hashline" ? hashlineSubtypes : undefined, - chunkEditSubtypes: config.editVariant === "chunk" ? chunkSubtypes : undefined, mutationIntentMatched: mutationIntentValidation?.matched, mutationIntentReason: mutationIntentValidation?.reason, timeoutTelemetry, @@ -1335,7 +1306,6 @@ async function _runRpcBenchmarkRun( totalInputChars: 0, }; const hashlineSubtypes: Record = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0])); - const chunkSubtypes: Record = Object.fromEntries(CHUNK_OP_SUBTYPES.map(k => [k, 0])); const logFile = path.join(sessionDir, `run-${task.id}-${runIndex}.jsonl`); const logEvent = async (event: unknown) => { @@ -1483,12 +1453,6 @@ async function _runRpcBenchmarkRun( hashlineSubtypes[key] += counts[key]; } } - if (config.editVariant === "chunk" && args) { - const counts = countChunkEditSubtypes(args); - for (const key of CHUNK_OP_SUBTYPES) { - chunkSubtypes[key] += counts[key]; - } - } if (e.isError) { toolStats.editFailures++; const toolError = await appendNoChangeMutationHint( @@ -1609,7 +1573,6 @@ async function _runRpcBenchmarkRun( editWarnings, editAutocorrectCount, hashlineEditSubtypes: config.editVariant === "hashline" ? hashlineSubtypes : undefined, - chunkEditSubtypes: config.editVariant === "chunk" ? chunkSubtypes : undefined, mutationIntentMatched: mutationIntentValidation?.matched, mutationIntentReason: mutationIntentValidation?.reason, timeoutTelemetry, @@ -2161,15 +2124,6 @@ export async function runBenchmark( ) : undefined; - const chunkEditSubtypes: Record | undefined = - config.editVariant === "chunk" - ? Object.fromEntries( - CHUNK_OP_SUBTYPES.map(key => [ - key, - allRuns.reduce((sum, r) => sum + (r.chunkEditSubtypes?.[key] ?? 0), 0), - ]), - ) - : undefined; const denom = effectiveRuns || 1; const summary: BenchmarkSummary = { @@ -2212,7 +2166,6 @@ export async function runBenchmark( transportFailureRuns, mutationIntentMatchRate, hashlineEditSubtypes, - chunkEditSubtypes, }; return { diff --git a/scripts/edit-benchmark.py b/scripts/edit-benchmark.py index 3f749d01b..86d1926dc 100755 --- a/scripts/edit-benchmark.py +++ b/scripts/edit-benchmark.py @@ -2,11 +2,10 @@ """ Edit benchmark: tests the edit tool across models with a simple edit task. -Select the edit variant via the PI_EDIT_VARIANT env var (e.g. `chunk`, `vim`, +Select the edit variant via the PI_EDIT_VARIANT env var (e.g. `vim`, `hashline`, `replace`, `patch`, `apply_patch`) or `--variant`. Examples: - PI_EDIT_VARIANT=chunk scripts/edit-benchmark.py PI_EDIT_VARIANT=vim scripts/edit-benchmark.py scripts/edit-benchmark.py --variant hashline """ @@ -54,11 +53,7 @@ def build_spec(variant: str) -> BenchmarkSpec: f"```rust\n" f"{EXPECTED_CONTENT}```\n" ) - retry = ( - 'Use `read(path="test.rs")` to refresh chunk selectors if needed, then try again using the edit tool.' - if variant == "chunk" - else f"Please try again using the edit tool {mode_phrase}." - ) + retry = f"Please try again using the edit tool {mode_phrase}." return BenchmarkSpec( description=f"Benchmark edit tool in {variant} mode across models with simple edit tasks.", workspace_prefix=f"{variant}-benchmark", diff --git a/scripts/rate-edit-tool.py b/scripts/rate-edit-tool.py index 98e1b90bf..c624f73c6 100755 --- a/scripts/rate-edit-tool.py +++ b/scripts/rate-edit-tool.py @@ -720,7 +720,7 @@ MARKDOWN_FIXTURE = ( | Surface | Expected stress | | --- | --- | - | `main.ts` | Structural chunk addressing | + | `main.ts` | Type/interface and class member edits | | `main.rs` | Enum and impl member edits | | `main.py` | Indentation-sensitive blocks | | `main.md` | Prose and block-level text edits |