feat: removed chunk-mode modules and read/edit entrypoints from pi-natives
- Removed `pi-natives` chunk language classifier modules and all core chunk subsystems (kind, state, render, edit, resolve). - Removed chunk-mode CLI/read/edit entrypoints, including `read` command and chunk mode registration/prompt tooling. - Removed chunk selectors from `read` and `grep` tools, switching behavior to raw/L-range handling. - Fixed poll wait parsing to keep defaulting to `30s` when the provided value is empty.
This commit is contained in:
@@ -1,373 +1,12 @@
|
||||
use std::{
|
||||
collections::{BTreeMap, BTreeSet, HashMap},
|
||||
env,
|
||||
fmt::Write as _,
|
||||
fs,
|
||||
path::{Path, PathBuf},
|
||||
};
|
||||
|
||||
use serde::Deserialize;
|
||||
|
||||
const IDENTIFIER_FIELD_PRIORITY: &[&str] = &[
|
||||
"name",
|
||||
"identifier",
|
||||
"attrpath",
|
||||
"key",
|
||||
"label",
|
||||
"alias",
|
||||
"field",
|
||||
"member",
|
||||
"property",
|
||||
"tag",
|
||||
"target",
|
||||
"variable",
|
||||
];
|
||||
|
||||
const BODY_FIELD_PRIORITY: &[&str] = &["body", "value", "declaration_list", "block", "members"];
|
||||
const PROMOTION_FIELD_PRIORITY: &[&str] = &["definition", "declaration", "item", "member"];
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
struct GrammarSpec {
|
||||
language: &'static str,
|
||||
package: &'static str,
|
||||
node_types_rel: &'static str,
|
||||
}
|
||||
|
||||
struct LockedPackage {
|
||||
version: String,
|
||||
source: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct RawTypeRef {
|
||||
#[serde(rename = "type")]
|
||||
kind: Option<String>,
|
||||
named: bool,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct RawFieldSpec {
|
||||
#[serde(default)]
|
||||
multiple: bool,
|
||||
#[serde(default)]
|
||||
types: Vec<RawTypeRef>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct RawNodeType {
|
||||
#[serde(rename = "type")]
|
||||
kind: Option<String>,
|
||||
fields: Option<BTreeMap<String, RawFieldSpec>>,
|
||||
children: Option<RawFieldSpec>,
|
||||
subtypes: Option<Vec<RawTypeRef>>,
|
||||
}
|
||||
|
||||
#[derive(serde::Serialize)]
|
||||
struct GeneratedSchema {
|
||||
languages: BTreeMap<String, BTreeMap<String, GeneratedNodeTypeSchema>>,
|
||||
}
|
||||
|
||||
#[derive(serde::Serialize)]
|
||||
struct GeneratedNodeTypeSchema {
|
||||
identifier_fields: Vec<String>,
|
||||
body_fields: Vec<String>,
|
||||
promotion_fields: Vec<String>,
|
||||
container_child_kinds: Vec<String>,
|
||||
is_supertype: bool,
|
||||
has_structural_children: bool,
|
||||
}
|
||||
|
||||
const GRAMMARS: &[GrammarSpec] = &[
|
||||
GrammarSpec {
|
||||
language: "astro",
|
||||
package: "tree-sitter-astro-next",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "bash",
|
||||
package: "tree-sitter-bash",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "c",
|
||||
package: "tree-sitter-c",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "clojure",
|
||||
package: "tree-sitter-clojure",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "cmake",
|
||||
package: "tree-sitter-cmake",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "cpp",
|
||||
package: "tree-sitter-cpp",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "csharp",
|
||||
package: "tree-sitter-c-sharp",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "dart",
|
||||
package: "tree-sitter-dart",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "css",
|
||||
package: "tree-sitter-css",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "diff",
|
||||
package: "tree-sitter-diff",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "dockerfile",
|
||||
package: "tree-sitter-dockerfile-updated",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "elixir",
|
||||
package: "tree-sitter-elixir",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "erlang",
|
||||
package: "tree-sitter-erlang",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "go",
|
||||
package: "tree-sitter-go",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "graphql",
|
||||
package: "tree-sitter-graphql",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "handlebars",
|
||||
package: "tree-sitter-glimmer",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "haskell",
|
||||
package: "tree-sitter-haskell",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "hcl",
|
||||
package: "tree-sitter-hcl",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "html",
|
||||
package: "tree-sitter-html",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "ini",
|
||||
package: "tree-sitter-ini",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "java",
|
||||
package: "tree-sitter-java",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "javascript",
|
||||
package: "tree-sitter-javascript",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "json",
|
||||
package: "tree-sitter-json",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "toml",
|
||||
package: "tree-sitter-toml-ng",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "just",
|
||||
package: "tree-sitter-just",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "julia",
|
||||
package: "tree-sitter-julia",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "kotlin",
|
||||
package: "tree-sitter-kotlin-sg",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "lua",
|
||||
package: "tree-sitter-lua",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "make",
|
||||
package: "tree-sitter-make",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "markdown",
|
||||
package: "tree-sitter-md",
|
||||
node_types_rel: "tree-sitter-markdown/src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "nix",
|
||||
package: "tree-sitter-nix",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "objc",
|
||||
package: "tree-sitter-objc",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "odin",
|
||||
package: "tree-sitter-odin",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "perl",
|
||||
package: "tree-sitter-perl-next",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "php",
|
||||
package: "tree-sitter-php",
|
||||
node_types_rel: "php/src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "powershell",
|
||||
package: "tree-sitter-powershell",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "protobuf",
|
||||
package: "tree-sitter-proto",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "python",
|
||||
package: "tree-sitter-python",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "r",
|
||||
package: "tree-sitter-r",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "regex",
|
||||
package: "tree-sitter-regex",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "ruby",
|
||||
package: "tree-sitter-ruby",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "rust",
|
||||
package: "tree-sitter-rust",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "scala",
|
||||
package: "tree-sitter-scala",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "solidity",
|
||||
package: "tree-sitter-solidity",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "sql",
|
||||
package: "tree-sitter-sequel",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "starlark",
|
||||
package: "tree-sitter-starlark",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "svelte",
|
||||
package: "tree-sitter-svelte-next",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "swift",
|
||||
package: "tree-sitter-swift",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "toml",
|
||||
package: "tree-sitter-toml-ng",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "tlaplus",
|
||||
package: "tree-sitter-tlaplus",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "tsx",
|
||||
package: "tree-sitter-typescript",
|
||||
node_types_rel: "tsx/src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "typescript",
|
||||
package: "tree-sitter-typescript",
|
||||
node_types_rel: "typescript/src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "verilog",
|
||||
package: "tree-sitter-verilog",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "vue",
|
||||
package: "tree-sitter-vue-next",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "xml",
|
||||
package: "tree-sitter-xml",
|
||||
node_types_rel: "xml/src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "yaml",
|
||||
package: "tree-sitter-yaml",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
GrammarSpec {
|
||||
language: "zig",
|
||||
package: "tree-sitter-zig",
|
||||
node_types_rel: "src/node-types.json",
|
||||
},
|
||||
];
|
||||
|
||||
fn main() {
|
||||
napi_build::setup();
|
||||
generate_chunk_schema();
|
||||
generate_minimizer_builtin_filters();
|
||||
}
|
||||
|
||||
@@ -423,471 +62,3 @@ fn generate_minimizer_builtin_filters() {
|
||||
fs::write(&output_path, concatenated)
|
||||
.unwrap_or_else(|e| panic!("failed to write {}: {e}", output_path.display()));
|
||||
}
|
||||
|
||||
fn generate_chunk_schema() {
|
||||
let manifest_dir = env::var("CARGO_MANIFEST_DIR").expect("CARGO_MANIFEST_DIR should be set");
|
||||
let workspace_root = Path::new(&manifest_dir)
|
||||
.parent()
|
||||
.and_then(Path::parent)
|
||||
.expect("pi-natives should live under the workspace root");
|
||||
let out_dir = PathBuf::from(env::var("OUT_DIR").expect("OUT_DIR should be set"));
|
||||
let output_path = out_dir.join("chunk_schema.json");
|
||||
let locked_packages = locked_packages(&workspace_root.join("Cargo.lock"));
|
||||
let registry_roots = cargo_registry_roots();
|
||||
let git_roots = cargo_git_checkout_roots();
|
||||
let mut languages = BTreeMap::new();
|
||||
|
||||
for grammar in GRAMMARS {
|
||||
let Some(locked) = locked_packages.get(grammar.package) else {
|
||||
continue;
|
||||
};
|
||||
let Some(package_dir) =
|
||||
find_locked_package_dir(®istry_roots, &git_roots, grammar.package, locked)
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
let node_types_path = package_dir.join(grammar.node_types_rel);
|
||||
if !node_types_path.exists() {
|
||||
continue;
|
||||
}
|
||||
|
||||
println!("cargo:rerun-if-changed={}", node_types_path.display());
|
||||
let source =
|
||||
fs::read_to_string(&node_types_path).expect("node-types.json should be readable");
|
||||
let raw_nodes: Vec<RawNodeType> =
|
||||
serde_json::from_str(&source).expect("node-types.json should parse");
|
||||
let schemas = build_language_schema(raw_nodes);
|
||||
if !schemas.is_empty() {
|
||||
languages.insert(grammar.language.to_string(), schemas);
|
||||
}
|
||||
}
|
||||
|
||||
let generated = GeneratedSchema { languages };
|
||||
let json = serde_json::to_string(&generated).expect("schema JSON should serialize");
|
||||
fs::write(output_path, json).expect("schema JSON should write");
|
||||
}
|
||||
|
||||
fn build_language_schema(raw_nodes: Vec<RawNodeType>) -> BTreeMap<String, GeneratedNodeTypeSchema> {
|
||||
let mut raw_by_kind = HashMap::new();
|
||||
for raw in raw_nodes {
|
||||
let Some(kind) = raw.kind.clone() else {
|
||||
continue;
|
||||
};
|
||||
raw_by_kind.insert(kind, raw);
|
||||
}
|
||||
|
||||
let structural_state = compute_structural_state(&raw_by_kind);
|
||||
let mut out = BTreeMap::new();
|
||||
for (kind, raw) in &raw_by_kind {
|
||||
let identifier_fields = pick_priority_fields(raw.fields.as_ref(), IDENTIFIER_FIELD_PRIORITY);
|
||||
let body_fields = pick_priority_fields(raw.fields.as_ref(), BODY_FIELD_PRIORITY);
|
||||
let promotion_fields =
|
||||
collect_promotion_fields(raw, &structural_state, &identifier_fields, &body_fields);
|
||||
let container_child_kinds = collect_child_container_kinds(raw, &structural_state);
|
||||
let is_supertype = is_supertype(raw);
|
||||
let has_structural_children = structural_state
|
||||
.get(kind)
|
||||
.is_some_and(|state| state.has_structural_children);
|
||||
|
||||
if identifier_fields.is_empty()
|
||||
&& body_fields.is_empty()
|
||||
&& promotion_fields.is_empty()
|
||||
&& container_child_kinds.is_empty()
|
||||
&& !is_supertype
|
||||
&& !has_structural_children
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
out.insert(kind.clone(), GeneratedNodeTypeSchema {
|
||||
identifier_fields,
|
||||
body_fields,
|
||||
promotion_fields,
|
||||
container_child_kinds,
|
||||
is_supertype,
|
||||
has_structural_children,
|
||||
});
|
||||
}
|
||||
|
||||
out
|
||||
}
|
||||
|
||||
fn pick_priority_fields(
|
||||
fields: Option<&BTreeMap<String, RawFieldSpec>>,
|
||||
priority: &[&str],
|
||||
) -> Vec<String> {
|
||||
let Some(fields) = fields else {
|
||||
return Vec::new();
|
||||
};
|
||||
|
||||
priority
|
||||
.iter()
|
||||
.filter(|field| fields.contains_key(**field))
|
||||
.map(|field| (*field).to_string())
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn collect_child_container_kinds(
|
||||
raw: &RawNodeType,
|
||||
structural_state: &HashMap<String, StructuralState>,
|
||||
) -> Vec<String> {
|
||||
let mut kinds = BTreeSet::new();
|
||||
let child_types = raw
|
||||
.children
|
||||
.as_ref()
|
||||
.map(|children| children.types.as_slice())
|
||||
.unwrap_or_default();
|
||||
|
||||
for child in child_types {
|
||||
if !child.named {
|
||||
continue;
|
||||
}
|
||||
let Some(kind) = child.kind.as_deref() else {
|
||||
continue;
|
||||
};
|
||||
if structural_state
|
||||
.get(kind)
|
||||
.copied()
|
||||
.is_some_and(StructuralState::is_structural)
|
||||
{
|
||||
kinds.insert(kind.to_string());
|
||||
}
|
||||
}
|
||||
|
||||
kinds.into_iter().collect()
|
||||
}
|
||||
|
||||
fn collect_promotion_fields(
|
||||
raw: &RawNodeType,
|
||||
structural_state: &HashMap<String, StructuralState>,
|
||||
identifier_fields: &[String],
|
||||
body_fields: &[String],
|
||||
) -> Vec<String> {
|
||||
let Some(fields) = raw.fields.as_ref() else {
|
||||
return Vec::new();
|
||||
};
|
||||
|
||||
PROMOTION_FIELD_PRIORITY
|
||||
.iter()
|
||||
.filter_map(|field_name| {
|
||||
let spec = fields.get(*field_name)?;
|
||||
if spec.multiple
|
||||
|| identifier_fields.iter().any(|field| field == field_name)
|
||||
|| body_fields.iter().any(|field| field == field_name)
|
||||
{
|
||||
return None;
|
||||
}
|
||||
|
||||
let has_structural_type = spec.types.iter().any(|field_type| {
|
||||
field_type.named
|
||||
&& field_type
|
||||
.kind
|
||||
.as_deref()
|
||||
.and_then(|kind| structural_state.get(kind))
|
||||
.copied()
|
||||
.is_some_and(StructuralState::is_structural)
|
||||
});
|
||||
has_structural_type.then(|| (*field_name).to_string())
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Default)]
|
||||
struct StructuralState {
|
||||
is_structural: bool,
|
||||
has_structural_children: bool,
|
||||
}
|
||||
|
||||
impl StructuralState {
|
||||
const fn is_structural(self) -> bool {
|
||||
self.is_structural
|
||||
}
|
||||
}
|
||||
|
||||
fn compute_structural_state(
|
||||
raw_by_kind: &HashMap<String, RawNodeType>,
|
||||
) -> HashMap<String, StructuralState> {
|
||||
let mut state = raw_by_kind
|
||||
.iter()
|
||||
.map(|(kind, raw)| {
|
||||
let base_structural = is_supertype(raw)
|
||||
|| raw.fields.as_ref().is_some_and(|fields| !fields.is_empty())
|
||||
|| !named_child_type_kinds(raw).is_empty();
|
||||
(kind.clone(), StructuralState {
|
||||
is_structural: base_structural,
|
||||
has_structural_children: false,
|
||||
})
|
||||
})
|
||||
.collect::<HashMap<_, _>>();
|
||||
|
||||
loop {
|
||||
let mut changed = false;
|
||||
for (kind, raw) in raw_by_kind {
|
||||
let next_has_structural_children =
|
||||
named_child_type_kinds(raw).into_iter().any(|child_kind| {
|
||||
state
|
||||
.get(child_kind.as_str())
|
||||
.copied()
|
||||
.is_some_and(StructuralState::is_structural)
|
||||
});
|
||||
|
||||
let entry = state
|
||||
.get_mut(kind.as_str())
|
||||
.expect("every raw node should have structural state");
|
||||
let next_is_structural = entry.is_structural || next_has_structural_children;
|
||||
if next_is_structural != entry.is_structural
|
||||
|| next_has_structural_children != entry.has_structural_children
|
||||
{
|
||||
entry.is_structural = next_is_structural;
|
||||
entry.has_structural_children = next_has_structural_children;
|
||||
changed = true;
|
||||
}
|
||||
}
|
||||
if !changed {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
state
|
||||
}
|
||||
|
||||
fn is_supertype(raw: &RawNodeType) -> bool {
|
||||
raw.subtypes
|
||||
.as_ref()
|
||||
.is_some_and(|subtypes| !subtypes.is_empty())
|
||||
}
|
||||
|
||||
fn named_child_type_kinds(raw: &RawNodeType) -> BTreeSet<String> {
|
||||
let mut kinds = BTreeSet::new();
|
||||
if let Some(fields) = raw.fields.as_ref() {
|
||||
for field in fields.values() {
|
||||
for field_type in &field.types {
|
||||
if field_type.named
|
||||
&& let Some(kind) = &field_type.kind
|
||||
{
|
||||
kinds.insert(kind.clone());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(children) = raw.children.as_ref() {
|
||||
for child in &children.types {
|
||||
if child.named
|
||||
&& let Some(kind) = &child.kind
|
||||
{
|
||||
kinds.insert(kind.clone());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
kinds
|
||||
}
|
||||
|
||||
fn cargo_registry_roots() -> Vec<PathBuf> {
|
||||
let mut roots = Vec::new();
|
||||
if let Some(cargo_home) = env::var_os("CARGO_HOME") {
|
||||
roots.push(PathBuf::from(cargo_home).join("registry").join("src"));
|
||||
}
|
||||
if let Some(home) = env::var_os("HOME") {
|
||||
roots.push(
|
||||
PathBuf::from(home)
|
||||
.join(".cargo")
|
||||
.join("registry")
|
||||
.join("src"),
|
||||
);
|
||||
}
|
||||
roots
|
||||
}
|
||||
|
||||
fn cargo_git_checkout_roots() -> Vec<PathBuf> {
|
||||
let mut roots = Vec::new();
|
||||
if let Some(cargo_home) = env::var_os("CARGO_HOME") {
|
||||
roots.push(PathBuf::from(cargo_home).join("git").join("checkouts"));
|
||||
}
|
||||
if let Some(home) = env::var_os("HOME") {
|
||||
roots.push(
|
||||
PathBuf::from(home)
|
||||
.join(".cargo")
|
||||
.join("git")
|
||||
.join("checkouts"),
|
||||
);
|
||||
}
|
||||
roots
|
||||
}
|
||||
|
||||
fn find_locked_package_dir(
|
||||
registry_roots: &[PathBuf],
|
||||
git_roots: &[PathBuf],
|
||||
package: &str,
|
||||
locked: &LockedPackage,
|
||||
) -> Option<PathBuf> {
|
||||
match locked.source.as_deref() {
|
||||
Some(source) if source.starts_with("git+") => {
|
||||
find_git_package_dir(git_roots, package, &locked.version, git_revision(source))
|
||||
},
|
||||
_ => find_registry_package_dir(registry_roots, package, &locked.version),
|
||||
}
|
||||
}
|
||||
|
||||
fn find_registry_package_dir(
|
||||
registry_roots: &[PathBuf],
|
||||
package: &str,
|
||||
version: &str,
|
||||
) -> Option<PathBuf> {
|
||||
for registry_root in registry_roots {
|
||||
let Ok(registry_dirs) = fs::read_dir(registry_root) else {
|
||||
continue;
|
||||
};
|
||||
for registry_dir in registry_dirs.flatten() {
|
||||
let candidate = registry_dir.path().join(format!("{package}-{version}"));
|
||||
if candidate.exists() {
|
||||
return Some(candidate);
|
||||
}
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn find_git_package_dir(
|
||||
git_roots: &[PathBuf],
|
||||
package: &str,
|
||||
version: &str,
|
||||
revision: Option<&str>,
|
||||
) -> Option<PathBuf> {
|
||||
for git_root in git_roots {
|
||||
let Ok(checkout_dirs) = fs::read_dir(git_root) else {
|
||||
continue;
|
||||
};
|
||||
for checkout_dir in checkout_dirs.flatten() {
|
||||
let Ok(revision_dirs) = fs::read_dir(checkout_dir.path()) else {
|
||||
continue;
|
||||
};
|
||||
for revision_dir in revision_dirs.flatten() {
|
||||
let revision_path = revision_dir.path();
|
||||
let Some(revision_name) = revision_path.file_name().and_then(|name| name.to_str())
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
if !revision_matches(revision_name, revision) {
|
||||
continue;
|
||||
}
|
||||
if let Some(package_dir) = find_manifest_package_dir(&revision_path, package, version) {
|
||||
return Some(package_dir);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn revision_matches(revision_name: &str, revision: Option<&str>) -> bool {
|
||||
revision.is_none_or(|revision| {
|
||||
revision.starts_with(revision_name) || revision_name.starts_with(revision)
|
||||
})
|
||||
}
|
||||
|
||||
fn find_manifest_package_dir(root: &Path, package: &str, version: &str) -> Option<PathBuf> {
|
||||
if manifest_matches_package(&root.join("Cargo.toml"), package, version) {
|
||||
return Some(root.to_path_buf());
|
||||
}
|
||||
|
||||
let Ok(entries) = fs::read_dir(root) else {
|
||||
return None;
|
||||
};
|
||||
for entry in entries.flatten() {
|
||||
let candidate = entry.path();
|
||||
if candidate.is_dir()
|
||||
&& manifest_matches_package(&candidate.join("Cargo.toml"), package, version)
|
||||
{
|
||||
return Some(candidate);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn manifest_matches_package(manifest_path: &Path, package: &str, version: &str) -> bool {
|
||||
let Ok(source) = fs::read_to_string(manifest_path) else {
|
||||
return false;
|
||||
};
|
||||
let mut in_package = false;
|
||||
let mut name_matches = false;
|
||||
let mut version_matches = false;
|
||||
|
||||
for line in source.lines() {
|
||||
let trimmed = line.trim();
|
||||
if trimmed.starts_with('[') {
|
||||
in_package = trimmed == "[package]";
|
||||
continue;
|
||||
}
|
||||
if !in_package {
|
||||
continue;
|
||||
}
|
||||
if let Some(value) = toml_string_value(trimmed, "name") {
|
||||
name_matches = value == package;
|
||||
continue;
|
||||
}
|
||||
if let Some(value) = toml_string_value(trimmed, "version") {
|
||||
version_matches = value == version;
|
||||
}
|
||||
}
|
||||
|
||||
name_matches && version_matches
|
||||
}
|
||||
|
||||
fn git_revision(source: &str) -> Option<&str> {
|
||||
source.rsplit_once('#').and_then(|(_, revision)| {
|
||||
if revision.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(revision)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn locked_packages(lock_path: &Path) -> HashMap<String, LockedPackage> {
|
||||
let source = fs::read_to_string(lock_path).expect("Cargo.lock should be readable");
|
||||
let mut packages = HashMap::new();
|
||||
let mut current_name = None;
|
||||
let mut current_version = None;
|
||||
let mut current_source = None;
|
||||
|
||||
for line in source.lines() {
|
||||
let trimmed = line.trim();
|
||||
if trimmed == "[[package]]" {
|
||||
if let (Some(name), Some(version)) = (current_name.take(), current_version.take()) {
|
||||
packages.insert(name, LockedPackage { version, source: current_source.take() });
|
||||
}
|
||||
current_source = None;
|
||||
continue;
|
||||
}
|
||||
if let Some(value) = toml_string_value(trimmed, "name") {
|
||||
current_name = Some(value.to_string());
|
||||
continue;
|
||||
}
|
||||
if let Some(value) = toml_string_value(trimmed, "version") {
|
||||
current_version = Some(value.to_string());
|
||||
continue;
|
||||
}
|
||||
if let Some(value) = toml_string_value(trimmed, "source") {
|
||||
current_source = Some(value.to_string());
|
||||
}
|
||||
}
|
||||
|
||||
if let (Some(name), Some(version)) = (current_name, current_version) {
|
||||
packages.insert(name, LockedPackage { version, source: current_source });
|
||||
}
|
||||
|
||||
packages
|
||||
}
|
||||
|
||||
fn toml_string_value<'a>(line: &'a str, key: &str) -> Option<&'a str> {
|
||||
let value = line
|
||||
.strip_prefix(key)?
|
||||
.trim_start()
|
||||
.strip_prefix('=')?
|
||||
.trim_start()
|
||||
.strip_prefix('"')?;
|
||||
let end = value.find('"')?;
|
||||
Some(&value[..end])
|
||||
}
|
||||
|
||||
@@ -1,231 +0,0 @@
|
||||
//! Language-specific chunk classifiers for Astro.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
use crate::language::SupportLang;
|
||||
|
||||
pub struct AstroClassifier;
|
||||
|
||||
impl LangClassifier for AstroClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &["document"],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
_context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
classify_astro_node(node, source)
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_astro_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"frontmatter" => Some(classify_frontmatter(node, source)),
|
||||
"frontmatter_js_block" => Some(group_candidate(node, ChunkKind::Code, source)),
|
||||
"element" => classify_element(node, source),
|
||||
"script_element" => Some(classify_script_element(node, source)),
|
||||
"style_element" => Some(classify_style_element(node, source)),
|
||||
"html_interpolation" => Some(classify_html_interpolation(node, source)),
|
||||
"attribute_interpolation" => Some(classify_attribute_interpolation(node, source)),
|
||||
"attribute_js_expr" => Some(group_candidate(node, ChunkKind::Expression, source)),
|
||||
"text" => Some(group_candidate(node, ChunkKind::Text, source)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_frontmatter<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let Some(content_node) = child_by_kind(node, &["frontmatter_js_block"]) else {
|
||||
return make_kind_chunk(node, ChunkKind::Frontmatter, None, source, None);
|
||||
};
|
||||
let candidate = with_region_node(
|
||||
make_kind_chunk(node, ChunkKind::Frontmatter, None, source, None),
|
||||
Some(content_node),
|
||||
);
|
||||
with_injected_subtree(candidate, SupportLang::TypeScript, content_node)
|
||||
}
|
||||
|
||||
fn classify_element<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
let tag_name = extract_tag_name(node, source)?;
|
||||
let recurse = Some(recurse_self(node, ChunkContext::ClassBody));
|
||||
if is_component_name(tag_name.as_str()) {
|
||||
Some(force_container(make_explicit_candidate(
|
||||
node,
|
||||
ChunkKind::Tag,
|
||||
format!("component_{tag_name}"),
|
||||
source,
|
||||
recurse,
|
||||
)))
|
||||
} else {
|
||||
Some(force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Tag,
|
||||
Some(tag_name),
|
||||
source,
|
||||
recurse,
|
||||
)))
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let identifier = has_attribute(node, "is:inline", source).then_some("inline".to_string());
|
||||
classify_raw_text_block(node, ChunkKind::Script, identifier, source, SupportLang::TypeScript)
|
||||
}
|
||||
|
||||
fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let identifier = if has_attribute(node, "define:vars", source) {
|
||||
Some("vars".to_string())
|
||||
} else if has_attribute(node, "is:global", source) {
|
||||
Some("global".to_string())
|
||||
} else {
|
||||
None
|
||||
};
|
||||
classify_raw_text_block(node, ChunkKind::Style, identifier, source, SupportLang::Css)
|
||||
}
|
||||
|
||||
fn classify_html_interpolation<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let identifier = child_by_kind(node, &["permissible_text"])
|
||||
.and_then(|expr| sanitize_identifier(node_text(source, expr.start_byte(), expr.end_byte())));
|
||||
|
||||
if let Some(nested_element) =
|
||||
child_by_kind(node, &["element", "script_element", "style_element"])
|
||||
{
|
||||
force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Expression,
|
||||
identifier,
|
||||
source,
|
||||
Some(recurse_self(nested_element, ChunkContext::ClassBody)),
|
||||
))
|
||||
} else {
|
||||
make_kind_chunk(node, ChunkKind::Expression, identifier, source, None)
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_attribute_interpolation<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let identifier = child_by_kind(node, &["attribute_js_expr"])
|
||||
.and_then(|expr| sanitize_identifier(node_text(source, expr.start_byte(), expr.end_byte())))
|
||||
.map_or_else(|| "expr".to_string(), |expr| format!("expr_{expr}"));
|
||||
make_kind_chunk(node, ChunkKind::Attr, Some(identifier), source, None)
|
||||
}
|
||||
|
||||
fn make_explicit_candidate<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
identifier: impl Into<Option<String>>,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'t>>,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
make_candidate(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse,
|
||||
source,
|
||||
)
|
||||
}
|
||||
|
||||
const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> {
|
||||
candidate.force_recurse = true;
|
||||
candidate
|
||||
}
|
||||
|
||||
fn classify_raw_text_block<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
identifier: Option<String>,
|
||||
source: &str,
|
||||
default_language: SupportLang,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
let Some(content_node) = child_by_kind(node, &["raw_text"]) else {
|
||||
return make_kind_chunk(node, kind, identifier, source, None);
|
||||
};
|
||||
let candidate =
|
||||
with_region_node(make_kind_chunk(node, kind, identifier, source, None), Some(content_node));
|
||||
match resolve_embedded_language(node, source, default_language) {
|
||||
Some(language) => with_injected_subtree(candidate, language, content_node),
|
||||
None => candidate,
|
||||
}
|
||||
}
|
||||
|
||||
fn extract_tag_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["start_tag", "self_closing_tag"])
|
||||
.and_then(|tag| child_by_kind(tag, &["tag_name"]))
|
||||
.and_then(|tag_name| {
|
||||
sanitize_identifier(node_text(source, tag_name.start_byte(), tag_name.end_byte()))
|
||||
})
|
||||
}
|
||||
|
||||
fn has_attribute(node: Node<'_>, name: &str, source: &str) -> bool {
|
||||
child_by_kind(node, &["start_tag", "self_closing_tag"])
|
||||
.into_iter()
|
||||
.flat_map(named_children)
|
||||
.filter(|child| child.kind() == "attribute")
|
||||
.filter_map(|attr| extract_attribute_name(attr, source))
|
||||
.any(|attr_name| attr_name == name)
|
||||
}
|
||||
|
||||
fn extract_attribute_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["attribute_name"]).map(|name| {
|
||||
node_text(source, name.start_byte(), name.end_byte())
|
||||
.trim()
|
||||
.to_string()
|
||||
})
|
||||
}
|
||||
|
||||
fn resolve_embedded_language(
|
||||
node: Node<'_>,
|
||||
source: &str,
|
||||
default_language: SupportLang,
|
||||
) -> Option<SupportLang> {
|
||||
if let Some(language) = attribute_value(node, "lang", source) {
|
||||
return SupportLang::from_alias(language.as_str());
|
||||
}
|
||||
Some(default_language)
|
||||
}
|
||||
|
||||
fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option<String> {
|
||||
let start = child_by_kind(node, &["start_tag", "self_closing_tag"])?;
|
||||
for child in named_children(start) {
|
||||
if child.kind() != "attribute" {
|
||||
continue;
|
||||
}
|
||||
if extract_attribute_name(child, source).as_deref() != Some(name) {
|
||||
continue;
|
||||
}
|
||||
if let Some(value) = child_by_kind(child, &["attribute_value", "quoted_attribute_value"]) {
|
||||
return sanitize_identifier(&unquote_text(node_text(
|
||||
source,
|
||||
value.start_byte(),
|
||||
value.end_byte(),
|
||||
)));
|
||||
}
|
||||
return Some(name.to_string());
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn is_component_name(tag_name: &str) -> bool {
|
||||
tag_name.chars().next().is_some_and(char::is_uppercase)
|
||||
}
|
||||
@@ -1,275 +0,0 @@
|
||||
//! Language-specific chunk classifiers for Bash, Make, and Diff.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct ShellBuildClassifier;
|
||||
|
||||
impl ShellBuildClassifier {
|
||||
/// Extract a Make rule target name (child node of kind `targets`).
|
||||
fn extract_rule_target(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["targets"])
|
||||
.and_then(|t| sanitize_identifier(node_text(source, t.start_byte(), t.end_byte())))
|
||||
}
|
||||
|
||||
/// Extract a Make variable/define name (field `name`).
|
||||
fn extract_var_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
node
|
||||
.child_by_field_name("name")
|
||||
.and_then(|n| sanitize_identifier(node_text(source, n.start_byte(), n.end_byte())))
|
||||
}
|
||||
|
||||
/// Strip the conventional `a/` or `b/` prefix from git diff paths.
|
||||
fn strip_ab_prefix(path: &str) -> &str {
|
||||
path
|
||||
.strip_prefix("a/")
|
||||
.or_else(|| path.strip_prefix("b/"))
|
||||
.unwrap_or(path)
|
||||
}
|
||||
|
||||
/// Extract the file path from a diff `block` node.
|
||||
///
|
||||
/// Extraction priority:
|
||||
/// 1. `new_file` child -> `filename` child text (skip if `/dev/null`)
|
||||
/// 2. `old_file` child -> `filename` child text (skip if `/dev/null`)
|
||||
/// 3. `command` child -> parse `a/path b/path` from filename children
|
||||
fn extract_diff_filename(node: Node<'_>, source: &str) -> Option<String> {
|
||||
// Try new_file first (most diffs have it)
|
||||
if let Some(new_file) = child_by_kind(node, &["new_file"])
|
||||
&& let Some(filename) = child_by_kind(new_file, &["filename"])
|
||||
{
|
||||
let text = node_text(source, filename.start_byte(), filename.end_byte()).trim();
|
||||
if text != "/dev/null" {
|
||||
return sanitize_identifier(Self::strip_ab_prefix(text));
|
||||
}
|
||||
}
|
||||
|
||||
// Fall back to old_file (deleted files)
|
||||
if let Some(old_file) = child_by_kind(node, &["old_file"])
|
||||
&& let Some(filename) = child_by_kind(old_file, &["filename"])
|
||||
{
|
||||
let text = node_text(source, filename.start_byte(), filename.end_byte()).trim();
|
||||
if text != "/dev/null" {
|
||||
return sanitize_identifier(Self::strip_ab_prefix(text));
|
||||
}
|
||||
}
|
||||
|
||||
// Last resort: extract from the `command` line ("diff --git a/path b/path").
|
||||
// The grammar's `filename` rule is `repeat1(/\S+/)`, so it captures both
|
||||
// paths as a single node like "a/foo.ts b/foo.ts". Take the last
|
||||
// space-delimited segment (the b-side path).
|
||||
if let Some(command) = child_by_kind(node, &["command"])
|
||||
&& let Some(filename) = child_by_kind(command, &["filename"])
|
||||
{
|
||||
let text = node_text(source, filename.start_byte(), filename.end_byte()).trim();
|
||||
let b_side = text.rsplit_once(' ').map_or(text, |(_, b)| b);
|
||||
return sanitize_identifier(Self::strip_ab_prefix(b_side));
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
impl LangClassifier for ShellBuildClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[
|
||||
semantic_rule(
|
||||
"conditional",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"command",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"pipeline",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"if_statement",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"case_statement",
|
||||
ChunkKind::Switch,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"while_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"for_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"function_definition",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"hunks",
|
||||
ChunkKind::Hunks,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
],
|
||||
class: &[semantic_rule(
|
||||
"hunk",
|
||||
ChunkKind::Hunk,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
)],
|
||||
function: &[
|
||||
semantic_rule(
|
||||
"if_statement",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"case_statement",
|
||||
ChunkKind::Switch,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"while_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"for_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"command",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"pipeline",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"subshell",
|
||||
ChunkKind::Block,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &["makefile"],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_root_custom(node, source),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn preserve_children(
|
||||
&self,
|
||||
_parent: &RawChunkCandidate<'_>,
|
||||
children: &[RawChunkCandidate<'_>],
|
||||
) -> bool {
|
||||
// Diff file blocks should always preserve hunk children
|
||||
children.iter().any(|c| c.kind == ChunkKind::Hunk)
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"rule" => {
|
||||
let name = ShellBuildClassifier::extract_rule_target(node, source)
|
||||
.unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Rule,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["recipe"]),
|
||||
))
|
||||
},
|
||||
"variable_assignment" | "shell_assignment" => {
|
||||
let name = ShellBuildClassifier::extract_var_name(node, source)
|
||||
.unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None))
|
||||
},
|
||||
"define_directive" => {
|
||||
let name = ShellBuildClassifier::extract_var_name(node, source)
|
||||
.unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_kind_chunk(node, ChunkKind::Define, Some(name), source, None))
|
||||
},
|
||||
"block" => {
|
||||
let identifier = ShellBuildClassifier::extract_diff_filename(node, source);
|
||||
let recurse = recurse_into(node, ChunkContext::ClassBody, &[], &["hunks"]);
|
||||
let mut candidate =
|
||||
make_container_chunk(node, ChunkKind::File, identifier, source, recurse);
|
||||
// Always expand hunks so individual @@ sections are addressable,
|
||||
// even for small diffs below the leaf threshold.
|
||||
candidate.force_recurse = recurse.is_some();
|
||||
Some(candidate)
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
@@ -1,431 +0,0 @@
|
||||
//! Language-specific chunk classifiers for C, C++, and Objective-C.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
defaults::classify_var_decl,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct CCppClassifier;
|
||||
|
||||
/// Extract the function name from a C/C++ `function_definition` or
|
||||
/// `function_declaration` node by traversing into the `declarator` chain.
|
||||
fn extract_c_function_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let decl = node.child_by_field_name("declarator")?;
|
||||
extract_c_declarator_name(decl, source)
|
||||
}
|
||||
|
||||
/// Recursively resolve a C/C++ declarator to its leaf identifier.
|
||||
/// Handles `function_declarator`, `pointer_declarator`, `reference_declarator`,
|
||||
/// `qualified_identifier`, `destructor_name`, `template_function`, etc.
|
||||
fn extract_c_declarator_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
match node.kind() {
|
||||
"identifier" | "field_identifier" | "type_identifier" => {
|
||||
sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()))
|
||||
},
|
||||
"destructor_name" => {
|
||||
// ~ClassName
|
||||
sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()))
|
||||
},
|
||||
"qualified_identifier" | "scoped_identifier" => {
|
||||
// e.g. Entity::update — extract the "name" field or last identifier
|
||||
node
|
||||
.child_by_field_name("name")
|
||||
.and_then(|n| extract_c_declarator_name(n, source))
|
||||
.or_else(|| {
|
||||
named_children(node)
|
||||
.into_iter()
|
||||
.rev()
|
||||
.find(|c| {
|
||||
matches!(
|
||||
c.kind(),
|
||||
"identifier" | "destructor_name" | "template_function" | "field_identifier"
|
||||
)
|
||||
})
|
||||
.and_then(|c| extract_c_declarator_name(c, source))
|
||||
})
|
||||
},
|
||||
"template_function" => {
|
||||
// template_function has a "name" field or direct identifier child
|
||||
node
|
||||
.child_by_field_name("name")
|
||||
.and_then(|n| sanitize_identifier(node_text(source, n.start_byte(), n.end_byte())))
|
||||
.or_else(|| {
|
||||
named_children(node)
|
||||
.into_iter()
|
||||
.find(|c| c.kind() == "identifier")
|
||||
.and_then(|c| {
|
||||
sanitize_identifier(node_text(source, c.start_byte(), c.end_byte()))
|
||||
})
|
||||
})
|
||||
},
|
||||
_ => {
|
||||
// function_declarator, pointer_declarator, reference_declarator, etc.
|
||||
// recurse into the "declarator" field
|
||||
node
|
||||
.child_by_field_name("declarator")
|
||||
.and_then(|inner| extract_c_declarator_name(inner, source))
|
||||
.or_else(|| {
|
||||
// fallback: look for direct identifier-like child
|
||||
named_children(node)
|
||||
.into_iter()
|
||||
.find(|c| {
|
||||
matches!(
|
||||
c.kind(),
|
||||
"identifier"
|
||||
| "field_identifier"
|
||||
| "qualified_identifier"
|
||||
| "scoped_identifier"
|
||||
| "destructor_name"
|
||||
| "template_function"
|
||||
)
|
||||
})
|
||||
.and_then(|c| extract_c_declarator_name(c, source))
|
||||
})
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract the field name from a C/C++ `field_declaration` node.
|
||||
/// The name sits in the `declarator` field which may be a plain
|
||||
/// `field_identifier`, or a `function_declarator` / `pointer_declarator` etc.
|
||||
fn extract_c_field_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let decl = node.child_by_field_name("declarator")?;
|
||||
extract_c_declarator_name(decl, source)
|
||||
}
|
||||
|
||||
const C_CPP_ROOT_RULES: &[super::classify::SemanticRule] = &[
|
||||
// ── Imports ──
|
||||
semantic_rule(
|
||||
"include_directive",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"preproc_include",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"using_directive",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"using_statement",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"import_declaration",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"module_import",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
// ── Statements ──
|
||||
semantic_rule(
|
||||
"expression_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const C_CPP_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: C_CPP_ROOT_RULES,
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
impl LangClassifier for CCppClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&C_CPP_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_root_custom(self, node, source),
|
||||
ChunkContext::ClassBody => classify_class_custom(self, node, source),
|
||||
ChunkContext::FunctionBody => Some(classify_function_c(node, source)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_root_custom<'t>(
|
||||
classifier: &CCppClassifier,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// ── Functions ──
|
||||
"function_definition" | "function_declaration" => Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
extract_c_function_name(node, source),
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
)),
|
||||
"constructor_definition" => Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Constructor,
|
||||
None,
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
)),
|
||||
|
||||
// ── Templates (unwrap to find the inner declaration) ──
|
||||
"template_declaration" => {
|
||||
// Find the inner function_definition / class_specifier / etc.
|
||||
let inner = named_children(node).into_iter().find(|c| {
|
||||
matches!(
|
||||
c.kind(),
|
||||
"function_definition"
|
||||
| "function_declaration"
|
||||
| "class_specifier"
|
||||
| "struct_specifier"
|
||||
| "type_alias_declaration"
|
||||
)
|
||||
});
|
||||
match inner {
|
||||
Some(inner) => {
|
||||
let mut candidate =
|
||||
classifier.classify_override(ChunkContext::Root, inner, source)?;
|
||||
// Expand range to include the template<...> prefix
|
||||
candidate.range_start_byte = node.start_byte();
|
||||
candidate.range_start_line = node.start_position().row + 1;
|
||||
candidate.checksum_start_byte = node.start_byte();
|
||||
Some(candidate)
|
||||
},
|
||||
None => Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::Template,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
source,
|
||||
)),
|
||||
}
|
||||
},
|
||||
|
||||
// ── Containers ──
|
||||
"class_specifier" | "class_declaration" | "class_interface" | "class_implementation" => {
|
||||
Some(container_candidate(node, ChunkKind::Class, source, recurse_class(node)))
|
||||
},
|
||||
"struct_specifier" | "struct_declaration" => {
|
||||
Some(container_candidate(node, ChunkKind::Struct, source, recurse_class(node)))
|
||||
},
|
||||
"enum_specifier" | "enum_declaration" => {
|
||||
Some(container_candidate(node, ChunkKind::Enum, source, recurse_enum(node)))
|
||||
},
|
||||
"namespace_definition" => {
|
||||
Some(container_candidate(node, ChunkKind::Module, source, recurse_class(node)))
|
||||
},
|
||||
"union_declaration" => {
|
||||
Some(container_candidate(node, ChunkKind::Union, source, recurse_class(node)))
|
||||
},
|
||||
|
||||
// ── Types ──
|
||||
"type_alias_declaration" | "user_defined_type_definition" => {
|
||||
Some(named_candidate(node, ChunkKind::Type, source, recurse_class(node)))
|
||||
},
|
||||
|
||||
// ── Variables / assignments ──
|
||||
"variable_declaration" => Some(classify_var_decl(node, source)),
|
||||
"assignment_statement" | "property_declaration" => {
|
||||
Some(group_candidate(node, ChunkKind::Declarations, source))
|
||||
},
|
||||
|
||||
// ── Macros ──
|
||||
"macro_definition" => Some(named_candidate(
|
||||
node,
|
||||
ChunkKind::Macro,
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
)),
|
||||
|
||||
// ── Control flow (top-level scripts) ──
|
||||
"if_statement" | "switch_statement" | "for_statement" | "while_statement"
|
||||
| "do_statement" | "try_block" => Some(classify_function_c(node, source)),
|
||||
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_class_custom<'t>(
|
||||
classifier: &CCppClassifier,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// ── Methods ──
|
||||
"function_definition" | "function_declaration" | "method_declaration" => {
|
||||
let name = extract_c_function_name(node, source)
|
||||
.or_else(|| extract_identifier(node, source))
|
||||
.unwrap_or_else(|| "anonymous".to_string());
|
||||
if name == "constructor" {
|
||||
Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Constructor,
|
||||
None,
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
))
|
||||
} else {
|
||||
Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
))
|
||||
}
|
||||
},
|
||||
|
||||
// ── Constructors ──
|
||||
"constructor_definition" | "constructor_declaration" => Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Constructor,
|
||||
None,
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
)),
|
||||
|
||||
// ── Fields ──
|
||||
"field_declaration" => Some(match extract_c_field_name(node, source) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Fields, source),
|
||||
}),
|
||||
|
||||
// ── Enum variants ──
|
||||
"enum_constant" => Some(match extract_identifier(node, source) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Variants, source),
|
||||
}),
|
||||
|
||||
// ── Nested containers ──
|
||||
"class_specifier" | "class_declaration" | "class_interface" | "class_implementation" => {
|
||||
Some(container_candidate(node, ChunkKind::Class, source, recurse_class(node)))
|
||||
},
|
||||
"struct_specifier" | "struct_declaration" => {
|
||||
Some(container_candidate(node, ChunkKind::Struct, source, recurse_class(node)))
|
||||
},
|
||||
"enum_specifier" | "enum_declaration" => {
|
||||
Some(container_candidate(node, ChunkKind::Enum, source, recurse_enum(node)))
|
||||
},
|
||||
"union_declaration" => {
|
||||
Some(container_candidate(node, ChunkKind::Union, source, recurse_class(node)))
|
||||
},
|
||||
"namespace_definition" => {
|
||||
Some(container_candidate(node, ChunkKind::Module, source, recurse_class(node)))
|
||||
},
|
||||
|
||||
// ── Templates (class body) ──
|
||||
"template_declaration" => {
|
||||
let inner = named_children(node).into_iter().find(|c| {
|
||||
matches!(
|
||||
c.kind(),
|
||||
"function_definition"
|
||||
| "function_declaration"
|
||||
| "class_specifier"
|
||||
| "struct_specifier"
|
||||
| "type_alias_declaration"
|
||||
)
|
||||
});
|
||||
match inner {
|
||||
Some(inner) => {
|
||||
let mut candidate =
|
||||
classifier.classify_override(ChunkContext::ClassBody, inner, source)?;
|
||||
candidate.range_start_byte = node.start_byte();
|
||||
candidate.range_start_line = node.start_position().row + 1;
|
||||
candidate.checksum_start_byte = node.start_byte();
|
||||
Some(candidate)
|
||||
},
|
||||
None => Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::Template,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
source,
|
||||
)),
|
||||
}
|
||||
},
|
||||
|
||||
// ── Types ──
|
||||
"type_alias_declaration" => Some(named_candidate(node, ChunkKind::Type, source, None)),
|
||||
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_function_c<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
|
||||
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
|
||||
match node.kind() {
|
||||
"if_statement" => {
|
||||
make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"switch_statement" => {
|
||||
make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"try_block" | "catch_clause" | "finally_clause" => {
|
||||
make_candidate(node, ChunkKind::Try, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"for_statement" => {
|
||||
make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"while_statement" => {
|
||||
make_candidate(node, ChunkKind::While, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"do_statement" => {
|
||||
make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"variable_declaration" => {
|
||||
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
|
||||
if span > 1 {
|
||||
if let Some(name) = extract_single_declarator_name(node, source) {
|
||||
make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)
|
||||
} else {
|
||||
group_candidate(node, ChunkKind::Variable, source)
|
||||
}
|
||||
} else {
|
||||
group_candidate(node, ChunkKind::Variable, source)
|
||||
}
|
||||
},
|
||||
_ => {
|
||||
let kind_name = sanitize_node_kind(node.kind());
|
||||
let kind = ChunkKind::from_sanitized_kind(kind_name);
|
||||
group_candidate(node, kind, source)
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -1,78 +0,0 @@
|
||||
//! Language-specific chunk classifier for Clojure.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct ClojureClassifier;
|
||||
|
||||
/// Extract the head symbol of a Clojure list form (first
|
||||
/// `sym_lit`/`kwd_lit`/`symbol`/`word`).
|
||||
fn form_head(node: Node<'_>, source: &str) -> Option<String> {
|
||||
named_children(node).into_iter().find_map(|child| {
|
||||
matches!(child.kind(), "sym_lit" | "kwd_lit" | "symbol" | "word")
|
||||
.then(|| node_text(source, child.start_byte(), child.end_byte()).to_string())
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract the name from a Clojure form: the second named child after the head
|
||||
/// symbol (e.g. `greet` in `(defn greet [x] x)`).
|
||||
fn form_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let mut children = named_children(node).into_iter();
|
||||
let _head = children.next()?;
|
||||
children.find_map(|child| {
|
||||
sanitize_identifier(node_text(source, child.start_byte(), child.end_byte()))
|
||||
})
|
||||
}
|
||||
|
||||
/// Classify a `list_lit` Clojure form based on its head symbol.
|
||||
fn classify_form<'t>(node: Node<'t>, source: &str, at_root: bool) -> RawChunkCandidate<'t> {
|
||||
let Some(head) = form_head(node, source) else {
|
||||
return positional_candidate(node, ChunkKind::Form, source);
|
||||
};
|
||||
match head.as_str() {
|
||||
"ns" | "require" | "use" | "import" | "refer-clojure" => {
|
||||
group_candidate(node, ChunkKind::Imports, source)
|
||||
},
|
||||
"defn" | "defn-" | "defmacro" | "defmulti" | "defmethod" => {
|
||||
make_kind_chunk(node, ChunkKind::Function, form_name(node, source), source, None)
|
||||
},
|
||||
"def" | "defonce" => {
|
||||
make_kind_chunk(node, ChunkKind::Decl, form_name(node, source), source, None)
|
||||
},
|
||||
"defprotocol" => {
|
||||
make_container_chunk(node, ChunkKind::Proto, form_name(node, source), source, None)
|
||||
},
|
||||
"deftype" | "defrecord" | "extend-type" | "extend-protocol" => {
|
||||
make_container_chunk(node, ChunkKind::Type, form_name(node, source), source, None)
|
||||
},
|
||||
_ if at_root => positional_candidate(node, ChunkKind::Form, source),
|
||||
_ => group_candidate(node, ChunkKind::Block, source),
|
||||
}
|
||||
}
|
||||
|
||||
impl LangClassifier for ClojureClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: super::classify::StructuralOverrides::EMPTY,
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
(node.kind() == "list_lit")
|
||||
.then(|| classify_form(node, source, matches!(context, ChunkContext::Root)))
|
||||
}
|
||||
}
|
||||
@@ -1,174 +0,0 @@
|
||||
//! CMake-specific chunk classifier.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct CMakeClassifier;
|
||||
|
||||
fn child_text<'a>(source: &'a str, node: Node<'_>) -> &'a str {
|
||||
node_text(source, node.start_byte(), node.end_byte())
|
||||
}
|
||||
|
||||
fn first_named_child(node: Node<'_>) -> Option<Node<'_>> {
|
||||
named_children(node).into_iter().next()
|
||||
}
|
||||
|
||||
fn first_named_child_of_kind<'t>(node: Node<'t>, kind: &str) -> Option<Node<'t>> {
|
||||
named_children(node)
|
||||
.into_iter()
|
||||
.find(|child| child.kind() == kind)
|
||||
}
|
||||
|
||||
fn command_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
first_named_child(node).and_then(|child| sanitize_identifier(child_text(source, child)))
|
||||
}
|
||||
|
||||
fn argument_nodes(node: Node<'_>) -> Vec<Node<'_>> {
|
||||
first_named_child_of_kind(node, "argument_list")
|
||||
.map(named_children)
|
||||
.unwrap_or_default()
|
||||
.into_iter()
|
||||
.filter(|child| child.kind() == "argument")
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn nth_argument_name(node: Node<'_>, index: usize, source: &str) -> Option<String> {
|
||||
argument_nodes(node)
|
||||
.into_iter()
|
||||
.nth(index)
|
||||
.and_then(|arg| sanitize_identifier(child_text(source, arg)))
|
||||
}
|
||||
|
||||
fn classify_definition<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"function_def" => {
|
||||
let header = first_named_child_of_kind(node, "function_command")?;
|
||||
let name = nth_argument_name(header, 0, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["body"]),
|
||||
))
|
||||
},
|
||||
"macro_def" => {
|
||||
let header = first_named_child_of_kind(node, "macro_command")?;
|
||||
let name = nth_argument_name(header, 0, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Macro,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["body"]),
|
||||
))
|
||||
},
|
||||
"if_condition" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::If,
|
||||
None,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
)),
|
||||
"foreach_loop" | "while_loop" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Loop,
|
||||
None,
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["body"]),
|
||||
)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_command<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
if node.kind() != "normal_command" {
|
||||
return None;
|
||||
}
|
||||
|
||||
let command = command_name(node, source)?;
|
||||
Some(match command.as_str() {
|
||||
"cmake_minimum_required" => make_kind_chunk(node, ChunkKind::VersionGate, None, source, None),
|
||||
"project" => {
|
||||
let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
make_kind_chunk(node, ChunkKind::Project, Some(name), source, None)
|
||||
},
|
||||
"include" | "find_package" => group_candidate(node, ChunkKind::Imports, source),
|
||||
"option" => {
|
||||
let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
make_kind_chunk(node, ChunkKind::Option, Some(name), source, None)
|
||||
},
|
||||
"set" => {
|
||||
let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)
|
||||
},
|
||||
"add_library" | "add_executable" | "add_custom_target" => {
|
||||
let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
make_kind_chunk(node, ChunkKind::Target, Some(name), source, None)
|
||||
},
|
||||
"install" | "export" => group_candidate(node, ChunkKind::Install, source),
|
||||
other => make_kind_chunk(node, ChunkKind::Cmd, Some(other.to_string()), source, None),
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_if_child<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"if_command" => Some(group_candidate(node, ChunkKind::Cond, source)),
|
||||
"elseif_command" => Some(positional_candidate(node, ChunkKind::Elif, source)),
|
||||
"else_command" => Some(positional_candidate(node, ChunkKind::Else, source)),
|
||||
"body" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Block,
|
||||
None,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
impl LangClassifier for CMakeClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[
|
||||
"endif_command",
|
||||
"endforeach_command",
|
||||
"endwhile_command",
|
||||
"endfunction_command",
|
||||
"endmacro_command",
|
||||
],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &[],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => {
|
||||
classify_definition(node, source).or_else(|| classify_command(node, source))
|
||||
},
|
||||
ChunkContext::FunctionBody => classify_definition(node, source)
|
||||
.or_else(|| classify_if_child(node, source))
|
||||
.or_else(|| classify_command(node, source)),
|
||||
ChunkContext::ClassBody => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,372 +0,0 @@
|
||||
//! Language-specific chunk classifiers for C# and Java.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
defaults::classify_var_decl,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct CSharpJavaClassifier;
|
||||
|
||||
const CSHARP_JAVA_ROOT_RULES: &[super::classify::SemanticRule] = &[
|
||||
// ── Imports ──
|
||||
semantic_rule(
|
||||
"import_declaration",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"using_directive",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"package_declaration",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"namespace_statement",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
// ── Functions ──
|
||||
semantic_rule(
|
||||
"method_declaration",
|
||||
ChunkKind::Method,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"function_declaration",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"function_definition",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
// ── Constructors ──
|
||||
semantic_rule(
|
||||
"constructor_declaration",
|
||||
ChunkKind::Constructor,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
// ── Containers ──
|
||||
semantic_rule(
|
||||
"class_declaration",
|
||||
ChunkKind::Class,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"interface_declaration",
|
||||
ChunkKind::Iface,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"enum_declaration",
|
||||
ChunkKind::Enum,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"struct_declaration",
|
||||
ChunkKind::Struct,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"record_declaration",
|
||||
ChunkKind::Struct,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"namespace_declaration",
|
||||
ChunkKind::Module,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"file_scoped_namespace_declaration",
|
||||
ChunkKind::Module,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
// ── Types ──
|
||||
semantic_rule(
|
||||
"type_alias_declaration",
|
||||
ChunkKind::Type,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
// ── Declarations ──
|
||||
semantic_rule(
|
||||
"property_declaration",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"state_variable_declaration",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
// ── Statements ──
|
||||
semantic_rule(
|
||||
"expression_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const CSHARP_JAVA_CLASS_RULES: &[super::classify::SemanticRule] = &[
|
||||
// ── Containers ──
|
||||
semantic_rule(
|
||||
"class_declaration",
|
||||
ChunkKind::Class,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"interface_declaration",
|
||||
ChunkKind::Iface,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"enum_declaration",
|
||||
ChunkKind::Enum,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"struct_declaration",
|
||||
ChunkKind::Struct,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"record_declaration",
|
||||
ChunkKind::Struct,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"namespace_declaration",
|
||||
ChunkKind::Module,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"file_scoped_namespace_declaration",
|
||||
ChunkKind::Module,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
// ── Static blocks ──
|
||||
semantic_rule(
|
||||
"class_static_block",
|
||||
ChunkKind::StaticInit,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const CSHARP_JAVA_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: CSHARP_JAVA_ROOT_RULES,
|
||||
class: CSHARP_JAVA_CLASS_RULES,
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
impl LangClassifier for CSharpJavaClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&CSHARP_JAVA_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => match node.kind() {
|
||||
// ── Variables / assignments ──
|
||||
"variable_declaration" | "lexical_declaration" => Some(classify_var_decl(node, source)),
|
||||
// ── Control flow (top-level scripts) ──
|
||||
"if_statement" | "switch_statement" | "switch_expression" | "for_statement"
|
||||
| "foreach_statement" | "while_statement" | "do_statement" | "try_statement" => {
|
||||
Some(classify_function_csharp_java(node, source))
|
||||
},
|
||||
_ => None,
|
||||
},
|
||||
ChunkContext::ClassBody => match node.kind() {
|
||||
// ── Methods (conditional constructor detection) ──
|
||||
"method_declaration" | "function_declaration" | "function_definition" => {
|
||||
let name =
|
||||
extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
if name == "constructor" {
|
||||
Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Constructor,
|
||||
None,
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
))
|
||||
} else {
|
||||
Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
))
|
||||
}
|
||||
},
|
||||
// ── Constructors ──
|
||||
"constructor_declaration" | "secondary_constructor" => Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Constructor,
|
||||
None,
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
)),
|
||||
// ── Fields ──
|
||||
"field_declaration"
|
||||
| "property_declaration"
|
||||
| "constant_declaration"
|
||||
| "event_field_declaration" => Some(match extract_field_name(node, source) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Fields, source),
|
||||
}),
|
||||
// ── Enum members ──
|
||||
"enum_member_declaration" | "enum_constant" | "enum_entry" => {
|
||||
Some(match extract_identifier(node, source) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Variants, source),
|
||||
})
|
||||
},
|
||||
_ => None,
|
||||
},
|
||||
ChunkContext::FunctionBody => Some(classify_function_csharp_java(node, source)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract the variable name from a field/constant declaration.
|
||||
///
|
||||
/// Java `field_declaration` has the structure:
|
||||
/// `field_declaration` { modifiers, type: `type_identifier`, declarator:
|
||||
/// `variable_declarator` { name: identifier } }
|
||||
///
|
||||
/// `extract_identifier` would find `type_identifier` first, so we look into
|
||||
/// `variable_declarator` children for the actual variable name.
|
||||
fn extract_field_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
for child in named_children(node) {
|
||||
if child.kind() == "variable_declarator" {
|
||||
return extract_identifier(child, source);
|
||||
}
|
||||
}
|
||||
extract_identifier(node, source)
|
||||
}
|
||||
|
||||
fn classify_function_csharp_java<'tree>(
|
||||
node: Node<'tree>,
|
||||
source: &str,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
|
||||
match node.kind() {
|
||||
"if_statement" => {
|
||||
make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"switch_statement" | "switch_expression" => {
|
||||
make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"try_statement" | "catch_clause" | "finally_clause" => {
|
||||
make_candidate(node, ChunkKind::Try, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"for_statement" => {
|
||||
make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"foreach_statement" => {
|
||||
make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"while_statement" => {
|
||||
make_candidate(node, ChunkKind::While, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"do_statement" => {
|
||||
make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"variable_declaration" | "lexical_declaration" => {
|
||||
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
|
||||
if span > 1 {
|
||||
if let Some(name) = extract_single_declarator_name(node, source) {
|
||||
make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)
|
||||
} else {
|
||||
group_from_sanitized(node, source)
|
||||
}
|
||||
} else {
|
||||
group_from_sanitized(node, source)
|
||||
}
|
||||
},
|
||||
_ => group_from_sanitized(node, source),
|
||||
}
|
||||
}
|
||||
|
||||
fn group_from_sanitized<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let sanitized = sanitize_node_kind(node.kind());
|
||||
let kind = ChunkKind::from_sanitized_kind(sanitized);
|
||||
let identifier = if kind == ChunkKind::Chunk {
|
||||
Some(sanitized.to_string())
|
||||
} else {
|
||||
None
|
||||
};
|
||||
make_candidate(node, kind, identifier, NameStyle::Group, None, None, source)
|
||||
}
|
||||
@@ -1,124 +0,0 @@
|
||||
//! Language-specific chunk classifiers for CSS and SCSS.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct CssClassifier;
|
||||
|
||||
const CSS_SHARED_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"keyframe_block",
|
||||
ChunkKind::Frame,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"declaration",
|
||||
ChunkKind::Fields,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const CSS_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: CSS_SHARED_RULES,
|
||||
class: CSS_SHARED_RULES,
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &["stylesheet"],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
|
||||
/// Extract a CSS selector name from a `rule_set` or `at_rule` node.
|
||||
///
|
||||
/// Tries known child kinds first (`selectors`, `selector_query`, `identifier`),
|
||||
/// then falls back to parsing the normalised header text.
|
||||
fn extract_css_selector(node: Node<'_>, source: &str) -> Option<String> {
|
||||
if let Some(sel) = child_by_kind(node, &["selectors", "selector_query", "identifier"]) {
|
||||
return sanitize_identifier(node_text(source, sel.start_byte(), sel.end_byte()));
|
||||
}
|
||||
let header = normalized_header(source, node.start_byte(), node.end_byte());
|
||||
let selector = header
|
||||
.trim_start_matches('@')
|
||||
.split('{')
|
||||
.next()
|
||||
.unwrap_or(header.as_str())
|
||||
.trim();
|
||||
sanitize_identifier(selector)
|
||||
}
|
||||
|
||||
/// Classify a CSS `rule_set` as a named container.
|
||||
fn classify_rule_set<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let name = extract_css_selector(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Rule,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["block"]),
|
||||
)
|
||||
}
|
||||
|
||||
/// Classify a CSS at-rule (`@media`, `@keyframes`, `@supports`, etc.) as a
|
||||
/// named container.
|
||||
fn classify_at_rule<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let name = extract_css_selector(node, source).unwrap_or_else(|| "rule".to_string());
|
||||
make_container_chunk(
|
||||
node,
|
||||
ChunkKind::At,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["block", "keyframe_block_list"]),
|
||||
)
|
||||
}
|
||||
|
||||
/// Shared dispatch for CSS node kinds, used in both root and class-body
|
||||
/// contexts.
|
||||
fn classify_css_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"rule_set" => Some(classify_rule_set(node, source)),
|
||||
"at_rule" | "media_statement" | "keyframes_statement" | "supports_statement" => {
|
||||
Some(classify_at_rule(node, source))
|
||||
},
|
||||
"keyframe_block" => Some(named_candidate(
|
||||
node,
|
||||
ChunkKind::Frame,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
)),
|
||||
"declaration" => Some(group_candidate(node, ChunkKind::Fields, source)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
impl LangClassifier for CssClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&CSS_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
if matches!(context, ChunkContext::Root | ChunkContext::ClassBody) {
|
||||
return classify_css_node(node, source);
|
||||
}
|
||||
None
|
||||
}
|
||||
}
|
||||
@@ -1,278 +0,0 @@
|
||||
//! Chunk classifiers for data formats: JSON, TOML, YAML.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct DataFormatsClassifier;
|
||||
|
||||
const DATA_FORMAT_STRUCTURAL_OVERRIDES: StructuralOverrides = StructuralOverrides {
|
||||
extra_trivia: &["bare_key", "quoted_key", "dotted_key"],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &[
|
||||
"array",
|
||||
"block_mapping",
|
||||
"block_node",
|
||||
"block_sequence",
|
||||
"document",
|
||||
"flow_mapping",
|
||||
"flow_node",
|
||||
"flow_sequence",
|
||||
"object",
|
||||
"stream",
|
||||
],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
};
|
||||
|
||||
const DATA_FORMAT_ROOT_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"inline_table",
|
||||
ChunkKind::Table,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"object",
|
||||
ChunkKind::Object,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"array",
|
||||
ChunkKind::Array,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"block_mapping",
|
||||
ChunkKind::Map,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"flow_mapping",
|
||||
ChunkKind::Map,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"block_sequence",
|
||||
ChunkKind::List,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"flow_sequence",
|
||||
ChunkKind::List,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"attribute",
|
||||
ChunkKind::Attr,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::ValueContainer,
|
||||
),
|
||||
];
|
||||
|
||||
const DATA_FORMAT_CLASS_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"inline_table",
|
||||
ChunkKind::Table,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"object",
|
||||
ChunkKind::Object,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"array",
|
||||
ChunkKind::Array,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"block_mapping",
|
||||
ChunkKind::Map,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"flow_mapping",
|
||||
ChunkKind::Map,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"block_sequence",
|
||||
ChunkKind::List,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"flow_sequence",
|
||||
ChunkKind::List,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::SelfNode(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"block_sequence_item",
|
||||
ChunkKind::Item,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"attribute",
|
||||
ChunkKind::Attr,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::ValueContainer,
|
||||
),
|
||||
];
|
||||
|
||||
const DATA_FORMAT_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: DATA_FORMAT_ROOT_RULES,
|
||||
class: DATA_FORMAT_CLASS_RULES,
|
||||
function: &[],
|
||||
structural_overrides: DATA_FORMAT_STRUCTURAL_OVERRIDES,
|
||||
};
|
||||
|
||||
impl LangClassifier for DataFormatsClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&DATA_FORMAT_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_data_node(node, source, true),
|
||||
ChunkContext::ClassBody => classify_data_node(node, source, false),
|
||||
ChunkContext::FunctionBody => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn preserve_children(
|
||||
&self,
|
||||
parent: &RawChunkCandidate<'_>,
|
||||
_children: &[RawChunkCandidate<'_>],
|
||||
) -> bool {
|
||||
// YAML keys with container values should always expose sub-chunks
|
||||
// so that deeply nested keys are individually addressable.
|
||||
parent.force_recurse && parent.kind == ChunkKind::Key
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_data_node<'t>(
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
is_root: bool,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// Key-value pairs (JSON pairs, YAML mappings)
|
||||
"pair" => {
|
||||
let name = extract_pair_key(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Key,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_value_container(node),
|
||||
))
|
||||
},
|
||||
"block_mapping_pair" | "flow_pair" => {
|
||||
let name = extract_yaml_key(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
let recurse = recurse_value_container(node);
|
||||
let mut candidate = make_kind_chunk(node, ChunkKind::Key, Some(name), source, recurse);
|
||||
// YAML structure is inherently hierarchical. Keys whose value is a
|
||||
// container (mapping/sequence) should always produce sub-chunks so
|
||||
// that deeply nested keys are individually addressable.
|
||||
if candidate.recurse.is_some() {
|
||||
candidate.force_recurse = true;
|
||||
}
|
||||
Some(candidate)
|
||||
},
|
||||
// TOML tables
|
||||
"table" => {
|
||||
let name = extract_toml_table_name(node, source);
|
||||
Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Table,
|
||||
name,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
))
|
||||
},
|
||||
// TOML array tables
|
||||
"table_array_element" => Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::Table,
|
||||
extract_toml_table_name(node, source).unwrap_or_else(|| "table_array".to_string()),
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
source,
|
||||
)),
|
||||
// YAML sequence items (only when nested, not at root level)
|
||||
"block_sequence_item" if !is_root => {
|
||||
Some(positional_candidate(node, ChunkKind::Item, source))
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract key from a `pair` node (JSON or TOML).
|
||||
/// JSON pairs have a `"key"` field; TOML pairs have no field names, so we fall
|
||||
/// back to looking for the first `bare_key`, `quoted_key`, or `dotted_key`
|
||||
/// child.
|
||||
fn extract_pair_key(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let key = node
|
||||
.child_by_field_name("key")
|
||||
.or_else(|| child_by_kind(node, &["bare_key", "quoted_key", "dotted_key"]))?;
|
||||
sanitize_identifier(unquote_text(node_text(source, key.start_byte(), key.end_byte())).as_str())
|
||||
}
|
||||
|
||||
fn extract_toml_table_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let key = child_by_kind(node, &["dotted_key", "bare_key", "quoted_key"])?;
|
||||
sanitize_identifier(node_text(source, key.start_byte(), key.end_byte()))
|
||||
}
|
||||
|
||||
/// Extract key from a YAML `block_mapping_pair` or `flow_pair` node.
|
||||
/// Descends into the key to find the first scalar child for complex keys.
|
||||
fn extract_yaml_key(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let key = node.child_by_field_name("key")?;
|
||||
let key_node = first_scalar_child(key).unwrap_or(key);
|
||||
sanitize_identifier(
|
||||
unquote_text(node_text(source, key_node.start_byte(), key_node.end_byte())).as_str(),
|
||||
)
|
||||
}
|
||||
@@ -1,184 +0,0 @@
|
||||
//! Chunk classifier for Dockerfile syntax.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct DockerfileClassifier;
|
||||
|
||||
const DOCKERFILE_ROOT_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"run_instruction",
|
||||
ChunkKind::Cmd,
|
||||
RuleStyle::Named,
|
||||
NamingMode::SanitizedKind,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"cmd_instruction",
|
||||
ChunkKind::Cmd,
|
||||
RuleStyle::Named,
|
||||
NamingMode::SanitizedKind,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"entrypoint_instruction",
|
||||
ChunkKind::Cmd,
|
||||
RuleStyle::Named,
|
||||
NamingMode::SanitizedKind,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"copy_instruction",
|
||||
ChunkKind::Copy,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"add_instruction",
|
||||
ChunkKind::Add,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"workdir_instruction",
|
||||
ChunkKind::Workdir,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"expose_instruction",
|
||||
ChunkKind::Expose,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"user_instruction",
|
||||
ChunkKind::User,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const DOCKERFILE_FUNCTION_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"cmd_instruction",
|
||||
ChunkKind::Cmd,
|
||||
RuleStyle::Named,
|
||||
NamingMode::SanitizedKind,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"shell_command",
|
||||
ChunkKind::Shell,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"json_string_array",
|
||||
ChunkKind::Argv,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const DOCKERFILE_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: DOCKERFILE_ROOT_RULES,
|
||||
class: &[],
|
||||
function: DOCKERFILE_FUNCTION_RULES,
|
||||
structural_overrides: super::classify::StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
fn child_text<'a>(source: &'a str, node: Node<'_>) -> &'a str {
|
||||
node_text(source, node.start_byte(), node.end_byte())
|
||||
}
|
||||
|
||||
fn first_named_child(node: Node<'_>) -> Option<Node<'_>> {
|
||||
named_children(node).into_iter().next()
|
||||
}
|
||||
|
||||
fn first_named_child_of_kind<'t>(node: Node<'t>, kind: &str) -> Option<Node<'t>> {
|
||||
named_children(node)
|
||||
.into_iter()
|
||||
.find(|child| child.kind() == kind)
|
||||
}
|
||||
|
||||
fn extract_stage_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
if let Some(alias) = child_by_kind(node, &["image_alias"]) {
|
||||
return sanitize_identifier(child_text(source, alias));
|
||||
}
|
||||
|
||||
child_by_kind(node, &["image_spec"]).and_then(|image| {
|
||||
let image_name = child_by_kind(image, &["image_name"]).unwrap_or(image);
|
||||
sanitize_identifier(child_text(source, image_name))
|
||||
})
|
||||
}
|
||||
|
||||
fn extract_pair_key(node: Node<'_>, pair_kind: &str, source: &str) -> Option<String> {
|
||||
first_named_child_of_kind(node, pair_kind)
|
||||
.and_then(first_named_child)
|
||||
.and_then(|key| sanitize_identifier(unquote_text(child_text(source, key)).as_str()))
|
||||
}
|
||||
|
||||
fn extract_arg_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
first_named_child(node).and_then(|name| sanitize_identifier(child_text(source, name)))
|
||||
}
|
||||
|
||||
impl LangClassifier for DockerfileClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&DOCKERFILE_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => match node.kind() {
|
||||
"from_instruction" => {
|
||||
let name =
|
||||
extract_stage_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_kind_chunk(node, ChunkKind::Stage, Some(name), source, None))
|
||||
},
|
||||
"arg_instruction" => {
|
||||
let name = extract_arg_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_kind_chunk(node, ChunkKind::Arg, Some(name), source, None))
|
||||
},
|
||||
"env_instruction" => {
|
||||
let name = extract_pair_key(node, "env_pair", source)
|
||||
.unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_kind_chunk(node, ChunkKind::Env, Some(name), source, None))
|
||||
},
|
||||
"label_instruction" => {
|
||||
let name = extract_pair_key(node, "label_pair", source)
|
||||
.unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_kind_chunk(node, ChunkKind::Label, Some(name), source, None))
|
||||
},
|
||||
"healthcheck_instruction" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Healthcheck,
|
||||
None,
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["cmd_instruction"]),
|
||||
)),
|
||||
_ => None,
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,159 +0,0 @@
|
||||
//! Language-specific chunk classifier for Elixir.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct ElixirClassifier;
|
||||
|
||||
/// Extract the call target: `target` field, or first named child.
|
||||
fn call_target(node: Node<'_>, source: &str) -> Option<String> {
|
||||
node
|
||||
.child_by_field_name("target")
|
||||
.or_else(|| named_children(node).into_iter().next())
|
||||
.map(|n| node_text(source, n.start_byte(), n.end_byte()).to_string())
|
||||
}
|
||||
|
||||
/// Classify an Elixir `call` node based on its target keyword.
|
||||
fn classify_call<'t>(node: Node<'t>, source: &str, at_root: bool) -> RawChunkCandidate<'t> {
|
||||
let target = call_target(node, source).unwrap_or_default();
|
||||
let name = || call_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
match target.as_str() {
|
||||
"defmodule" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Module,
|
||||
Some(name()),
|
||||
source,
|
||||
recurse_body(node, ChunkContext::ClassBody),
|
||||
),
|
||||
"defprotocol" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Proto,
|
||||
Some(name()),
|
||||
source,
|
||||
recurse_body(node, ChunkContext::ClassBody),
|
||||
),
|
||||
"defimpl" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Impl,
|
||||
Some(name()),
|
||||
source,
|
||||
recurse_body(node, ChunkContext::ClassBody),
|
||||
),
|
||||
"def" | "defp" | "defdelegate" | "defguard" | "defguardp" | "defn" | "defnp" => {
|
||||
make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
Some(name()),
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
)
|
||||
},
|
||||
"defmacro" | "defmacrop" => make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Macro,
|
||||
Some(name()),
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
),
|
||||
"alias" | "import" | "require" | "use" => group_candidate(node, ChunkKind::Imports, source),
|
||||
"defstruct" | "defexception" => group_candidate(node, ChunkKind::Declarations, source),
|
||||
"if" | "unless" => positional_candidate(node, ChunkKind::If, source),
|
||||
"case" | "cond" | "receive" => positional_candidate(node, ChunkKind::Switch, source),
|
||||
"for" => positional_candidate(node, ChunkKind::For, source),
|
||||
"try" | "with" => positional_candidate(node, ChunkKind::Block, source),
|
||||
_ if at_root => group_candidate(node, ChunkKind::Statements, source),
|
||||
_ => group_candidate(node, ChunkKind::Block, source),
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract the name from an Elixir `call` node.
|
||||
///
|
||||
/// Skips keyword-only calls (imports, control flow) that have no meaningful
|
||||
/// identifier, then returns the first non-`do_block` named child after the
|
||||
/// target.
|
||||
fn call_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let target = call_target(node, source)?;
|
||||
if matches!(
|
||||
target.as_str(),
|
||||
"alias"
|
||||
| "import"
|
||||
| "require"
|
||||
| "use"
|
||||
| "if" | "case"
|
||||
| "cond"
|
||||
| "for"
|
||||
| "try"
|
||||
| "with"
|
||||
| "unless"
|
||||
| "receive"
|
||||
) {
|
||||
return None;
|
||||
}
|
||||
|
||||
// The first named child after the target is typically `arguments`.
|
||||
// For `def run(x)`, arguments contains a `call` node whose target is `run`.
|
||||
// For `defmodule App`, arguments contains an `alias` node with text `App`.
|
||||
// For `def run(x) when is_integer(x)`, arguments contains a `binary_operator`
|
||||
// with the call on the left and the guard on the right.
|
||||
// Extract the meaningful name, not the full text with parameters.
|
||||
named_children(node).into_iter().skip(1).find_map(|child| {
|
||||
if child.kind() == "do_block" {
|
||||
return None;
|
||||
}
|
||||
if child.kind() == "arguments" {
|
||||
// Dig into arguments to find the actual name.
|
||||
return named_children(child).into_iter().next().and_then(|arg| {
|
||||
if arg.kind() == "call" {
|
||||
// `def run(x)` → arguments has call(target=run), extract target name
|
||||
call_target(arg, source).and_then(|t| sanitize_identifier(&t))
|
||||
} else if arg.kind() == "binary_operator" {
|
||||
// `def run(x) when guard` → binary_operator(left=call, right=guard)
|
||||
// Extract name from the left side (the actual function call).
|
||||
arg.child_by_field_name("left").and_then(|left| {
|
||||
if left.kind() == "call" {
|
||||
call_target(left, source).and_then(|t| sanitize_identifier(&t))
|
||||
} else {
|
||||
sanitize_identifier(node_text(source, left.start_byte(), left.end_byte()))
|
||||
}
|
||||
})
|
||||
} else {
|
||||
// `defmodule App` → arguments has alias("App")
|
||||
sanitize_identifier(node_text(source, arg.start_byte(), arg.end_byte()))
|
||||
}
|
||||
});
|
||||
}
|
||||
sanitize_identifier(node_text(source, child.start_byte(), child.end_byte()))
|
||||
})
|
||||
}
|
||||
|
||||
impl LangClassifier for ElixirClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &["unary_operator"],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &[],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
(node.kind() == "call").then(|| classify_call(node, source, context == ChunkContext::Root))
|
||||
}
|
||||
}
|
||||
@@ -1,242 +0,0 @@
|
||||
//! Language-specific chunk classifier for Erlang.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct ErlangClassifier;
|
||||
|
||||
fn find_named_descendant_by_kind<'t>(node: Node<'t>, kinds: &[&str]) -> Option<Node<'t>> {
|
||||
if kinds.iter().any(|kind| node.kind() == *kind) {
|
||||
return Some(node);
|
||||
}
|
||||
|
||||
for child in named_children(node) {
|
||||
if let Some(found) = find_named_descendant_by_kind(child, kinds) {
|
||||
return Some(found);
|
||||
}
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
|
||||
fn named_text(node: Node<'_>, source: &str) -> Option<String> {
|
||||
sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()))
|
||||
}
|
||||
|
||||
fn erlang_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let name_node = match node.kind() {
|
||||
"module_attribute" | "record_decl" | "record_field" => {
|
||||
child_by_field_or_kind(node, &["name"], &["atom"])
|
||||
},
|
||||
"type_alias" => node
|
||||
.child_by_field_name("name")
|
||||
.and_then(|name| find_named_descendant_by_kind(name, &["atom"])),
|
||||
"spec" => child_by_field_or_kind(node, &["fun"], &["atom"]),
|
||||
"pp_define" => node
|
||||
.child_by_field_name("lhs")
|
||||
.and_then(|lhs| find_named_descendant_by_kind(lhs, &["var"])),
|
||||
"fun_decl" => node
|
||||
.child_by_field_name("clause")
|
||||
.and_then(|clause| child_by_field_or_kind(clause, &["name"], &["atom"])),
|
||||
"function_clause" => child_by_field_or_kind(node, &["name"], &["atom"]),
|
||||
_ => child_by_kind(node, &["atom", "var"]),
|
||||
}?;
|
||||
|
||||
named_text(name_node, source)
|
||||
}
|
||||
|
||||
fn recurse_clause_body(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
recurse_into(node, ChunkContext::FunctionBody, &["body"], &["clause_body"])
|
||||
}
|
||||
|
||||
impl LangClassifier for ErlangClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[
|
||||
semantic_rule(
|
||||
"export_attribute",
|
||||
ChunkKind::Exports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"export_type_attribute",
|
||||
ChunkKind::Exports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"import_attribute",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"pp_include",
|
||||
ChunkKind::Includes,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"pp_include_lib",
|
||||
ChunkKind::Includes,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &[],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &["spec"],
|
||||
},
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_erlang_root(node, source),
|
||||
ChunkContext::ClassBody => classify_erlang_class(node, source),
|
||||
ChunkContext::FunctionBody => classify_erlang_function(node, source),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_erlang_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"module_attribute" => {
|
||||
make_kind_chunk(node, ChunkKind::Module, erlang_name(node, source), source, None)
|
||||
},
|
||||
"pp_define" => {
|
||||
make_kind_chunk(node, ChunkKind::Macro, erlang_name(node, source), source, None)
|
||||
},
|
||||
"record_decl" => make_candidate(
|
||||
node,
|
||||
ChunkKind::Struct,
|
||||
format!("record_{}", erlang_name(node, source)?),
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
source,
|
||||
),
|
||||
"type_alias" => {
|
||||
make_kind_chunk(node, ChunkKind::Type, erlang_name(node, source), source, None)
|
||||
},
|
||||
// The Erlang grammar exposes each top-level clause as its own `fun_decl`.
|
||||
// Keep that shape instead of inventing a synthetic merged function node.
|
||||
"fun_decl" => make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
erlang_name(node, source),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
),
|
||||
"spec" => return None,
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_erlang_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"record_field" => {
|
||||
make_kind_chunk(node, ChunkKind::Field, erlang_name(node, source), source, None)
|
||||
},
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_erlang_function<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"function_clause" => make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Clause,
|
||||
erlang_name(node, source),
|
||||
source,
|
||||
recurse_clause_body(node),
|
||||
),
|
||||
"fun_clause" | "cr_clause" => make_candidate(
|
||||
node,
|
||||
ChunkKind::Clause,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse_clause_body(node),
|
||||
source,
|
||||
),
|
||||
"receive_after" => make_candidate(
|
||||
node,
|
||||
ChunkKind::After,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse_clause_body(node),
|
||||
source,
|
||||
),
|
||||
"catch_clause" => make_candidate(
|
||||
node,
|
||||
ChunkKind::Catch,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse_clause_body(node),
|
||||
source,
|
||||
),
|
||||
"receive_expr" => make_candidate(
|
||||
node,
|
||||
ChunkKind::Receive,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
source,
|
||||
),
|
||||
"case_expr" => make_candidate(
|
||||
node,
|
||||
ChunkKind::Case,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
source,
|
||||
),
|
||||
"try_expr" => make_candidate(
|
||||
node,
|
||||
ChunkKind::Try,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
source,
|
||||
),
|
||||
"anonymous_fun" => make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
Some("anonymous".to_string()),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
),
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
@@ -1,320 +0,0 @@
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct GoClassifier;
|
||||
|
||||
const ROOT_RULES: &[super::classify::SemanticRule] = &[
|
||||
// ── Imports / package ──
|
||||
semantic_rule(
|
||||
"package_clause",
|
||||
ChunkKind::Module,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"import_declaration",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
// ── Functions ──
|
||||
semantic_rule(
|
||||
"function_declaration",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"method_declaration",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
// ── Statements ──
|
||||
semantic_rule(
|
||||
"expression_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"go_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"defer_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"send_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const CLASS_RULES: &[super::classify::SemanticRule] = &[
|
||||
// ── Methods ──
|
||||
semantic_rule(
|
||||
"method_spec",
|
||||
ChunkKind::Method,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::None,
|
||||
),
|
||||
// ── Field / method lists ──
|
||||
semantic_rule(
|
||||
"field_declaration_list",
|
||||
ChunkKind::Fields,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"method_spec_list",
|
||||
ChunkKind::Methods,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const FUNCTION_RULES: &[super::classify::SemanticRule] = &[
|
||||
// ── Control flow ──
|
||||
semantic_rule(
|
||||
"if_statement",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"for_statement",
|
||||
ChunkKind::For,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"switch_statement",
|
||||
ChunkKind::Switch,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"expression_switch_statement",
|
||||
ChunkKind::Switch,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"type_switch_statement",
|
||||
ChunkKind::Switch,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"select_statement",
|
||||
ChunkKind::Switch,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
// ── Statements ──
|
||||
semantic_rule(
|
||||
"go_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"defer_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"send_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const GO_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: ROOT_RULES,
|
||||
class: CLASS_RULES,
|
||||
function: FUNCTION_RULES,
|
||||
structural_overrides: StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
impl LangClassifier for GoClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&GO_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_root_custom(node, source),
|
||||
ChunkContext::ClassBody => classify_class_custom(node, source),
|
||||
ChunkContext::FunctionBody => classify_function_custom(node, source),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// ── Variables ──
|
||||
"const_declaration" | "var_declaration" | "short_var_declaration" => {
|
||||
Some(match extract_identifier(node, source) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Declarations, source),
|
||||
})
|
||||
},
|
||||
|
||||
// ── Containers ──
|
||||
"type_declaration" => Some(classify_type_decl(node, source)),
|
||||
|
||||
// ── Control flow (top-level scripts) ──
|
||||
"if_statement"
|
||||
| "switch_statement"
|
||||
| "expression_switch_statement"
|
||||
| "type_switch_statement"
|
||||
| "select_statement"
|
||||
| "for_statement" => Some(classify_function_go(node, source)),
|
||||
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// ── Fields ──
|
||||
"field_declaration" | "embedded_field" => Some(match extract_identifier(node, source) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Fields, source),
|
||||
}),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// ── Variables ──
|
||||
"short_var_declaration" | "var_declaration" | "const_declaration" => {
|
||||
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
|
||||
Some(if span > 1 {
|
||||
if let Some(name) = extract_identifier(node, source) {
|
||||
make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)
|
||||
} else {
|
||||
group_from_sanitized(node, source)
|
||||
}
|
||||
} else {
|
||||
group_from_sanitized(node, source)
|
||||
})
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Classify Go function-level nodes (reused for top-level control flow
|
||||
/// delegation).
|
||||
fn classify_function_go<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
|
||||
match node.kind() {
|
||||
"if_statement" => {
|
||||
make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"switch_statement"
|
||||
| "expression_switch_statement"
|
||||
| "type_switch_statement"
|
||||
| "select_statement" => {
|
||||
make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"for_statement" => {
|
||||
make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
_ => group_candidate(node, ChunkKind::Statements, source),
|
||||
}
|
||||
}
|
||||
|
||||
/// Classify Go `type_declaration` nodes.
|
||||
///
|
||||
/// A single `type_spec` with a struct/interface body becomes a container;
|
||||
/// a single `type_spec` without one becomes a named leaf.
|
||||
/// Multiple `type_spec` children (type group) become a group.
|
||||
fn classify_type_decl<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let specs: Vec<Node<'t>> = named_children(node)
|
||||
.into_iter()
|
||||
.filter(|c| c.kind() == "type_spec")
|
||||
.collect();
|
||||
|
||||
if specs.len() == 1 {
|
||||
let spec = specs[0];
|
||||
let name = extract_identifier(spec, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
if let Some(recurse) = recurse_type_spec(spec) {
|
||||
return make_container_chunk_from(
|
||||
node,
|
||||
spec,
|
||||
ChunkKind::Type,
|
||||
Some(name),
|
||||
source,
|
||||
Some(recurse),
|
||||
);
|
||||
}
|
||||
return make_kind_chunk_from(node, spec, ChunkKind::Type, Some(name), source, None);
|
||||
}
|
||||
|
||||
group_candidate(node, ChunkKind::Declarations, source)
|
||||
}
|
||||
|
||||
fn group_from_sanitized<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let sanitized = sanitize_node_kind(node.kind());
|
||||
let kind = ChunkKind::from_sanitized_kind(sanitized);
|
||||
let identifier = if kind == ChunkKind::Chunk {
|
||||
Some(sanitized.to_string())
|
||||
} else {
|
||||
None
|
||||
};
|
||||
make_candidate(node, kind, identifier, NameStyle::Group, None, None, source)
|
||||
}
|
||||
|
||||
/// For a `type_spec`, find a `struct_type` or `interface_type` child and return
|
||||
/// its body (`field_declaration_list` or `method_spec_list`) as a recurse spec.
|
||||
fn recurse_type_spec(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
let container = child_by_kind(node, &["struct_type", "interface_type"])?;
|
||||
let body = child_by_kind(container, &["field_declaration_list", "method_spec_list"])
|
||||
.unwrap_or(container);
|
||||
Some(RecurseSpec { node: body, context: ChunkContext::ClassBody })
|
||||
}
|
||||
@@ -1,294 +0,0 @@
|
||||
//! GraphQL-specific chunk classifier.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct GraphqlClassifier;
|
||||
|
||||
impl LangClassifier for GraphqlClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &["comma"],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &[
|
||||
"document",
|
||||
"definition",
|
||||
"type_system_definition",
|
||||
"type_definition",
|
||||
"executable_definition",
|
||||
],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_graphql_root(node, source),
|
||||
ChunkContext::ClassBody => classify_graphql_class(node, source),
|
||||
ChunkContext::FunctionBody => classify_graphql_function(node, source),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_graphql_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"schema_definition" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Schema,
|
||||
None,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
)),
|
||||
"directive_definition" => Some(make_named_graphql_chunk(
|
||||
node,
|
||||
ChunkKind::Directive,
|
||||
extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string()),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["arguments_definition"]),
|
||||
)),
|
||||
"scalar_type_definition" => Some(make_named_graphql_chunk(
|
||||
node,
|
||||
ChunkKind::Type,
|
||||
format!(
|
||||
"scalar_{}",
|
||||
extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string())
|
||||
),
|
||||
source,
|
||||
None,
|
||||
)),
|
||||
"object_type_definition" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Type,
|
||||
extract_graphql_name(node, source),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["fields_definition"]),
|
||||
)),
|
||||
"interface_type_definition" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Interface,
|
||||
extract_graphql_name(node, source),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["fields_definition"]),
|
||||
)),
|
||||
"union_type_definition" => Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Union,
|
||||
extract_graphql_name(node, source),
|
||||
source,
|
||||
None,
|
||||
)),
|
||||
"enum_type_definition" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Enum,
|
||||
extract_graphql_name(node, source),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["enum_values_definition"]),
|
||||
)),
|
||||
"input_object_type_definition" => Some(make_named_graphql_chunk(
|
||||
node,
|
||||
ChunkKind::Type,
|
||||
format!(
|
||||
"input_{}",
|
||||
extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string())
|
||||
),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["input_fields_definition"]),
|
||||
)),
|
||||
"operation_definition" => Some(make_named_graphql_chunk(
|
||||
node,
|
||||
ChunkKind::Operation,
|
||||
extract_graphql_operation_chunk_name(node, source),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["selection_set"]),
|
||||
)),
|
||||
"fragment_definition" => Some(make_named_graphql_chunk(
|
||||
node,
|
||||
ChunkKind::Operation,
|
||||
format!(
|
||||
"fragment_{}",
|
||||
extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string())
|
||||
),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["selection_set"]),
|
||||
)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_graphql_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"root_operation_type_definition" => Some(make_named_graphql_chunk(
|
||||
node,
|
||||
ChunkKind::Root,
|
||||
extract_graphql_operation_type(node, source).unwrap_or_else(|| "anonymous".to_string()),
|
||||
source,
|
||||
None,
|
||||
)),
|
||||
"field_definition" => {
|
||||
let name = extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
let recurse = recurse_into(node, ChunkContext::ClassBody, &[], &["arguments_definition"]);
|
||||
Some(match recurse {
|
||||
Some(recurse) => {
|
||||
make_container_chunk(node, ChunkKind::Field, Some(name), source, Some(recurse))
|
||||
},
|
||||
None => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
|
||||
})
|
||||
},
|
||||
"input_value_definition" => Some(classify_graphql_input_value(node, source)),
|
||||
"enum_value_definition" => Some(make_named_graphql_chunk(
|
||||
node,
|
||||
ChunkKind::Variant,
|
||||
format!(
|
||||
"value_{}",
|
||||
extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string())
|
||||
),
|
||||
source,
|
||||
None,
|
||||
)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_graphql_function<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"selection" => classify_graphql_selection(node, source),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_graphql_selection<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
let child = first_named_child(node)?;
|
||||
match child.kind() {
|
||||
"field" => {
|
||||
let name = extract_graphql_name(child, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
let recurse = recurse_into(child, ChunkContext::FunctionBody, &[], &["selection_set"]);
|
||||
Some(match recurse {
|
||||
Some(recurse) => make_container_chunk_from(
|
||||
node,
|
||||
child,
|
||||
ChunkKind::Field,
|
||||
Some(name),
|
||||
source,
|
||||
Some(recurse),
|
||||
),
|
||||
None => make_kind_chunk_from(node, child, ChunkKind::Field, Some(name), source, None),
|
||||
})
|
||||
},
|
||||
"fragment_spread" => Some(make_named_graphql_chunk_from(
|
||||
node,
|
||||
child,
|
||||
ChunkKind::Operation,
|
||||
format!(
|
||||
"spread_{}",
|
||||
extract_graphql_name(child, source).unwrap_or_else(|| "anonymous".to_string())
|
||||
),
|
||||
source,
|
||||
None,
|
||||
)),
|
||||
"inline_fragment" => Some(make_container_chunk_from(
|
||||
node,
|
||||
child,
|
||||
ChunkKind::InlineFragment,
|
||||
None,
|
||||
source,
|
||||
recurse_into(child, ChunkContext::FunctionBody, &[], &["selection_set"]),
|
||||
)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_graphql_input_value<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let name = extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
match node.parent().map(|parent| parent.kind()) {
|
||||
Some("input_fields_definition") => {
|
||||
make_kind_chunk(node, ChunkKind::Field, Some(name), source, None)
|
||||
},
|
||||
_ => make_kind_chunk(node, ChunkKind::Arg, Some(name), source, None),
|
||||
}
|
||||
}
|
||||
|
||||
fn make_named_graphql_chunk<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
identifier: impl Into<Option<String>>,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'t>>,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
make_candidate(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse,
|
||||
source,
|
||||
)
|
||||
}
|
||||
|
||||
fn make_named_graphql_chunk_from<'t>(
|
||||
range_node: Node<'t>,
|
||||
signature_node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
identifier: impl Into<Option<String>>,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'t>>,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
make_candidate(
|
||||
range_node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(signature_node, source),
|
||||
recurse,
|
||||
source,
|
||||
)
|
||||
}
|
||||
|
||||
fn extract_graphql_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
find_graphql_name_node(node)
|
||||
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
|
||||
}
|
||||
|
||||
fn find_graphql_name_node(node: Node<'_>) -> Option<Node<'_>> {
|
||||
match node.kind() {
|
||||
"name" | "fragment_name" => Some(node),
|
||||
_ => named_children(node)
|
||||
.into_iter()
|
||||
.find_map(find_graphql_name_node),
|
||||
}
|
||||
}
|
||||
|
||||
fn extract_graphql_operation_type(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["operation_type"])
|
||||
.and_then(|kind| sanitize_identifier(node_text(source, kind.start_byte(), kind.end_byte())))
|
||||
}
|
||||
|
||||
fn extract_graphql_operation_chunk_name(node: Node<'_>, source: &str) -> String {
|
||||
let operation =
|
||||
extract_graphql_operation_type(node, source).unwrap_or_else(|| "operation".to_string());
|
||||
match extract_graphql_name(node, source) {
|
||||
Some(name) => format!("{operation}_{name}"),
|
||||
None => operation,
|
||||
}
|
||||
}
|
||||
|
||||
fn first_named_child(node: Node<'_>) -> Option<Node<'_>> {
|
||||
(0..node.named_child_count()).find_map(|index| node.named_child(index))
|
||||
}
|
||||
@@ -1,195 +0,0 @@
|
||||
//! Language-specific chunk classifiers for Haskell and Scala.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct HaskellScalaClassifier;
|
||||
|
||||
const HASKELL_SCALA_ROOT_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"import_declaration",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"package_declaration",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"module",
|
||||
ChunkKind::Module,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"function_declaration",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"function_definition",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"class_definition",
|
||||
ChunkKind::Class,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"object_definition",
|
||||
ChunkKind::Module,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"trait_definition",
|
||||
ChunkKind::Iface,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"type_alias_declaration",
|
||||
ChunkKind::Type,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"type_item",
|
||||
ChunkKind::Type,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"variable_declaration",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"assignment",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"expression_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const HASKELL_SCALA_FUNCTION_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"if_statement",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"match_expression",
|
||||
ChunkKind::Match,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"for_expression",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"while_expression",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"block_expression",
|
||||
ChunkKind::Block,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
];
|
||||
|
||||
const HASKELL_SCALA_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: HASKELL_SCALA_ROOT_RULES,
|
||||
class: &[],
|
||||
function: HASKELL_SCALA_FUNCTION_RULES,
|
||||
structural_overrides: super::classify::StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
impl LangClassifier for HaskellScalaClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&HASKELL_SCALA_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
if context != ChunkContext::ClassBody {
|
||||
return None;
|
||||
}
|
||||
|
||||
match node.kind() {
|
||||
"function_declaration" | "function_definition" | "method_definition" => {
|
||||
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
let kind = if name == "constructor" {
|
||||
ChunkKind::Constructor
|
||||
} else {
|
||||
ChunkKind::Function
|
||||
};
|
||||
let identifier = (kind != ChunkKind::Constructor).then_some(name);
|
||||
Some(make_kind_chunk(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
source,
|
||||
resolve_recurse(node, ChunkContext::FunctionBody),
|
||||
))
|
||||
},
|
||||
"variable_declaration" | "property_declaration" => {
|
||||
Some(extract_identifier(node, source).map_or_else(
|
||||
|| group_candidate(node, ChunkKind::Fields, source),
|
||||
|name| make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
|
||||
))
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,163 +0,0 @@
|
||||
//! Language-specific chunk classifiers for HTML and XML.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
use crate::language::SupportLang;
|
||||
|
||||
pub struct HtmlXmlClassifier;
|
||||
|
||||
const HTML_XML_SHARED_RULES: &[super::classify::SemanticRule] = &[semantic_rule(
|
||||
"text_node",
|
||||
ChunkKind::Text,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
)];
|
||||
|
||||
const HTML_XML_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: HTML_XML_SHARED_RULES,
|
||||
class: HTML_XML_SHARED_RULES,
|
||||
function: &[],
|
||||
structural_overrides: super::classify::StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
/// Classify an element-like node as a container with tag semantics.
|
||||
///
|
||||
/// Uses `extract_markup_tag_name` directly because the shared
|
||||
/// `extract_identifier` does not handle HTML/XML start-tag structures.
|
||||
fn classify_element<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"script_element" => Some(classify_script_element(node, source)),
|
||||
"style_element" => Some(classify_style_element(node, source)),
|
||||
"element" => {
|
||||
let tag_name =
|
||||
extract_markup_tag_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
// HTML: child elements are direct children of `element`.
|
||||
// XML: child elements are inside a `content` wrapper node.
|
||||
let recurse_target = child_by_kind(node, &["content"]).unwrap_or(node);
|
||||
Some(force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Tag,
|
||||
Some(tag_name),
|
||||
source,
|
||||
Some(recurse_self(recurse_target, ChunkContext::ClassBody)),
|
||||
)))
|
||||
},
|
||||
"text_node" => Some(group_candidate(node, ChunkKind::Text, source)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
classify_injected_raw_text_block(node, ChunkKind::Script, source, SupportLang::JavaScript)
|
||||
}
|
||||
|
||||
fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
classify_injected_raw_text_block(node, ChunkKind::Style, source, SupportLang::Css)
|
||||
}
|
||||
|
||||
fn classify_injected_raw_text_block<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
source: &str,
|
||||
default_language: SupportLang,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
let Some(content_node) = child_by_kind(node, &["raw_text"]) else {
|
||||
return positional_candidate(node, kind, source);
|
||||
};
|
||||
|
||||
let candidate = with_region_node(positional_candidate(node, kind, source), Some(content_node));
|
||||
match resolve_embedded_language(node, source, default_language) {
|
||||
Some(language) => with_injected_subtree(candidate, language, content_node),
|
||||
None => candidate,
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract the tag name from an HTML/XML element node.
|
||||
///
|
||||
/// HTML: `element` → `start_tag`/`self_closing_tag` → `tag_name`
|
||||
/// XML (tree-sitter-xml): `element` → `STag`/`EmptyElemTag` → `Name`
|
||||
fn extract_markup_tag_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
named_children(node).into_iter().find_map(|child| {
|
||||
let tag_name_kinds: &[&str] = match child.kind() {
|
||||
// HTML
|
||||
"start_tag" | "self_closing_tag" => &["tag_name"],
|
||||
// XML (tree-sitter-xml grammar)
|
||||
"STag" | "EmptyElemTag" => &["Name"],
|
||||
_ => return None,
|
||||
};
|
||||
child_by_kind(child, tag_name_kinds)
|
||||
.and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte())))
|
||||
})
|
||||
}
|
||||
|
||||
fn resolve_embedded_language(
|
||||
node: Node<'_>,
|
||||
source: &str,
|
||||
default_language: SupportLang,
|
||||
) -> Option<SupportLang> {
|
||||
if let Some(language) = attribute_value(node, "lang", source) {
|
||||
return SupportLang::from_alias(language.as_str());
|
||||
}
|
||||
Some(default_language)
|
||||
}
|
||||
|
||||
fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option<String> {
|
||||
let start = start_like(node)?;
|
||||
for child in named_children(start) {
|
||||
if child.kind() != "attribute" {
|
||||
continue;
|
||||
}
|
||||
if extract_attribute_name(child, source).as_deref() != Some(name) {
|
||||
continue;
|
||||
}
|
||||
if let Some(value) = child_by_kind(child, &["attribute_value", "quoted_attribute_value"]) {
|
||||
return sanitize_identifier(&unquote_text(node_text(
|
||||
source,
|
||||
value.start_byte(),
|
||||
value.end_byte(),
|
||||
)));
|
||||
}
|
||||
return Some(name.to_string());
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn extract_attribute_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["attribute_name"])
|
||||
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
|
||||
}
|
||||
|
||||
fn start_like(node: Node<'_>) -> Option<Node<'_>> {
|
||||
child_by_kind(node, &["start_tag", "self_closing_tag"])
|
||||
}
|
||||
|
||||
const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> {
|
||||
candidate.force_recurse = true;
|
||||
candidate
|
||||
}
|
||||
|
||||
impl LangClassifier for HtmlXmlClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&HTML_XML_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
if matches!(context, ChunkContext::Root | ChunkContext::ClassBody) {
|
||||
return classify_element(node, source);
|
||||
}
|
||||
None
|
||||
}
|
||||
}
|
||||
@@ -1,88 +0,0 @@
|
||||
//! Chunk classifier for INI.
|
||||
//!
|
||||
//! The tree-sitter INI grammar is intentionally flat: a document contains
|
||||
//! root-level `setting` nodes and `section` containers, and a section contains
|
||||
//! only its own `setting` children. Mirror that structure directly instead of
|
||||
//! inventing deeper hierarchy.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct IniClassifier;
|
||||
|
||||
impl LangClassifier for IniClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: super::classify::StructuralOverrides::EMPTY,
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_ini_root(node, source),
|
||||
ChunkContext::ClassBody => classify_ini_class(node, source),
|
||||
ChunkContext::FunctionBody => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_ini_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"section" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Section,
|
||||
Some(ini_name(node, source)?),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
),
|
||||
// INI permits settings before any section header; keep them as first-class
|
||||
// chunks instead of forcing them under a synthetic container.
|
||||
"setting" => {
|
||||
make_kind_chunk(node, ChunkKind::Key, Some(ini_name(node, source)?), source, None)
|
||||
},
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_ini_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"setting" => {
|
||||
make_kind_chunk(node, ChunkKind::Key, Some(ini_name(node, source)?), source, None)
|
||||
},
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn ini_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
find_named_text(node, source, &["section_name", "setting_name", "text"]).and_then(|text| {
|
||||
sanitize_identifier(text.trim().trim_start_matches('[').trim_end_matches(']'))
|
||||
})
|
||||
}
|
||||
|
||||
fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> {
|
||||
if kinds.iter().any(|kind| node.kind() == *kind) {
|
||||
return Some(node_text(source, node.start_byte(), node.end_byte()));
|
||||
}
|
||||
|
||||
for child in named_children(node) {
|
||||
if let Some(text) = find_named_text(child, source, kinds) {
|
||||
return Some(text);
|
||||
}
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
@@ -1,918 +0,0 @@
|
||||
//! Jupyter notebook (`.ipynb`) chunker.
|
||||
//!
|
||||
//! Notebooks are JSON documents whose `cells` array carries the code/markdown
|
||||
//! that users actually edit. This module parses the JSON, extracts each cell
|
||||
//! as its own source fragment, and assembles a *virtual source* — the
|
||||
//! concatenation of all cell bodies with one-line marker headers — that the
|
||||
//! rest of the chunk pipeline can treat like any other text file.
|
||||
//!
|
||||
//! The per-cell sub-chunks come from recursively running [`build_chunk_tree`]
|
||||
//! on the individual cell sources with their appropriate language (code cells
|
||||
//! use the notebook's kernel language, defaulting to Python; markdown cells
|
||||
//! use `markdown`; raw cells fall through to the blank-line fallback). Their
|
||||
//! byte and line offsets are shifted into the virtual source and then
|
||||
//! rewrapped under `cell_<n>` parent chunks so the resulting tree looks to
|
||||
//! edit.rs exactly like a normal multi-symbol file.
|
||||
//!
|
||||
//! On write-back, [`notebook_to_json`] walks the (possibly edited) virtual
|
||||
//! source, splits it at the cell markers, updates each cell's `source` field
|
||||
//! in a preserved [`NotebookContext`], and serializes the whole notebook back
|
||||
//! to JSON. Cell metadata (`metadata`, `outputs`, `execution_count`, `id`,
|
||||
//! attachments, etc.) is preserved verbatim.
|
||||
|
||||
use std::sync::Arc;
|
||||
|
||||
use serde::Serialize;
|
||||
use serde_json::{Map, Value};
|
||||
|
||||
use crate::chunk::{
|
||||
build_chunk_tree, chunk_checksum,
|
||||
kind::ChunkKind,
|
||||
line_start_offsets,
|
||||
types::{ChunkNode, ChunkTree},
|
||||
};
|
||||
|
||||
/// Marker line prefix placed before every cell body in the virtual source.
|
||||
///
|
||||
/// Format: `# %%% oh-my-pi cell_<N> [<type>]`
|
||||
///
|
||||
/// The leading `#` makes the marker a valid comment in Python and most other
|
||||
/// code languages, and the `oh-my-pi` tag makes accidental collision with
|
||||
/// user content vanishingly unlikely. Markdown cells get the same marker —
|
||||
/// `#` in markdown is a heading, but the marker line itself is stripped
|
||||
/// before the markdown chunker parses the cell body (see
|
||||
/// [`build_cell_sub_tree`]).
|
||||
const MARKER_PREFIX: &str = "# %%% oh-my-pi cell_";
|
||||
|
||||
/// Returns true if `line` is a cell marker; parses the cell index and type.
|
||||
fn parse_marker_line(line: &str) -> Option<(usize, &str)> {
|
||||
let rest = line.strip_prefix(MARKER_PREFIX)?;
|
||||
let (num_str, after_num) = rest.split_once(' ')?;
|
||||
let index: usize = num_str.parse().ok()?;
|
||||
let cell_type = after_num.strip_prefix('[')?.strip_suffix(']')?;
|
||||
Some((index, cell_type))
|
||||
}
|
||||
|
||||
fn format_marker(index: usize, cell_type: &str) -> String {
|
||||
format!("{MARKER_PREFIX}{index} [{cell_type}]")
|
||||
}
|
||||
|
||||
/// Metadata for a single notebook cell. Everything except the joined `source`
|
||||
/// is preserved verbatim for JSON round-tripping.
|
||||
#[derive(Clone)]
|
||||
pub struct NotebookCell {
|
||||
pub cell_type: String,
|
||||
pub source: String,
|
||||
pub metadata: Value,
|
||||
pub outputs: Option<Value>,
|
||||
pub execution_count: Option<Value>,
|
||||
/// Additional fields on the cell object (`id`, `attachments`, …) so we
|
||||
/// emit the same keys we consumed.
|
||||
pub other: Map<String, Value>,
|
||||
/// Whether the original cell used `"source"` as a string (`true`) or an
|
||||
/// array of lines (`false`). Preserved so round-trips stay byte-identical
|
||||
/// when only metadata or unrelated cells change.
|
||||
pub source_was_string: bool,
|
||||
}
|
||||
|
||||
/// Preserved notebook-level state used for rebuilding the JSON after edits.
|
||||
#[derive(Clone)]
|
||||
pub struct NotebookContext {
|
||||
/// Cells in document order.
|
||||
pub cells: Vec<NotebookCell>,
|
||||
/// Top-level notebook fields other than `cells`, in original key order.
|
||||
pub top_fields: Map<String, Value>,
|
||||
/// Indent string detected from the original JSON (typically `" "`).
|
||||
pub indent: String,
|
||||
/// Whether the original file ended with a trailing `\n`.
|
||||
pub trailing_newline: bool,
|
||||
/// Normalized kernel language used for code cells (e.g. `python`).
|
||||
pub kernel_language: String,
|
||||
}
|
||||
|
||||
/// Result of a notebook parse: the virtual source ready for chunking plus
|
||||
/// the context needed to rebuild the JSON later.
|
||||
pub struct NotebookParse {
|
||||
pub virtual_source: String,
|
||||
pub context: NotebookContext,
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
// JSON parsing
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Parse the raw ipynb JSON into a [`NotebookContext`] and the derived
|
||||
/// virtual source text. Returns a descriptive error if the JSON is invalid
|
||||
/// or not a notebook document.
|
||||
pub fn parse_notebook(source: &str) -> Result<NotebookParse, String> {
|
||||
let normalized = strip_bom(source);
|
||||
let value: Value = serde_json::from_str(normalized)
|
||||
.map_err(|err| format!("Invalid Jupyter notebook JSON: {err}"))?;
|
||||
let Value::Object(obj) = value else {
|
||||
return Err("Invalid Jupyter notebook: top-level value must be an object.".to_string());
|
||||
};
|
||||
|
||||
let cells_val = obj
|
||||
.get("cells")
|
||||
.ok_or_else(|| "Invalid Jupyter notebook: missing `cells` array.".to_string())?;
|
||||
let cells_arr = cells_val
|
||||
.as_array()
|
||||
.ok_or_else(|| "Invalid Jupyter notebook: `cells` is not an array.".to_string())?;
|
||||
|
||||
let mut cells = Vec::with_capacity(cells_arr.len());
|
||||
for (i, raw) in cells_arr.iter().enumerate() {
|
||||
cells.push(parse_cell(raw, i)?);
|
||||
}
|
||||
|
||||
// Top-level fields minus `cells`, preserving insertion order.
|
||||
let mut top_fields = Map::new();
|
||||
for (k, v) in &obj {
|
||||
if k != "cells" {
|
||||
top_fields.insert(k.clone(), v.clone());
|
||||
}
|
||||
}
|
||||
|
||||
let kernel_language =
|
||||
extract_kernel_language(&top_fields).unwrap_or_else(|| "python".to_string());
|
||||
let indent = detect_json_indent(normalized);
|
||||
let trailing_newline = normalized.ends_with('\n');
|
||||
|
||||
let ctx = NotebookContext { cells, top_fields, indent, trailing_newline, kernel_language };
|
||||
let virtual_source = build_virtual_source(&ctx);
|
||||
Ok(NotebookParse { virtual_source, context: ctx })
|
||||
}
|
||||
|
||||
fn parse_cell(raw: &Value, index: usize) -> Result<NotebookCell, String> {
|
||||
let obj = raw
|
||||
.as_object()
|
||||
.ok_or_else(|| format!("Invalid Jupyter notebook: cell {} is not an object.", index + 1))?;
|
||||
|
||||
let cell_type = obj
|
||||
.get("cell_type")
|
||||
.and_then(Value::as_str)
|
||||
.unwrap_or("code")
|
||||
.to_string();
|
||||
|
||||
let (source, source_was_string) = match obj.get("source") {
|
||||
Some(Value::String(s)) => (s.clone(), true),
|
||||
Some(Value::Array(arr)) => {
|
||||
let mut joined = String::new();
|
||||
for item in arr {
|
||||
match item {
|
||||
Value::String(s) => joined.push_str(s),
|
||||
_ => {
|
||||
return Err(format!(
|
||||
"Invalid Jupyter notebook: cell {} source array contains non-string element.",
|
||||
index + 1
|
||||
));
|
||||
},
|
||||
}
|
||||
}
|
||||
(joined, false)
|
||||
},
|
||||
Some(Value::Null) | None => (String::new(), false),
|
||||
Some(_) => {
|
||||
return Err(format!(
|
||||
"Invalid Jupyter notebook: cell {} has a non-string `source` field.",
|
||||
index + 1
|
||||
));
|
||||
},
|
||||
};
|
||||
|
||||
let metadata = obj
|
||||
.get("metadata")
|
||||
.cloned()
|
||||
.unwrap_or_else(|| Value::Object(Map::new()));
|
||||
let outputs = obj.get("outputs").cloned();
|
||||
let execution_count = obj.get("execution_count").cloned();
|
||||
|
||||
// Additional fields (id, attachments, …) that we pass through untouched.
|
||||
let mut other = Map::new();
|
||||
for (k, v) in obj {
|
||||
if !matches!(k.as_str(), "cell_type" | "source" | "metadata" | "outputs" | "execution_count")
|
||||
{
|
||||
other.insert(k.clone(), v.clone());
|
||||
}
|
||||
}
|
||||
|
||||
Ok(NotebookCell {
|
||||
cell_type,
|
||||
source,
|
||||
metadata,
|
||||
outputs,
|
||||
execution_count,
|
||||
other,
|
||||
source_was_string,
|
||||
})
|
||||
}
|
||||
|
||||
fn strip_bom(source: &str) -> &str {
|
||||
source.strip_prefix('\u{feff}').unwrap_or(source)
|
||||
}
|
||||
|
||||
fn extract_kernel_language(top_fields: &Map<String, Value>) -> Option<String> {
|
||||
if let Some(meta) = top_fields.get("metadata").and_then(Value::as_object) {
|
||||
if let Some(lang) = meta
|
||||
.get("kernelspec")
|
||||
.and_then(Value::as_object)
|
||||
.and_then(|k| k.get("language"))
|
||||
.and_then(Value::as_str)
|
||||
&& !lang.is_empty()
|
||||
{
|
||||
return Some(lang.to_ascii_lowercase());
|
||||
}
|
||||
if let Some(lang) = meta
|
||||
.get("language_info")
|
||||
.and_then(Value::as_object)
|
||||
.and_then(|k| k.get("name"))
|
||||
.and_then(Value::as_str)
|
||||
&& !lang.is_empty()
|
||||
{
|
||||
return Some(lang.to_ascii_lowercase());
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn detect_json_indent(source: &str) -> String {
|
||||
// Look for the first `\n` followed by whitespace inside the top-level object
|
||||
// (i.e. after the opening `{`). This is a heuristic; Jupyter canonically
|
||||
// uses a single space per level.
|
||||
let Some(brace) = source.find('{') else {
|
||||
return " ".to_string();
|
||||
};
|
||||
let rest = &source[brace + 1..];
|
||||
let Some(nl) = rest.find('\n') else {
|
||||
return " ".to_string();
|
||||
};
|
||||
let after_nl = &rest[nl + 1..];
|
||||
let mut end = 0usize;
|
||||
for ch in after_nl.chars() {
|
||||
if ch == ' ' || ch == '\t' {
|
||||
end += ch.len_utf8();
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if end == 0 {
|
||||
" ".to_string()
|
||||
} else {
|
||||
after_nl[..end].to_string()
|
||||
}
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
// Virtual source assembly
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Build the virtual source text from the current cell list.
|
||||
///
|
||||
/// Each cell is preceded by a marker line and its body. Cells do not include
|
||||
/// trailing blank separators — we rely purely on the marker line to delimit
|
||||
/// adjacent cells so the reconstructed sources stay byte-identical to the
|
||||
/// originals after whole-cell edits.
|
||||
pub fn build_virtual_source(ctx: &NotebookContext) -> String {
|
||||
let mut out = String::new();
|
||||
for (i, cell) in ctx.cells.iter().enumerate() {
|
||||
out.push_str(&format_marker(i + 1, &cell.cell_type));
|
||||
out.push('\n');
|
||||
out.push_str(&cell.source);
|
||||
if !cell.source.is_empty() && !cell.source.ends_with('\n') {
|
||||
out.push('\n');
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
// Chunk tree construction
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Locate every cell marker in `source`, returning tuples of:
|
||||
/// (`cell_number`, `marker_line_byte_start`, `content_byte_start`,
|
||||
/// `content_byte_end`, `marker_line_number_1based`,
|
||||
/// `content_start_line_1based`,
|
||||
/// `content_end_line_1based_inclusive_or_zero_if_empty`)
|
||||
struct CellRegion {
|
||||
cell_num: usize,
|
||||
cell_type: String,
|
||||
marker_start: usize, // byte offset of the `#` starting the marker line
|
||||
content_start: usize, // byte offset of the first byte of the cell body
|
||||
content_end: usize, // byte offset one past the last byte of the cell body
|
||||
marker_line: u32, // 1-based line number of the marker
|
||||
content_line: u32, // 1-based line number of the first body line (or marker_line + 1)
|
||||
content_end_line: u32, /* 1-based line number of the last body line (== content_line - 1
|
||||
* for empty bodies) */
|
||||
}
|
||||
|
||||
/// Scan a virtual source text for cell markers and return the list of
|
||||
/// regions. Assumes markers occur at the very start of their line.
|
||||
fn scan_cells(virtual_source: &str) -> Vec<CellRegion> {
|
||||
let line_starts = line_start_offsets(virtual_source);
|
||||
let mut regions: Vec<CellRegion> = Vec::new();
|
||||
|
||||
for (line_idx, &line_start) in line_starts.iter().enumerate() {
|
||||
let line_end = if line_idx + 1 < line_starts.len() {
|
||||
// Exclude the trailing newline
|
||||
line_starts[line_idx + 1] - 1
|
||||
} else {
|
||||
virtual_source.len()
|
||||
};
|
||||
let line = &virtual_source[line_start..line_end];
|
||||
if let Some((cell_num, cell_type)) = parse_marker_line(line) {
|
||||
// Close the previous region if any.
|
||||
if let Some(prev) = regions.last_mut() {
|
||||
prev.content_end = line_start;
|
||||
// Trim trailing newline from content_end if present (i.e. the body ended with
|
||||
// \n). Actually we keep the newline: cell bodies end with \n except
|
||||
// possibly the last. content_end_line = line of the last body byte.
|
||||
if prev.content_end > prev.content_start {
|
||||
let body_last_char_line = line_idx; // line_idx is 0-based, so this is the previous line
|
||||
prev.content_end_line = body_last_char_line as u32;
|
||||
} else {
|
||||
prev.content_end_line = prev.content_line.saturating_sub(1);
|
||||
}
|
||||
}
|
||||
let content_start = if line_idx + 1 < line_starts.len() {
|
||||
line_starts[line_idx + 1]
|
||||
} else {
|
||||
virtual_source.len()
|
||||
};
|
||||
regions.push(CellRegion {
|
||||
cell_num,
|
||||
cell_type: cell_type.to_string(),
|
||||
marker_start: line_start,
|
||||
content_start,
|
||||
content_end: virtual_source.len(), // provisional, closed by the next marker
|
||||
marker_line: (line_idx as u32) + 1,
|
||||
content_line: (line_idx as u32) + 2,
|
||||
content_end_line: 0,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Close the last region.
|
||||
if let Some(last) = regions.last_mut() {
|
||||
last.content_end = virtual_source.len();
|
||||
if last.content_end > last.content_start {
|
||||
// Count lines inside the body.
|
||||
let body = &virtual_source[last.content_start..last.content_end];
|
||||
let body_lines = body.matches('\n').count();
|
||||
// If body doesn't end with '\n', the final partial line still counts.
|
||||
let has_trailing_nl = body.ends_with('\n');
|
||||
let content_lines = if has_trailing_nl {
|
||||
body_lines
|
||||
} else {
|
||||
body_lines + 1
|
||||
};
|
||||
if content_lines > 0 {
|
||||
last.content_end_line = last.content_line + content_lines as u32 - 1;
|
||||
} else {
|
||||
last.content_end_line = last.content_line.saturating_sub(1);
|
||||
}
|
||||
} else {
|
||||
last.content_end_line = last.content_line.saturating_sub(1);
|
||||
}
|
||||
}
|
||||
|
||||
regions
|
||||
}
|
||||
|
||||
/// Build a chunk tree from a virtual source text.
|
||||
///
|
||||
/// Re-scans the virtual source for cell markers, parses each cell body with
|
||||
/// its language, and wraps the results in `cell_<n>` parent chunks. This is
|
||||
/// the entry point used by both the initial JSON-based parse (via
|
||||
/// [`parse_notebook`] → `build_virtual_source` → this function) and the
|
||||
/// post-edit rebuilds that operate directly on the mutated virtual source.
|
||||
pub fn build_notebook_tree_from_virtual(
|
||||
virtual_source: &str,
|
||||
kernel_language: &str,
|
||||
) -> Result<ChunkTree, String> {
|
||||
let total_lines = total_line_count(virtual_source);
|
||||
let root_checksum = chunk_checksum(virtual_source.as_bytes());
|
||||
let regions = scan_cells(virtual_source);
|
||||
|
||||
// Accumulated chunk nodes. Index 0 is reserved for the synthetic root.
|
||||
let mut chunks: Vec<ChunkNode> = Vec::with_capacity(1 + regions.len() * 2);
|
||||
chunks.push(ChunkNode {
|
||||
path: String::new(),
|
||||
identifier: None,
|
||||
kind: ChunkKind::Root,
|
||||
leaf: false,
|
||||
virtual_content: None,
|
||||
parent_path: None,
|
||||
children: Vec::new(),
|
||||
signature: None,
|
||||
start_line: u32::from(total_lines != 0),
|
||||
end_line: total_lines as u32,
|
||||
line_count: total_lines as u32,
|
||||
start_byte: 0,
|
||||
end_byte: virtual_source.len() as u32,
|
||||
checksum_start_byte: 0,
|
||||
prologue_end_byte: Some(0),
|
||||
epilogue_start_byte: Some(virtual_source.len() as u32),
|
||||
checksum: root_checksum.clone(),
|
||||
error: false,
|
||||
indent: 0,
|
||||
indent_char: String::new(),
|
||||
group: false,
|
||||
});
|
||||
|
||||
let mut root_children: Vec<String> = Vec::with_capacity(regions.len());
|
||||
|
||||
for region in ®ions {
|
||||
let cell_path = format!("cell_{}", region.cell_num);
|
||||
let cell_language_str = match region.cell_type.as_str() {
|
||||
"code" => kernel_language.to_string(),
|
||||
"markdown" => "markdown".to_string(),
|
||||
_ => String::new(),
|
||||
};
|
||||
|
||||
let body = &virtual_source[region.content_start..region.content_end];
|
||||
let body_has_content = !body.is_empty();
|
||||
let cell_checksum = chunk_checksum(body.as_bytes());
|
||||
|
||||
// Build sub-chunks by parsing the cell body in isolation. Offsets in
|
||||
// the returned tree are relative to `body`; we translate them into
|
||||
// virtual-source coordinates by adding `region.content_start` bytes
|
||||
// and `region.content_line - 1` lines.
|
||||
let mut cell_children_paths: Vec<String> = Vec::new();
|
||||
if body_has_content {
|
||||
let sub_tree = build_chunk_tree(body, cell_language_str.as_str())
|
||||
.map_err(|err| format!("Failed to parse cell_{} body: {err}", region.cell_num))?;
|
||||
for sub_chunk in sub_tree.chunks.into_iter().skip(1) {
|
||||
let translated_path = format!("{}.{}", cell_path, sub_chunk.path);
|
||||
let translated_parent = match sub_chunk.parent_path.as_deref() {
|
||||
Some("") | None => Some(cell_path.clone()),
|
||||
Some(other) => Some(format!("{cell_path}.{other}")),
|
||||
};
|
||||
let translated_children: Vec<String> = sub_chunk
|
||||
.children
|
||||
.iter()
|
||||
.map(|c| format!("{cell_path}.{c}"))
|
||||
.collect();
|
||||
let shifted_start_byte = sub_chunk
|
||||
.start_byte
|
||||
.saturating_add(region.content_start as u32);
|
||||
let shifted_end_byte = sub_chunk
|
||||
.end_byte
|
||||
.saturating_add(region.content_start as u32);
|
||||
let line_shift = region.content_line.saturating_sub(1);
|
||||
chunks.push(ChunkNode {
|
||||
path: translated_path.clone(),
|
||||
identifier: sub_chunk.identifier,
|
||||
kind: sub_chunk.kind,
|
||||
leaf: sub_chunk.leaf,
|
||||
virtual_content: sub_chunk.virtual_content,
|
||||
parent_path: translated_parent,
|
||||
children: translated_children,
|
||||
signature: sub_chunk.signature,
|
||||
start_line: sub_chunk.start_line.saturating_add(line_shift),
|
||||
end_line: sub_chunk.end_line.saturating_add(line_shift),
|
||||
line_count: sub_chunk.line_count,
|
||||
start_byte: shifted_start_byte,
|
||||
end_byte: shifted_end_byte,
|
||||
checksum_start_byte: sub_chunk
|
||||
.checksum_start_byte
|
||||
.saturating_add(region.content_start as u32),
|
||||
prologue_end_byte: sub_chunk
|
||||
.prologue_end_byte
|
||||
.map(|b| b.saturating_add(region.content_start as u32)),
|
||||
epilogue_start_byte: sub_chunk
|
||||
.epilogue_start_byte
|
||||
.map(|b| b.saturating_add(region.content_start as u32)),
|
||||
checksum: sub_chunk.checksum,
|
||||
error: sub_chunk.error,
|
||||
indent: sub_chunk.indent,
|
||||
indent_char: sub_chunk.indent_char,
|
||||
group: false,
|
||||
});
|
||||
}
|
||||
for sub_path in sub_tree.root_children {
|
||||
cell_children_paths.push(format!("{cell_path}.{sub_path}"));
|
||||
}
|
||||
}
|
||||
|
||||
let cell_line_count = {
|
||||
let body_lines = if body_has_content {
|
||||
if body.ends_with('\n') {
|
||||
body.matches('\n').count()
|
||||
} else {
|
||||
body.matches('\n').count() + 1
|
||||
}
|
||||
} else {
|
||||
0
|
||||
};
|
||||
1 + body_lines as u32
|
||||
};
|
||||
let cell_end_line = region.marker_line + cell_line_count.saturating_sub(1);
|
||||
let cell_leaf = cell_children_paths.is_empty();
|
||||
chunks.push(ChunkNode {
|
||||
path: cell_path.clone(),
|
||||
identifier: Some(cell_path.clone()),
|
||||
kind: ChunkKind::Cell,
|
||||
leaf: cell_leaf,
|
||||
virtual_content: None,
|
||||
parent_path: Some(String::new()),
|
||||
children: cell_children_paths,
|
||||
signature: Some(format!("cell_{} ({})", region.cell_num, region.cell_type)),
|
||||
start_line: region.marker_line,
|
||||
end_line: cell_end_line,
|
||||
line_count: cell_line_count,
|
||||
start_byte: region.marker_start as u32,
|
||||
end_byte: region.content_end as u32,
|
||||
checksum_start_byte: region.content_start as u32,
|
||||
prologue_end_byte: Some(region.content_start as u32),
|
||||
epilogue_start_byte: Some(region.content_end as u32),
|
||||
checksum: cell_checksum,
|
||||
error: false,
|
||||
indent: 0,
|
||||
indent_char: String::new(),
|
||||
group: false,
|
||||
});
|
||||
root_children.push(cell_path);
|
||||
}
|
||||
|
||||
// Populate root children now that every cell is known.
|
||||
if let Some(root) = chunks.get_mut(0) {
|
||||
root.children.clone_from(&root_children);
|
||||
}
|
||||
|
||||
// Sort chunks so the cell parent always comes before its sub-chunks,
|
||||
// matching the invariant that other paths rely on (render, edit
|
||||
// scheduling, line-to-chunk lookup). Keep the root at index 0.
|
||||
// The insertion order above places sub-chunks before the cell parent, so
|
||||
// we need to reorder: for each cell region, move the cell parent ahead of
|
||||
// its sub-chunks.
|
||||
//
|
||||
// Simpler: rebuild the chunks list by iterating cells, emitting the cell
|
||||
// parent followed by its sub-chunks in path order.
|
||||
let mut reordered: Vec<ChunkNode> = Vec::with_capacity(chunks.len());
|
||||
reordered.push(chunks.remove(0)); // root
|
||||
|
||||
let mut remaining: Vec<ChunkNode> = chunks;
|
||||
for cell_path in &root_children {
|
||||
// Extract the cell parent first.
|
||||
if let Some(pos) = remaining.iter().position(|c| &c.path == cell_path) {
|
||||
reordered.push(remaining.remove(pos));
|
||||
}
|
||||
// Then any descendants of this cell.
|
||||
let prefix = format!("{cell_path}.");
|
||||
let mut i = 0;
|
||||
while i < remaining.len() {
|
||||
if remaining[i].path.starts_with(&prefix) {
|
||||
reordered.push(remaining.remove(i));
|
||||
} else {
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Anything left over (shouldn't happen, but be defensive).
|
||||
reordered.extend(remaining);
|
||||
|
||||
Ok(ChunkTree {
|
||||
language: "ipynb".to_string(),
|
||||
checksum: root_checksum,
|
||||
line_count: total_lines as u32,
|
||||
parse_errors: 0,
|
||||
parse_error_lines: Vec::new(),
|
||||
fallback: false,
|
||||
root_path: String::new(),
|
||||
root_children,
|
||||
chunks: reordered,
|
||||
})
|
||||
}
|
||||
|
||||
fn total_line_count(source: &str) -> usize {
|
||||
if source.is_empty() {
|
||||
0
|
||||
} else {
|
||||
source.bytes().filter(|b| *b == b'\n').count() + 1
|
||||
}
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
// Virtual → JSON round-trip
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Update a [`NotebookContext`] from a (possibly edited) virtual source,
|
||||
/// then serialize it back to JSON. Cells that no longer appear in the
|
||||
/// virtual source are dropped; cells whose markers survive get their
|
||||
/// `source` field replaced with the current body.
|
||||
///
|
||||
/// Returns the serialized JSON text ready to be written to disk.
|
||||
pub fn notebook_to_json(
|
||||
virtual_source: &str,
|
||||
base_ctx: &NotebookContext,
|
||||
) -> Result<String, String> {
|
||||
let mut ctx = base_ctx.clone();
|
||||
let regions = scan_cells(virtual_source);
|
||||
|
||||
// Rebuild the cells array in the order markers appear in the virtual
|
||||
// source. Look up each marker's original cell by 1-based cell_num so
|
||||
// edits that reorder cells via sibling insertion continue to track the
|
||||
// right metadata.
|
||||
let mut new_cells: Vec<NotebookCell> = Vec::with_capacity(regions.len());
|
||||
for region in ®ions {
|
||||
let body_slice = &virtual_source[region.content_start..region.content_end];
|
||||
// Trim the single trailing newline that the virtual source format
|
||||
// adds so edits that replace an entire cell body don't grow by one
|
||||
// line every round-trip.
|
||||
let body = trim_virtual_body(body_slice);
|
||||
|
||||
let original = ctx.cells.get(region.cell_num.saturating_sub(1)).cloned();
|
||||
let cell = match original {
|
||||
Some(mut cell) => {
|
||||
cell.source = body.to_string();
|
||||
cell.cell_type.clone_from(®ion.cell_type);
|
||||
cell
|
||||
},
|
||||
None => NotebookCell {
|
||||
cell_type: region.cell_type.clone(),
|
||||
source: body.to_string(),
|
||||
metadata: Value::Object(Map::new()),
|
||||
outputs: match region.cell_type.as_str() {
|
||||
"code" => Some(Value::Array(Vec::new())),
|
||||
_ => None,
|
||||
},
|
||||
execution_count: match region.cell_type.as_str() {
|
||||
"code" => Some(Value::Null),
|
||||
_ => None,
|
||||
},
|
||||
other: Map::new(),
|
||||
source_was_string: false,
|
||||
},
|
||||
};
|
||||
new_cells.push(cell);
|
||||
}
|
||||
ctx.cells = new_cells;
|
||||
|
||||
let json = serialize_notebook(&ctx)?;
|
||||
Ok(json)
|
||||
}
|
||||
|
||||
/// Strip a single trailing newline from `body`, if present. The virtual
|
||||
/// source always terminates each cell body with `\n` to make the markers
|
||||
/// start on a fresh line; we remove that byte so the cell's stored source
|
||||
/// matches the semantic content.
|
||||
fn trim_virtual_body(body: &str) -> &str {
|
||||
body.strip_suffix('\n').unwrap_or(body)
|
||||
}
|
||||
|
||||
fn serialize_notebook(ctx: &NotebookContext) -> Result<String, String> {
|
||||
// Build the cells array first.
|
||||
let mut cells_arr: Vec<Value> = Vec::with_capacity(ctx.cells.len());
|
||||
for cell in &ctx.cells {
|
||||
cells_arr.push(cell_to_value(cell));
|
||||
}
|
||||
|
||||
// Rebuild the top-level object preserving the original key order with
|
||||
// `cells` injected at the position it originally occupied. If the input
|
||||
// had no `cells` key (we wouldn't be here), we append.
|
||||
let mut top = Map::new();
|
||||
let mut cells_inserted = false;
|
||||
for (k, v) in &ctx.top_fields {
|
||||
top.insert(k.clone(), v.clone());
|
||||
if k == "metadata" && !cells_inserted {
|
||||
// Jupyter's canonical order is cells, metadata, nbformat,
|
||||
// nbformat_minor. We preserve whatever we found.
|
||||
}
|
||||
}
|
||||
// If the original document had `cells` somewhere, we want to re-insert
|
||||
// it at roughly the same slot. Jupyter always writes `cells` first, so
|
||||
// build a fresh Map in canonical order: cells then the preserved
|
||||
// top_fields.
|
||||
let mut final_top = Map::new();
|
||||
final_top.insert("cells".to_string(), Value::Array(cells_arr));
|
||||
cells_inserted = true;
|
||||
for (k, v) in top {
|
||||
if k != "cells" {
|
||||
final_top.insert(k, v);
|
||||
}
|
||||
}
|
||||
let _ = cells_inserted;
|
||||
|
||||
let indent_bytes = ctx.indent.as_bytes().to_vec();
|
||||
let formatter = serde_json::ser::PrettyFormatter::with_indent(&indent_bytes);
|
||||
let mut buf: Vec<u8> = Vec::with_capacity(1024);
|
||||
{
|
||||
let mut ser = serde_json::Serializer::with_formatter(&mut buf, formatter);
|
||||
Value::Object(final_top)
|
||||
.serialize(&mut ser)
|
||||
.map_err(|err| format!("Failed to serialize notebook JSON: {err}"))?;
|
||||
}
|
||||
let mut text = String::from_utf8(buf)
|
||||
.map_err(|err| format!("Serialized notebook is not valid UTF-8: {err}"))?;
|
||||
if ctx.trailing_newline && !text.ends_with('\n') {
|
||||
text.push('\n');
|
||||
}
|
||||
Ok(text)
|
||||
}
|
||||
|
||||
fn cell_to_value(cell: &NotebookCell) -> Value {
|
||||
let mut obj = Map::new();
|
||||
obj.insert("cell_type".to_string(), Value::String(cell.cell_type.clone()));
|
||||
// Preserve `id` and similar fields that idiomatically appear before
|
||||
// `metadata` in nbformat 4+.
|
||||
for (k, v) in &cell.other {
|
||||
if !matches!(k.as_str(), "metadata" | "outputs" | "execution_count" | "source") {
|
||||
obj.insert(k.clone(), v.clone());
|
||||
}
|
||||
}
|
||||
obj.insert("metadata".to_string(), cell.metadata.clone());
|
||||
if cell.cell_type == "code" {
|
||||
obj.insert(
|
||||
"execution_count".to_string(),
|
||||
cell.execution_count.clone().unwrap_or(Value::Null),
|
||||
);
|
||||
obj.insert(
|
||||
"outputs".to_string(),
|
||||
cell
|
||||
.outputs
|
||||
.clone()
|
||||
.unwrap_or_else(|| Value::Array(Vec::new())),
|
||||
);
|
||||
} else {
|
||||
if let Some(outputs) = &cell.outputs {
|
||||
obj.insert("outputs".to_string(), outputs.clone());
|
||||
}
|
||||
if let Some(ec) = &cell.execution_count {
|
||||
obj.insert("execution_count".to_string(), ec.clone());
|
||||
}
|
||||
}
|
||||
obj.insert("source".to_string(), source_to_value(&cell.source, cell.source_was_string));
|
||||
Value::Object(obj)
|
||||
}
|
||||
|
||||
/// Convert a flat source string to the Jupyter `source` field representation.
|
||||
///
|
||||
/// If the cell originally used a string (or a new cell was inserted), we keep
|
||||
/// it as a string. Otherwise we split into the canonical `Vec<String>` with
|
||||
/// each element preserving its trailing `\n`.
|
||||
fn source_to_value(source: &str, was_string: bool) -> Value {
|
||||
if was_string {
|
||||
return Value::String(source.to_string());
|
||||
}
|
||||
if source.is_empty() {
|
||||
return Value::Array(Vec::new());
|
||||
}
|
||||
let mut parts: Vec<Value> = Vec::new();
|
||||
for line in source.split_inclusive('\n') {
|
||||
parts.push(Value::String(line.to_string()));
|
||||
}
|
||||
Value::Array(parts)
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
// Shared helpers
|
||||
// ────────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Atomically-shareable notebook context; used by [`ChunkStateInner`] to
|
||||
/// carry the notebook metadata through edit cycles.
|
||||
pub type SharedNotebookContext = Arc<NotebookContext>;
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use serde_json::json;
|
||||
|
||||
use super::*;
|
||||
use crate::chunk::state::ChunkStateInner;
|
||||
|
||||
fn sample_notebook() -> String {
|
||||
let value = json!({
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "code",
|
||||
"source": ["def foo():\n", " return 1\n"],
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"execution_count": null
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": ["# Hello\n", "World\n"],
|
||||
"metadata": {}
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"source": ["class Bar:\n", " def baz(self):\n", " pass\n"],
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"execution_count": 3
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"kernelspec": {
|
||||
"language": "python"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 5
|
||||
});
|
||||
serde_json::to_string_pretty(&value).expect("static json")
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_notebook_into_cells() {
|
||||
let nb = parse_notebook(&sample_notebook()).expect("valid notebook");
|
||||
assert_eq!(nb.context.cells.len(), 3);
|
||||
assert_eq!(nb.context.cells[0].cell_type, "code");
|
||||
assert_eq!(nb.context.cells[1].cell_type, "markdown");
|
||||
assert_eq!(nb.context.cells[2].cell_type, "code");
|
||||
assert_eq!(nb.context.kernel_language, "python");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn virtual_source_contains_all_cells() {
|
||||
let nb = parse_notebook(&sample_notebook()).expect("valid notebook");
|
||||
let vs = &nb.virtual_source;
|
||||
assert!(vs.contains("def foo():"), "cell 1 body missing");
|
||||
assert!(vs.contains("# Hello"), "cell 2 body missing");
|
||||
assert!(vs.contains("class Bar:"), "cell 3 body missing");
|
||||
assert!(vs.contains("# %%% oh-my-pi cell_1 [code]"), "cell_1 marker missing");
|
||||
assert!(vs.contains("# %%% oh-my-pi cell_2 [markdown]"), "cell_2 marker missing");
|
||||
assert!(vs.contains("# %%% oh-my-pi cell_3 [code]"), "cell_3 marker missing");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn builds_cell_level_chunks() {
|
||||
let nb = parse_notebook(&sample_notebook()).expect("valid notebook");
|
||||
let tree =
|
||||
build_notebook_tree_from_virtual(&nb.virtual_source, "python").expect("tree should build");
|
||||
assert_eq!(
|
||||
tree.root_children,
|
||||
vec!["cell_1", "cell_2", "cell_3"],
|
||||
"root children should be the three cells"
|
||||
);
|
||||
let cell1 = tree
|
||||
.chunks
|
||||
.iter()
|
||||
.find(|c| c.path == "cell_1")
|
||||
.expect("cell_1 chunk");
|
||||
assert!(!cell1.leaf, "code cell with a function should not be a leaf");
|
||||
assert!(
|
||||
cell1
|
||||
.children
|
||||
.iter()
|
||||
.any(|p| p.starts_with("cell_1.fn_foo")),
|
||||
"cell_1 should contain fn_foo, got {:?}",
|
||||
cell1.children
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sub_chunk_paths_are_prefixed_with_cell() {
|
||||
let nb = parse_notebook(&sample_notebook()).expect("valid notebook");
|
||||
let tree =
|
||||
build_notebook_tree_from_virtual(&nb.virtual_source, "python").expect("tree should build");
|
||||
let cell3 = tree
|
||||
.chunks
|
||||
.iter()
|
||||
.find(|c| c.path == "cell_3")
|
||||
.expect("cell_3 chunk");
|
||||
assert!(
|
||||
cell3
|
||||
.children
|
||||
.iter()
|
||||
.any(|p| p.starts_with("cell_3.cls_Bar")),
|
||||
"cell_3 should contain cls_Bar, got {:?}",
|
||||
cell3.children
|
||||
);
|
||||
let bar_method = tree
|
||||
.chunks
|
||||
.iter()
|
||||
.find(|c| c.path == "cell_3.cls_Bar.fn_baz");
|
||||
assert!(bar_method.is_some(), "cell_3.cls_Bar.fn_baz should exist");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chunk_state_parse_ipynb_carries_notebook_context() {
|
||||
let json = sample_notebook();
|
||||
let state =
|
||||
ChunkStateInner::parse(json, "ipynb".to_string()).expect("ChunkState should parse ipynb");
|
||||
assert_eq!(state.language(), "ipynb");
|
||||
// Source is the virtual source, not the JSON
|
||||
assert!(state.source().contains("# %%% oh-my-pi cell_1"));
|
||||
// The notebook context is preserved for JSON round-trip
|
||||
let ctx = state
|
||||
.notebook
|
||||
.as_ref()
|
||||
.expect("notebook context should be set");
|
||||
let json_out = notebook_to_json(state.source(), ctx).expect("should serialize back to JSON");
|
||||
let reparsed: serde_json::Value =
|
||||
serde_json::from_str(&json_out).expect("output should be valid JSON");
|
||||
let cells = reparsed["cells"].as_array().expect("cells array");
|
||||
assert_eq!(cells.len(), 3, "should still have 3 cells");
|
||||
assert_eq!(cells[0]["cell_type"], "code");
|
||||
assert_eq!(cells[2]["execution_count"], 3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_notebook_produces_empty_tree() {
|
||||
let json = r#"{"cells": [], "metadata": {}, "nbformat": 4, "nbformat_minor": 5}"#;
|
||||
let state = ChunkStateInner::parse(json.to_string(), "ipynb".to_string())
|
||||
.expect("empty notebook should parse");
|
||||
assert!(state.tree().root_children.is_empty());
|
||||
}
|
||||
}
|
||||
@@ -1,564 +0,0 @@
|
||||
//! JavaScript / TypeScript / TSX chunk classifier.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, WrapperSignature,
|
||||
WrapperTransform, classify_with_defaults, first_wrapper_content_child,
|
||||
promote_wrapper_candidate, semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct JsTsClassifier;
|
||||
|
||||
fn recurse_internal_module(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
recurse_into(node, ChunkContext::ClassBody, &["body"], &["statement_block"])
|
||||
}
|
||||
|
||||
static JSTS_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[
|
||||
semantic_rule(
|
||||
"import_statement",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"import_declaration",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"function_declaration",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"function",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"function_expression",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"arrow_function",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"generator_function",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"generator_function_declaration",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"class_declaration",
|
||||
ChunkKind::Class,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"class",
|
||||
ChunkKind::Class,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"class_expression",
|
||||
ChunkKind::Class,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"interface_declaration",
|
||||
ChunkKind::Interface,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"enum_declaration",
|
||||
ChunkKind::Enum,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"type_alias_declaration",
|
||||
ChunkKind::Type,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::None,
|
||||
),
|
||||
],
|
||||
class: &[
|
||||
semantic_rule(
|
||||
"constructor",
|
||||
ChunkKind::Constructor,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"class_static_block",
|
||||
ChunkKind::StaticInit,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"type_alias_declaration",
|
||||
ChunkKind::Type,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::None,
|
||||
),
|
||||
],
|
||||
function: &[],
|
||||
structural_overrides: super::classify::StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
impl LangClassifier for JsTsClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&JSTS_TABLES
|
||||
}
|
||||
|
||||
fn is_trivia(&self, kind: &str) -> bool {
|
||||
// Whitespace/text runs between JSX elements carry no structure and
|
||||
// should be absorbed as leading trivia of the next element (matching
|
||||
// the existing comment-absorption semantics).
|
||||
kind == "jsx_text"
|
||||
}
|
||||
|
||||
fn should_skip_child(&self, kind: &str) -> bool {
|
||||
// JSX opening and closing elements are part of the enclosing
|
||||
// `jsx_element` chunk's framing, not children in their own right.
|
||||
// Skip them entirely when enumerating children so they don't pollute
|
||||
// the chunk tree with noisy 1‑line entries and, crucially, so they
|
||||
// don't get absorbed backward into the next real child.
|
||||
matches!(kind, "jsx_opening_element" | "jsx_closing_element")
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_root_custom(node, source),
|
||||
ChunkContext::ClassBody => classify_class_custom(node, source),
|
||||
ChunkContext::FunctionBody => Some(classify_function_js(node, source)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// ── Exports / decorators ──
|
||||
"export_statement" => Some(classify_export_statement(ChunkContext::Root, node, source)),
|
||||
"decorated_definition" => promote_wrapper_candidate(
|
||||
&JsTsClassifier,
|
||||
ChunkContext::Root,
|
||||
node,
|
||||
source,
|
||||
WrapperTransform { signature: WrapperSignature::Wrapper, ..WrapperTransform::default() },
|
||||
)
|
||||
.or_else(|| Some(positional_candidate(node, ChunkKind::Block, source))),
|
||||
|
||||
// ── Variables ──
|
||||
"lexical_declaration" | "variable_declaration" => Some(classify_var_decl_js(node, source)),
|
||||
|
||||
// ── Containers with custom recursion ──
|
||||
"internal_module" => {
|
||||
Some(container_candidate(node, ChunkKind::Module, source, recurse_internal_module(node)))
|
||||
},
|
||||
|
||||
// ── Control flow at top level ──
|
||||
"if_statement" | "switch_statement" | "switch_expression" | "try_statement"
|
||||
| "for_statement" | "for_in_statement" | "for_of_statement" | "while_statement"
|
||||
| "do_statement" | "with_statement" => Some(classify_function_js(node, source)),
|
||||
|
||||
// ── Statements ──
|
||||
"expression_statement" => {
|
||||
// Unwrap `expression_statement` wrapping an `internal_module` (namespace).
|
||||
let inner = named_children(node)
|
||||
.into_iter()
|
||||
.find(|c| c.kind() == "internal_module");
|
||||
if let Some(ns) = inner {
|
||||
Some(container_candidate(ns, ChunkKind::Module, source, recurse_internal_module(ns)))
|
||||
} else {
|
||||
Some(group_candidate(node, ChunkKind::Statements, source))
|
||||
}
|
||||
},
|
||||
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// ── Exports / decorators (re-exported members) ──
|
||||
"export_statement" => Some(classify_export_statement(ChunkContext::ClassBody, node, source)),
|
||||
"decorated_definition" => promote_wrapper_candidate(
|
||||
&JsTsClassifier,
|
||||
ChunkContext::ClassBody,
|
||||
node,
|
||||
source,
|
||||
WrapperTransform { signature: WrapperSignature::Wrapper, ..WrapperTransform::default() },
|
||||
)
|
||||
.or_else(|| Some(positional_candidate(node, ChunkKind::Block, source))),
|
||||
|
||||
// ── Variables ──
|
||||
"lexical_declaration" | "variable_declaration" => Some(classify_var_decl_js(node, source)),
|
||||
|
||||
// ── Methods ──
|
||||
"method_definition" | "method_signature" | "abstract_method_signature" => {
|
||||
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
if name == "constructor" {
|
||||
Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Constructor,
|
||||
None,
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
))
|
||||
} else {
|
||||
Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_body(node, ChunkContext::FunctionBody),
|
||||
))
|
||||
}
|
||||
},
|
||||
|
||||
// ── Fields ──
|
||||
"public_field_definition"
|
||||
| "field_definition"
|
||||
| "property_definition"
|
||||
| "property_signature"
|
||||
| "property_declaration"
|
||||
| "abstract_class_field" => match extract_identifier(node, source) {
|
||||
Some(name) => Some(make_kind_chunk(node, ChunkKind::Field, Some(name), source, None)),
|
||||
None => Some(group_candidate(node, ChunkKind::Fields, source)),
|
||||
},
|
||||
|
||||
// ── Enum members ──
|
||||
"enum_assignment" | "enum_member_declaration" => match extract_identifier(node, source) {
|
||||
Some(name) => Some(make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None)),
|
||||
None => Some(group_candidate(node, ChunkKind::Variants, source)),
|
||||
},
|
||||
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Classify nodes inside a function body for JS/TS.
|
||||
fn classify_function_js<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
|
||||
match node.kind() {
|
||||
// ── Control flow ──
|
||||
"if_statement" => {
|
||||
make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"switch_statement" | "switch_expression" => {
|
||||
make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"try_statement" => {
|
||||
make_candidate(node, ChunkKind::Try, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
|
||||
// ── Loops ──
|
||||
"for_statement" => {
|
||||
make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"for_in_statement" => {
|
||||
make_candidate(node, ChunkKind::ForIn, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"for_of_statement" => {
|
||||
make_candidate(node, ChunkKind::ForOf, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"while_statement" => {
|
||||
make_candidate(node, ChunkKind::While, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
"do_statement" => {
|
||||
make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
|
||||
// ── Blocks ──
|
||||
"with_statement" => {
|
||||
make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source)
|
||||
},
|
||||
|
||||
// ── Variables ──
|
||||
"lexical_declaration" | "variable_declaration" => {
|
||||
if let Some(name) = extract_single_declarator_name(node, source) {
|
||||
make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)
|
||||
} else {
|
||||
group_from_sanitized(node, source)
|
||||
}
|
||||
},
|
||||
|
||||
// ── Return statements ──
|
||||
// A bare `return <Link>…</Link>` or `return (<Link>…</Link>)` creates a
|
||||
// huge monolithic leaf chunk in React components. Recurse into the JSX
|
||||
// so each child element inside the returned tree stays individually
|
||||
// addressable. Callback-with-trailing-block patterns such as
|
||||
// `return items.map(item => { … })` are handled by the shared
|
||||
// call-with-callback promotion in `classify_with_defaults` via the
|
||||
// `return_statement` arm below.
|
||||
"return_statement" => classify_return_statement_js(node, source),
|
||||
|
||||
// ── JSX elements ──
|
||||
// Inside function bodies, JSX elements become container chunks with
|
||||
// their tag name so React component trees are navigable instead of
|
||||
// opaque walls of markup.
|
||||
"jsx_element" => classify_jsx_element(node, source),
|
||||
"jsx_self_closing_element" => classify_jsx_self_closing_element(node, source),
|
||||
"jsx_fragment" => make_candidate(
|
||||
node,
|
||||
ChunkKind::Tag,
|
||||
Some("fragment".to_string()),
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
source,
|
||||
),
|
||||
|
||||
// ── Expression statements (enable call-with-callback promotion) ──
|
||||
"expression_statement" => group_candidate(node, ChunkKind::Statements, source),
|
||||
|
||||
// ── Fallback ──
|
||||
_ => group_from_sanitized(node, source),
|
||||
}
|
||||
}
|
||||
|
||||
/// Classify a `jsx_element` as a container chunk named after its tag.
|
||||
///
|
||||
/// The chunk recurses into itself so that nested JSX children are emitted as
|
||||
/// sub-chunks. Structural JSX nodes (opening/closing elements, text,
|
||||
/// attributes) are filtered out as trivia by the classifier's `is_trivia`
|
||||
/// override, so only meaningful children (child elements, expression
|
||||
/// containers) become chunks.
|
||||
fn classify_jsx_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let tag_name = extract_jsx_tag_name(node, source);
|
||||
let mut candidate = make_candidate(
|
||||
node,
|
||||
ChunkKind::Tag,
|
||||
tag_name,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
source,
|
||||
);
|
||||
// Force recursion for jsx_elements that span more than a single
|
||||
// source line. Without this, a `<div>` wrapping a single
|
||||
// near-equal-sized child fails `recursion_narrows_scope` and the
|
||||
// whole subtree collapses into one opaque chunk. One-line elements
|
||||
// keep natural collapse behavior so short inline JSX stays a leaf.
|
||||
if candidate
|
||||
.range_end_line
|
||||
.saturating_sub(candidate.range_start_line)
|
||||
> 0
|
||||
{
|
||||
candidate.force_recurse = true;
|
||||
}
|
||||
candidate
|
||||
}
|
||||
|
||||
/// Classify a `return_statement`.
|
||||
///
|
||||
/// Two patterns matter:
|
||||
///
|
||||
/// 1. `return <Link>…</Link>` / `return (<Link>…</Link>)` — unwrap any
|
||||
/// parentheses and recurse directly into the JSX tree so each nested JSX
|
||||
/// element is individually addressable.
|
||||
/// 2. `return items.map(item => { … })` — the shared call-with-trailing-
|
||||
/// callback promoter turns this into a named expression container that
|
||||
/// recurses into the callback body. We invoke it explicitly here because the
|
||||
/// shared promotion in `classify_with_defaults` only runs on groupable
|
||||
/// leaves, and `Return` is not groupable.
|
||||
fn classify_return_statement_js<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
if let Some(expr) = named_children(node).into_iter().next() {
|
||||
let target = unwrap_parenthesized(expr);
|
||||
if matches!(target.kind(), "jsx_element" | "jsx_fragment" | "jsx_self_closing_element") {
|
||||
let mut candidate = make_candidate(
|
||||
node,
|
||||
ChunkKind::Return,
|
||||
None::<String>,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(RecurseSpec { node: target, context: ChunkContext::FunctionBody }),
|
||||
source,
|
||||
);
|
||||
// The JSX tree may span nearly the entire return statement, which
|
||||
// would fail the `recursion_narrows_scope` check. Force recursion
|
||||
// so the JSX children are always individually addressable.
|
||||
candidate.force_recurse = true;
|
||||
return candidate;
|
||||
}
|
||||
}
|
||||
if let Some(mut promoted) = try_promote_call_with_callback(node, source) {
|
||||
// The callback body is the sole child of the return value, so it
|
||||
// spans nearly the entire return statement. Force recursion to
|
||||
// guarantee the callback internals are addressable.
|
||||
promoted.force_recurse = true;
|
||||
return promoted;
|
||||
}
|
||||
group_from_sanitized(node, source)
|
||||
}
|
||||
|
||||
/// Classify a `jsx_self_closing_element` as a leaf chunk named after its tag.
|
||||
fn classify_jsx_self_closing_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let tag_name = extract_jsx_tag_name(node, source);
|
||||
make_kind_chunk(node, ChunkKind::Tag, tag_name, source, None)
|
||||
}
|
||||
|
||||
/// Unwrap nested `parenthesized_expression` wrappers to reach the meaningful
|
||||
/// inner expression.
|
||||
fn unwrap_parenthesized(mut node: Node<'_>) -> Node<'_> {
|
||||
while node.kind() == "parenthesized_expression" {
|
||||
let Some(inner) = named_children(node).into_iter().next() else {
|
||||
break;
|
||||
};
|
||||
node = inner;
|
||||
}
|
||||
node
|
||||
}
|
||||
|
||||
/// Extract the tag name from a `jsx_element` or `jsx_self_closing_element`.
|
||||
///
|
||||
/// The tag may be an identifier (`div`, `Link`), a member expression
|
||||
/// (`Foo.Bar`), or a nested identifier. We sanitize the full text so path
|
||||
/// segments remain valid identifiers.
|
||||
fn extract_jsx_tag_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let name_holder = match node.kind() {
|
||||
"jsx_element" => child_by_kind(node, &["jsx_opening_element"])?,
|
||||
"jsx_self_closing_element" => node,
|
||||
_ => return None,
|
||||
};
|
||||
let name_node = named_children(name_holder).into_iter().find(|child| {
|
||||
matches!(
|
||||
child.kind(),
|
||||
"identifier" | "member_expression" | "nested_identifier" | "jsx_namespace_name"
|
||||
)
|
||||
})?;
|
||||
sanitize_identifier(node_text(source, name_node.start_byte(), name_node.end_byte()))
|
||||
}
|
||||
|
||||
/// Classify `const`/`let`/`var` declarations, promoting arrow functions
|
||||
/// and class expressions to fn_/class_ chunks.
|
||||
fn classify_var_decl_js<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
// Inline promotion logic — look for single variable_declarator with fn/class
|
||||
// value.
|
||||
let declarators: Vec<Node<'t>> = named_children(node)
|
||||
.into_iter()
|
||||
.filter(|c| c.kind() == "variable_declarator")
|
||||
.collect();
|
||||
if declarators.len() == 1 {
|
||||
let decl = declarators[0];
|
||||
if let Some(value) = decl.child_by_field_name("value") {
|
||||
let name = extract_identifier(decl, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
match value.kind() {
|
||||
"arrow_function" | "function_expression" | "function" => {
|
||||
let recurse = recurse_body(value, ChunkContext::FunctionBody);
|
||||
return make_kind_chunk(node, ChunkKind::Function, Some(name), source, recurse);
|
||||
},
|
||||
"class" | "class_expression" => {
|
||||
let recurse = recurse_class(value);
|
||||
return make_container_chunk(node, ChunkKind::Class, Some(name), source, recurse);
|
||||
},
|
||||
_ => {},
|
||||
}
|
||||
}
|
||||
}
|
||||
// Not promoted — fall back to var_NAME or group.
|
||||
if let Some(name) = extract_single_declarator_name(node, source) {
|
||||
return make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None);
|
||||
}
|
||||
group_candidate(node, ChunkKind::Declarations, source)
|
||||
}
|
||||
|
||||
fn group_from_sanitized<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let sanitized = sanitize_node_kind(node.kind());
|
||||
let kind = ChunkKind::from_sanitized_kind(sanitized);
|
||||
let identifier = if kind == ChunkKind::Chunk {
|
||||
Some(sanitized.to_string())
|
||||
} else {
|
||||
None
|
||||
};
|
||||
make_candidate(node, kind, identifier, NameStyle::Group, None, None, source)
|
||||
}
|
||||
|
||||
/// Unwrap `export` / `export default` to classify the inner declaration.
|
||||
///
|
||||
/// Wrapper promotion handles declaration-like exports automatically.
|
||||
/// `export default …` remaps the promoted child to `default_export`, while
|
||||
/// re-exports and bare expression exports still fall through to `stmts`.
|
||||
fn classify_export_statement<'t>(
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
let header = normalized_header(source, node.start_byte(), node.end_byte());
|
||||
let is_default = header.starts_with("export default");
|
||||
|
||||
if let Some(candidate) =
|
||||
promote_wrapper_candidate(&JsTsClassifier, context, node, source, WrapperTransform {
|
||||
kind: is_default.then_some(ChunkKind::DefaultExport),
|
||||
name_style: is_default.then_some(NameStyle::Named),
|
||||
clear_identifier: is_default,
|
||||
..WrapperTransform::default()
|
||||
}) {
|
||||
return candidate;
|
||||
}
|
||||
|
||||
let Some(child) = first_wrapper_content_child(&JsTsClassifier, node) else {
|
||||
return if is_default {
|
||||
make_kind_chunk(node, ChunkKind::DefaultExport, None, source, None)
|
||||
} else {
|
||||
group_candidate(node, ChunkKind::Statements, source)
|
||||
};
|
||||
};
|
||||
|
||||
if is_default {
|
||||
return make_kind_chunk(node, ChunkKind::DefaultExport, None, source, None);
|
||||
}
|
||||
|
||||
match child.kind() {
|
||||
"lexical_declaration" | "variable_declaration" => {
|
||||
classify_with_defaults(&JsTsClassifier, context, child, source)
|
||||
},
|
||||
_ => group_candidate(child, ChunkKind::Statements, source),
|
||||
}
|
||||
}
|
||||
@@ -1,129 +0,0 @@
|
||||
//! Language-specific chunk classifier for Just.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct JustClassifier;
|
||||
|
||||
const JUST_FUNCTION_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"recipe_line",
|
||||
ChunkKind::Cmd,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"shebang",
|
||||
ChunkKind::Shebang,
|
||||
RuleStyle::Named,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const JUST_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: JUST_FUNCTION_RULES,
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &["source_file"],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
|
||||
fn first_named_child(node: Node<'_>) -> Option<Node<'_>> {
|
||||
named_children(node).into_iter().next()
|
||||
}
|
||||
|
||||
fn first_named_child_of_kind<'t>(node: Node<'t>, kind: &str) -> Option<Node<'t>> {
|
||||
named_children(node)
|
||||
.into_iter()
|
||||
.find(|child| child.kind() == kind)
|
||||
}
|
||||
|
||||
fn child_text<'a>(source: &'a str, node: Node<'_>) -> &'a str {
|
||||
node_text(source, node.start_byte(), node.end_byte())
|
||||
}
|
||||
|
||||
/// `set shell := ...` uses a dedicated `shell` token instead of a named
|
||||
/// identifier, so parse the assignment head text instead of relying on fields.
|
||||
fn extract_setting_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let header = child_text(source, node).lines().next()?.trim();
|
||||
let rest = header.strip_prefix("set ")?;
|
||||
let name = rest.split_once(":=")?.0.trim();
|
||||
sanitize_identifier(name)
|
||||
}
|
||||
|
||||
fn extract_alias_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
first_named_child(node).and_then(|child| sanitize_identifier(child_text(source, child)))
|
||||
}
|
||||
|
||||
fn extract_recipe_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let header = first_named_child_of_kind(node, "recipe_header")?;
|
||||
first_named_child(header).and_then(|child| sanitize_identifier(child_text(source, child)))
|
||||
}
|
||||
|
||||
fn classify_just_root_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"setting" => {
|
||||
let name = extract_setting_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
make_kind_chunk(node, ChunkKind::Setting, Some(name), source, None)
|
||||
},
|
||||
"alias" => {
|
||||
let name = extract_alias_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
make_kind_chunk(node, ChunkKind::Alias, Some(name), source, None)
|
||||
},
|
||||
"recipe" => {
|
||||
let name = extract_recipe_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Recipe,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["recipe_body"]),
|
||||
)
|
||||
},
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_just_body_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
// Just recipe bodies are line-oriented; tree-sitter exposes shell lines as
|
||||
// `recipe_line` leaves rather than a nested shell AST.
|
||||
"recipe_line" => group_candidate(node, ChunkKind::Cmd, source),
|
||||
"shebang" => make_kind_chunk(node, ChunkKind::Shebang, None, source, None),
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
impl LangClassifier for JustClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&JUST_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_just_root_node(node, source),
|
||||
ChunkContext::FunctionBody => classify_just_body_node(node, source),
|
||||
ChunkContext::ClassBody => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,341 +0,0 @@
|
||||
//! Language-specific chunk classifiers for Markdown and Handlebars.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
chunk_checksum,
|
||||
classify::{ClassifierTables, LangClassifier},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
types::ChunkNode,
|
||||
};
|
||||
use crate::language::SupportLang;
|
||||
|
||||
pub struct MarkupClassifier;
|
||||
|
||||
impl MarkupClassifier {
|
||||
fn classify_section<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let name = extract_markdown_heading(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Section,
|
||||
Some(name),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
))
|
||||
}
|
||||
|
||||
fn classify_block_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let name =
|
||||
extract_glimmer_block_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Block,
|
||||
Some(name),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
))
|
||||
}
|
||||
|
||||
fn classify_mustache_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let name =
|
||||
extract_glimmer_mustache_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
make_kind_chunk(node, ChunkKind::Mustache, Some(name), source, None)
|
||||
}
|
||||
|
||||
/// Classify HTML-like element nodes that appear inside handlebars blocks.
|
||||
fn classify_element<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"element" | "script_element" | "style_element" | "element_node" => {
|
||||
let name =
|
||||
extract_element_tag_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Tag,
|
||||
Some(name),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
)))
|
||||
},
|
||||
"text_node" => Some(group_candidate(node, ChunkKind::Text, source)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl LangClassifier for MarkupClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: super::classify::StructuralOverrides::EMPTY,
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
if !matches!(context, ChunkContext::Root | ChunkContext::ClassBody) {
|
||||
return None;
|
||||
}
|
||||
match node.kind() {
|
||||
"section" => Some(Self::classify_section(node, source)),
|
||||
"fenced_code_block" => Some(classify_fenced_code_block(node, source)),
|
||||
"html_block" => Some(classify_html_block(node, source)),
|
||||
"block_statement" => Some(Self::classify_block_statement(node, source)),
|
||||
"mustache_statement" => Some(Self::classify_mustache_statement(node, source)),
|
||||
"element" | "script_element" | "style_element" | "element_node" | "text_node" => {
|
||||
Self::classify_element(node, source)
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn post_process(
|
||||
&self,
|
||||
chunks: &mut Vec<ChunkNode>,
|
||||
_root_children: &mut Vec<String>,
|
||||
source: &str,
|
||||
) {
|
||||
add_markdown_table_row_chunks(chunks, source);
|
||||
}
|
||||
}
|
||||
|
||||
const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> {
|
||||
candidate.force_recurse = true;
|
||||
candidate
|
||||
}
|
||||
|
||||
fn classify_fenced_code_block<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let embedded_language = fenced_code_language(node, source);
|
||||
let identifier = embedded_language
|
||||
.map(embedded_selector_token)
|
||||
.map(str::to_string);
|
||||
let candidate =
|
||||
with_region_node(make_kind_chunk(node, ChunkKind::Code, identifier, source, None), None);
|
||||
match (child_by_kind(node, &["code_fence_content"]), embedded_language) {
|
||||
(Some(content_node), Some(language)) => {
|
||||
with_injected_subtree(candidate, language, content_node)
|
||||
},
|
||||
_ => candidate,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_html_block<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let candidate =
|
||||
with_region_node(make_kind_chunk(node, ChunkKind::Html, None, source, None), None);
|
||||
with_injected_subtree(candidate, SupportLang::Html, node)
|
||||
}
|
||||
|
||||
fn add_markdown_table_row_chunks(chunks: &mut Vec<ChunkNode>, source: &str) {
|
||||
let original_len = chunks.len();
|
||||
let mut additions = Vec::<(usize, Vec<ChunkNode>)>::new();
|
||||
for (index, chunk) in chunks.iter().enumerate().take(original_len) {
|
||||
if !chunk.children.is_empty() || matches!(chunk.kind, ChunkKind::Code | ChunkKind::Html) {
|
||||
continue;
|
||||
}
|
||||
let rows = markdown_table_rows_for_chunk(source, chunk);
|
||||
if rows.len() < 2
|
||||
|| !rows
|
||||
.iter()
|
||||
.any(|row| markdown_table_separator_row(row.text))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let nodes = rows
|
||||
.into_iter()
|
||||
.enumerate()
|
||||
.map(|(row_index, row)| {
|
||||
let identifier = (row_index + 1).to_string();
|
||||
let path = format!("{}.row_{}", chunk.path, identifier);
|
||||
let (indent, indent_char) = detect_indent(source, row.start_byte);
|
||||
let row_source = source.get(row.start_byte..row.end_byte).unwrap_or_default();
|
||||
ChunkNode {
|
||||
path,
|
||||
identifier: Some(identifier),
|
||||
kind: ChunkKind::Row,
|
||||
leaf: true,
|
||||
virtual_content: None,
|
||||
parent_path: Some(chunk.path.clone()),
|
||||
children: Vec::new(),
|
||||
signature: Some(row.text.trim().to_owned()),
|
||||
start_line: row.line,
|
||||
end_line: row.line,
|
||||
line_count: 1,
|
||||
start_byte: row.start_byte as u32,
|
||||
end_byte: row.end_byte as u32,
|
||||
checksum_start_byte: row.start_byte as u32,
|
||||
prologue_end_byte: None,
|
||||
epilogue_start_byte: None,
|
||||
checksum: chunk_checksum(row_source.as_bytes()),
|
||||
error: false,
|
||||
indent,
|
||||
indent_char,
|
||||
group: false,
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
additions.push((index, nodes));
|
||||
}
|
||||
|
||||
for (index, nodes) in additions {
|
||||
let child_paths = nodes.iter().map(|node| node.path.clone()).collect();
|
||||
chunks[index].leaf = false;
|
||||
chunks[index].children = child_paths;
|
||||
chunks.extend(nodes);
|
||||
}
|
||||
}
|
||||
|
||||
struct MarkdownTableRow<'a> {
|
||||
line: u32,
|
||||
start_byte: usize,
|
||||
end_byte: usize,
|
||||
text: &'a str,
|
||||
}
|
||||
|
||||
fn markdown_table_rows_for_chunk<'a>(
|
||||
source: &'a str,
|
||||
chunk: &ChunkNode,
|
||||
) -> Vec<MarkdownTableRow<'a>> {
|
||||
let line_offsets = source_line_offsets(source);
|
||||
let mut rows = Vec::new();
|
||||
let mut saw_non_empty = false;
|
||||
for line in chunk.start_line..=chunk.end_line {
|
||||
let Some((start_byte, end_byte)) = line_bounds(source, &line_offsets, line) else {
|
||||
continue;
|
||||
};
|
||||
let text = source
|
||||
.get(start_byte..end_byte)
|
||||
.unwrap_or_default()
|
||||
.trim_end_matches('\n');
|
||||
if text.trim().is_empty() {
|
||||
continue;
|
||||
}
|
||||
saw_non_empty = true;
|
||||
if !markdown_table_row(text) {
|
||||
return Vec::new();
|
||||
}
|
||||
rows.push(MarkdownTableRow { line, start_byte, end_byte, text });
|
||||
}
|
||||
if saw_non_empty { rows } else { Vec::new() }
|
||||
}
|
||||
|
||||
fn source_line_offsets(source: &str) -> Vec<usize> {
|
||||
let mut offsets = vec![0usize];
|
||||
for (index, ch) in source.char_indices() {
|
||||
if ch == '\n' {
|
||||
offsets.push(index + 1);
|
||||
}
|
||||
}
|
||||
offsets
|
||||
}
|
||||
|
||||
fn line_bounds(source: &str, offsets: &[usize], line: u32) -> Option<(usize, usize)> {
|
||||
if line == 0 {
|
||||
return None;
|
||||
}
|
||||
let start = *offsets.get((line - 1) as usize)?;
|
||||
let end = offsets.get(line as usize).copied().unwrap_or(source.len());
|
||||
Some((start, end))
|
||||
}
|
||||
|
||||
fn markdown_table_row(line: &str) -> bool {
|
||||
let trimmed = line.trim();
|
||||
trimmed.starts_with('|') && trimmed.ends_with('|') && trimmed.matches('|').count() >= 2
|
||||
}
|
||||
|
||||
fn markdown_table_separator_row(line: &str) -> bool {
|
||||
let trimmed = line.trim();
|
||||
trimmed.contains('-')
|
||||
&& trimmed
|
||||
.chars()
|
||||
.all(|ch| matches!(ch, '|' | '-' | ':' | ' ' | '\t'))
|
||||
}
|
||||
|
||||
/// Extract heading text from a Markdown `section` node's `atx_heading` or
|
||||
/// `setext_heading` child.
|
||||
fn extract_markdown_heading(node: Node<'_>, source: &str) -> Option<String> {
|
||||
named_children(node)
|
||||
.into_iter()
|
||||
.find(|child| child.kind() == "atx_heading" || child.kind() == "setext_heading")
|
||||
.and_then(|heading| {
|
||||
sanitize_identifier(node_text(source, heading.start_byte(), heading.end_byte()))
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract name from a Handlebars `block_statement` via its
|
||||
/// `block_statement_start` child.
|
||||
fn extract_glimmer_block_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["block_statement_start"]).and_then(|start| {
|
||||
start
|
||||
.child_by_field_name("path")
|
||||
.or_else(|| child_by_kind(start, &["identifier"]))
|
||||
.and_then(|name| {
|
||||
sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
fn fenced_code_language(node: Node<'_>, source: &str) -> Option<SupportLang> {
|
||||
child_by_kind(node, &["info_string"])
|
||||
.and_then(|info| child_by_kind(info, &["language"]))
|
||||
.and_then(|lang| {
|
||||
SupportLang::from_alias(node_text(source, lang.start_byte(), lang.end_byte()))
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract name from a Handlebars `mustache_statement`:
|
||||
/// tries `helper_invocation`'s helper field first, then direct
|
||||
/// `identifier`/`path_expression`.
|
||||
fn extract_glimmer_mustache_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let children = named_children(node);
|
||||
for child in children {
|
||||
if child.kind() == "helper_invocation"
|
||||
&& let Some(helper) = child
|
||||
.child_by_field_name("helper")
|
||||
.or_else(|| child_by_kind(child, &["identifier", "path_expression"]))
|
||||
{
|
||||
return sanitize_identifier(node_text(source, helper.start_byte(), helper.end_byte()));
|
||||
}
|
||||
if matches!(child.kind(), "identifier" | "path_expression") {
|
||||
return sanitize_identifier(node_text(source, child.start_byte(), child.end_byte()));
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
/// Extract tag name from an HTML-like element node.
|
||||
///
|
||||
/// Handles both standard HTML (`element` → `start_tag`/`self_closing_tag` →
|
||||
/// `tag_name`) and Handlebars element nodes (`element_node` →
|
||||
/// `element_node_start`/`element_node_void` → `tag_name`).
|
||||
fn extract_element_tag_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
// Handlebars element_node uses element_node_start / element_node_void
|
||||
if node.kind() == "element_node" {
|
||||
return named_children(node).into_iter().find_map(|child| {
|
||||
if child.kind() == "element_node_start" || child.kind() == "element_node_void" {
|
||||
child_by_kind(child, &["tag_name"]).and_then(|tag| {
|
||||
sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte()))
|
||||
})
|
||||
} else {
|
||||
None
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
// Standard HTML: element → start_tag / self_closing_tag → tag_name
|
||||
named_children(node).into_iter().find_map(|child| {
|
||||
if child.kind() == "start_tag" || child.kind() == "self_closing_tag" {
|
||||
child_by_kind(child, &["tag_name"]).and_then(|tag| {
|
||||
sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte()))
|
||||
})
|
||||
} else {
|
||||
None
|
||||
}
|
||||
})
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,228 +0,0 @@
|
||||
//! Language-specific chunk classifiers for Nix and HCL (Terraform).
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct NixHclClassifier;
|
||||
|
||||
const NIX_HCL_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &["body"],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
|
||||
/// Extract a structured name from an HCL `block` node.
|
||||
///
|
||||
/// Shape: `block_type label1 label2 … { body }` where labels are `string_lit`.
|
||||
/// Returns e.g. `resource_aws_instance_web` for `resource "aws_instance" "web"
|
||||
/// { … }`.
|
||||
fn extract_hcl_block_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let mut children = named_children(node).into_iter();
|
||||
let block_type = children.next()?;
|
||||
let mut parts =
|
||||
vec![node_text(source, block_type.start_byte(), block_type.end_byte()).to_string()];
|
||||
for child in children {
|
||||
if child.kind() == "string_lit" {
|
||||
let text = unquote_text(node_text(source, child.start_byte(), child.end_byte()));
|
||||
if !text.is_empty() {
|
||||
parts.push(text);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if child.kind() == "body" || child.kind() == "block_end" || child.kind() == "block_start" {
|
||||
continue;
|
||||
}
|
||||
let text = node_text(source, child.start_byte(), child.end_byte());
|
||||
if !text.is_empty() {
|
||||
parts.push(text.to_string());
|
||||
}
|
||||
}
|
||||
sanitize_identifier(parts.join("_").as_str())
|
||||
}
|
||||
|
||||
/// Extract the attrpath name from a Nix `binding` node.
|
||||
fn extract_nix_binding_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
node.child_by_field_name("attrpath").and_then(|attrpath| {
|
||||
sanitize_identifier(node_text(source, attrpath.start_byte(), attrpath.end_byte()))
|
||||
})
|
||||
}
|
||||
|
||||
fn recurse_nix_attrset(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["binding_set"])
|
||||
}
|
||||
|
||||
fn recurse_nix_binding_value(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
let expression = node.child_by_field_name("expression")?;
|
||||
if matches!(
|
||||
expression.kind(),
|
||||
"attrset_expression" | "let_attrset_expression" | "rec_attrset_expression"
|
||||
) {
|
||||
return recurse_nix_attrset(expression);
|
||||
}
|
||||
recurse_value_container(node)
|
||||
}
|
||||
|
||||
fn classify_nix_binding<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let name = extract_nix_binding_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
let expression = node.child_by_field_name("expression");
|
||||
if let Some(expression) = expression
|
||||
&& matches!(
|
||||
expression.kind(),
|
||||
"attrset_expression" | "let_attrset_expression" | "rec_attrset_expression"
|
||||
) {
|
||||
return make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Attr,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_nix_attrset(expression),
|
||||
);
|
||||
}
|
||||
make_kind_chunk(node, ChunkKind::Attr, Some(name), source, recurse_nix_binding_value(node))
|
||||
}
|
||||
|
||||
impl LangClassifier for NixHclClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&NIX_HCL_TABLES
|
||||
}
|
||||
|
||||
fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// Nix top-level attrsets should recurse into their binding_set so the file exposes
|
||||
// structural attr chunks instead of a single opaque attrset_expr leaf.
|
||||
"attrset_expression" | "let_attrset_expression" | "rec_attrset_expression" => {
|
||||
Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::Attrs,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse_nix_attrset(node),
|
||||
source,
|
||||
))
|
||||
},
|
||||
// Older tree-sitter-nix revisions used `attribute`; current grammars expose `binding`.
|
||||
"attribute" | "binding" => Some(classify_nix_binding(node, source)),
|
||||
// HCL top-level block, or diff hunk fallback
|
||||
"block" => {
|
||||
if let Some(name) = extract_hcl_block_name(node, source) {
|
||||
Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Block,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["body"]),
|
||||
))
|
||||
} else {
|
||||
Some(group_candidate(node, ChunkKind::Hunks, source))
|
||||
}
|
||||
},
|
||||
// Nix expressions
|
||||
"function_expression" | "let_expression" => Some(named_candidate(
|
||||
node,
|
||||
ChunkKind::Expression,
|
||||
source,
|
||||
recurse_value_container(node),
|
||||
)),
|
||||
// Nix inherit
|
||||
"inherit" => Some(group_candidate(node, ChunkKind::Imports, source)),
|
||||
// Variable/assignment declarations
|
||||
"variable_declaration" | "assignment" => {
|
||||
Some(group_candidate(node, ChunkKind::Declarations, source))
|
||||
},
|
||||
// HCL top-level block types
|
||||
"provider" | "resource" | "data" | "locals" | "variable" | "output" | "module" => {
|
||||
let kind = match node.kind() {
|
||||
"locals" => ChunkKind::BlockLocals,
|
||||
"variable" => ChunkKind::Variable,
|
||||
"module" => ChunkKind::Module,
|
||||
_ => ChunkKind::Block,
|
||||
};
|
||||
Some(make_candidate(
|
||||
node,
|
||||
kind,
|
||||
prefixed_name(sanitize_node_kind(node.kind()), node, source),
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["body"]),
|
||||
source,
|
||||
))
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// Nested HCL block — only promote if it has an identifiable block name
|
||||
"block" => extract_hcl_block_name(node, source).map(|name| {
|
||||
make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Block,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["body"]),
|
||||
)
|
||||
}),
|
||||
// Nested Nix attrset values recurse into their binding_set just like top-level ones.
|
||||
"attrset_expression" | "let_attrset_expression" | "rec_attrset_expression" => {
|
||||
Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::Attrs,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse_nix_attrset(node),
|
||||
source,
|
||||
))
|
||||
},
|
||||
// Nix binding_set is a transparent wrapper around individual bindings.
|
||||
// Without this, the binding_set becomes an opaque leaf chunk hiding
|
||||
// all bindings inside a single massive block.
|
||||
"binding_set" => Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::Attrs,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
None,
|
||||
Some(RecurseSpec { node, context: ChunkContext::ClassBody }),
|
||||
source,
|
||||
)),
|
||||
// Nix let_expression inside a class-body context — recurse into its
|
||||
// binding_set so individual bindings are addressable.
|
||||
"let_expression" => Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::Expression,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse_value_container(node),
|
||||
source,
|
||||
)),
|
||||
// Nested Nix binding
|
||||
"binding" => Some(classify_nix_binding(node, source)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// Nix control flow
|
||||
"if_expression" => Some(positional_candidate(node, ChunkKind::If, source)),
|
||||
"let_expression" => Some(positional_candidate(node, ChunkKind::Block, source)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,236 +0,0 @@
|
||||
//! OCaml-specific chunk classifier.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct OcamlClassifier;
|
||||
|
||||
impl LangClassifier for OcamlClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides::EMPTY,
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_ocaml_item(node, source),
|
||||
ChunkContext::ClassBody => classify_class(node, source),
|
||||
ChunkContext::FunctionBody => classify_function(node, source),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"method_definition" => Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
ocaml_named_text(node, source, &["method_name"]),
|
||||
source,
|
||||
ocaml_method_recurse(node),
|
||||
)),
|
||||
"method_specification" => Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
ocaml_named_text(node, source, &["method_name"]),
|
||||
source,
|
||||
None,
|
||||
)),
|
||||
"instance_variable_definition" => {
|
||||
Some(match ocaml_named_text(node, source, &["instance_variable_name"]) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Fields, source),
|
||||
})
|
||||
},
|
||||
_ => classify_ocaml_item(node, source),
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_function<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"function_expression" | "match_expression" => Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::Match,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
source,
|
||||
)),
|
||||
"match_case" => Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::Case,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
source,
|
||||
)),
|
||||
"let_expression" => Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::Let,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
source,
|
||||
)),
|
||||
_ => classify_ocaml_item(node, source),
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_ocaml_item<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"open_module" => group_candidate(node, ChunkKind::Imports, source),
|
||||
"module_definition" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Module,
|
||||
ocaml_named_text(node, source, &["module_name"]),
|
||||
source,
|
||||
ocaml_module_recurse(node),
|
||||
),
|
||||
"module_type_definition" => make_candidate(
|
||||
node,
|
||||
ChunkKind::Interface,
|
||||
format!("modtype_{}", ocaml_named_text(node, source, &["module_type_name"])?),
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
ocaml_module_type_recurse(node),
|
||||
source,
|
||||
),
|
||||
"class_definition" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Class,
|
||||
ocaml_named_text(node, source, &["class_name"]),
|
||||
source,
|
||||
ocaml_class_recurse(node),
|
||||
),
|
||||
"class_type_definition" => make_candidate(
|
||||
node,
|
||||
ChunkKind::Iface,
|
||||
format!("classtype_{}", ocaml_named_text(node, source, &["class_type_name"])?),
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
ocaml_class_type_recurse(node),
|
||||
source,
|
||||
),
|
||||
"type_definition" => make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Type,
|
||||
ocaml_named_text(node, source, &["type_constructor"]),
|
||||
source,
|
||||
None,
|
||||
),
|
||||
"exception_definition" => make_candidate(
|
||||
node,
|
||||
ChunkKind::Constructor,
|
||||
format!("exception_{}", ocaml_named_text(node, source, &["constructor_name"])?),
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
None,
|
||||
source,
|
||||
),
|
||||
"value_definition" => classify_ocaml_value_definition(node, source)?,
|
||||
"value_specification" => make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Val,
|
||||
ocaml_named_text(node, source, &["value_name"]),
|
||||
source,
|
||||
None,
|
||||
),
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_ocaml_value_definition<'t>(
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
let name = ocaml_named_text(node, source, &["value_name"])?;
|
||||
let recurse = ocaml_value_recurse(node);
|
||||
if ocaml_value_definition_is_function(node) {
|
||||
Some(make_kind_chunk(node, ChunkKind::Function, Some(name), source, recurse))
|
||||
} else {
|
||||
Some(make_kind_chunk(node, ChunkKind::Val, Some(name), source, recurse))
|
||||
}
|
||||
}
|
||||
|
||||
fn ocaml_named_text(node: Node<'_>, source: &str, kinds: &[&str]) -> Option<String> {
|
||||
find_named_text(node, source, kinds).and_then(sanitize_identifier)
|
||||
}
|
||||
|
||||
fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> {
|
||||
if kinds.iter().any(|kind| node.kind() == *kind) {
|
||||
return Some(node_text(source, node.start_byte(), node.end_byte()));
|
||||
}
|
||||
for child in named_children(node) {
|
||||
if let Some(text) = find_named_text(child, source, kinds) {
|
||||
return Some(text);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn ocaml_module_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["module_binding"]).and_then(|binding| {
|
||||
recurse_into(binding.node, ChunkContext::ClassBody, &["body"], &["structure", "signature"])
|
||||
})
|
||||
}
|
||||
|
||||
fn ocaml_module_type_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
recurse_into(node, ChunkContext::ClassBody, &["body"], &["signature"])
|
||||
}
|
||||
|
||||
fn ocaml_class_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["class_binding"]).and_then(|binding| {
|
||||
recurse_into(binding.node, ChunkContext::ClassBody, &["body"], &["object_expression"])
|
||||
})
|
||||
}
|
||||
|
||||
fn ocaml_class_type_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["class_type_binding"]).and_then(|binding| {
|
||||
recurse_into(binding.node, ChunkContext::ClassBody, &["body"], &["class_body_type"])
|
||||
})
|
||||
}
|
||||
|
||||
fn ocaml_method_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
recurse_into(node, ChunkContext::FunctionBody, &["body"], &[
|
||||
"function_expression",
|
||||
"match_expression",
|
||||
"let_expression",
|
||||
])
|
||||
}
|
||||
|
||||
fn ocaml_value_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["let_binding"]).and_then(|binding| {
|
||||
named_children(binding.node)
|
||||
.into_iter()
|
||||
.find(|child| {
|
||||
matches!(child.kind(), "function_expression" | "match_expression" | "let_expression")
|
||||
})
|
||||
.map(|child| RecurseSpec { node: child, context: ChunkContext::FunctionBody })
|
||||
})
|
||||
}
|
||||
|
||||
fn ocaml_value_definition_is_function(node: Node<'_>) -> bool {
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["let_binding"]).is_some_and(|binding| {
|
||||
named_children(binding.node)
|
||||
.into_iter()
|
||||
.any(|child| matches!(child.kind(), "parameter" | "function_expression"))
|
||||
})
|
||||
}
|
||||
@@ -1,137 +0,0 @@
|
||||
//! Language-specific chunk classifier for Perl.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct PerlClassifier;
|
||||
|
||||
const PERL_SHARED_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"use_statement",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"conditional_statement",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"for_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"loop_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
];
|
||||
|
||||
const PERL_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: PERL_SHARED_RULES,
|
||||
class: &[],
|
||||
function: PERL_SHARED_RULES,
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &["statement_list"],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
|
||||
impl LangClassifier for PerlClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&PERL_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root | ChunkContext::FunctionBody => classify_perl_node(node, source),
|
||||
ChunkContext::ClassBody => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_perl_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
let body_recurse = || recurse_into(node, ChunkContext::FunctionBody, &["body"], &["block"]);
|
||||
|
||||
Some(match node.kind() {
|
||||
"package_statement" => {
|
||||
make_kind_chunk(node, ChunkKind::Module, Some(perl_name(node, source)?), source, None)
|
||||
},
|
||||
"subroutine_declaration_statement" => make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
Some(perl_name(node, source)?),
|
||||
source,
|
||||
body_recurse(),
|
||||
),
|
||||
"expression_statement" => classify_perl_statement(node, source),
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_perl_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
if perl_declares_variable(node) {
|
||||
group_candidate(node, ChunkKind::Declarations, source)
|
||||
} else {
|
||||
group_candidate(node, ChunkKind::Statements, source)
|
||||
}
|
||||
}
|
||||
|
||||
fn perl_declares_variable(node: Node<'_>) -> bool {
|
||||
if node.kind() == "variable_declaration" {
|
||||
return true;
|
||||
}
|
||||
|
||||
if node.kind() == "assignment_expression"
|
||||
&& named_children(node)
|
||||
.into_iter()
|
||||
.any(|child| child.kind() == "variable_declaration")
|
||||
{
|
||||
return true;
|
||||
}
|
||||
|
||||
named_children(node).into_iter().any(perl_declares_variable)
|
||||
}
|
||||
|
||||
fn perl_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
find_named_text(node, source, &["bareword", "package", "varname"]).and_then(sanitize_identifier)
|
||||
}
|
||||
|
||||
fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> {
|
||||
if kinds.iter().any(|kind| node.kind() == *kind) {
|
||||
return Some(node_text(source, node.start_byte(), node.end_byte()));
|
||||
}
|
||||
|
||||
for child in named_children(node) {
|
||||
if let Some(text) = find_named_text(child, source, kinds) {
|
||||
return Some(text);
|
||||
}
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
@@ -1,290 +0,0 @@
|
||||
//! PowerShell-specific chunk classifier.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct PowershellClassifier;
|
||||
|
||||
static POWERSHELL_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[
|
||||
semantic_rule(
|
||||
"param_block",
|
||||
ChunkKind::Parameters,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"flow_control_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
],
|
||||
class: &[],
|
||||
function: &[
|
||||
semantic_rule(
|
||||
"class_method_parameter_list",
|
||||
ChunkKind::Parameters,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"param_block",
|
||||
ChunkKind::Parameters,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"flow_control_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[
|
||||
"function_name",
|
||||
"simple_name",
|
||||
"type_literal",
|
||||
"switch_condition",
|
||||
],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &[],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
|
||||
impl LangClassifier for PowershellClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&POWERSHELL_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_root_custom(node, source),
|
||||
ChunkContext::ClassBody => classify_class_custom(node, source),
|
||||
ChunkContext::FunctionBody => classify_function_custom(node, source),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"statement_list" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Body,
|
||||
None,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::Root)),
|
||||
),
|
||||
"class_statement" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Class,
|
||||
Some(powershell_name(node, source)?),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
),
|
||||
"function_statement" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
Some(powershell_name(node, source)?),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
),
|
||||
"pipeline" => classify_powershell_pipeline(node, source),
|
||||
"switch_statement" | "if_statement" | "foreach_statement" => {
|
||||
return classify_function_custom(node, source);
|
||||
},
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"class_property_definition" => match powershell_name(node, source) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Fields, source),
|
||||
},
|
||||
"class_method_definition" => classify_class_method(node, source)?,
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"script_block" => make_container_chunk(
|
||||
node,
|
||||
block_kind_for_parent(node),
|
||||
None,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
),
|
||||
"script_block_body" | "statement_block" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Block,
|
||||
None,
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_list"]),
|
||||
),
|
||||
"pipeline" => classify_powershell_pipeline(node, source),
|
||||
"if_statement" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::If,
|
||||
None,
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_block"]),
|
||||
),
|
||||
"foreach_statement" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Loop,
|
||||
None,
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_block"]),
|
||||
),
|
||||
"switch_statement" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Switch,
|
||||
None,
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["switch_body"]),
|
||||
),
|
||||
"switch_clauses" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Cases,
|
||||
None,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
),
|
||||
"switch_clause" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Case,
|
||||
None,
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_block"]),
|
||||
),
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_class_method<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
let name = powershell_name(node, source)?;
|
||||
let class_name = powershell_name(node.parent()?, source)?;
|
||||
let (kind, identifier) = if name == "new" || name == class_name {
|
||||
(ChunkKind::Constructor, None)
|
||||
} else {
|
||||
(ChunkKind::Function, Some(name))
|
||||
};
|
||||
|
||||
Some(make_container_chunk(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
))
|
||||
}
|
||||
|
||||
fn classify_powershell_pipeline<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
if let Some(command_name) = powershell_command_name(node, source)
|
||||
&& matches!(command_name.as_str(), "using" | "using-module" | "Import-Module")
|
||||
{
|
||||
return group_candidate(node, ChunkKind::Imports, source);
|
||||
}
|
||||
|
||||
if let Some((name, script_block)) = assigned_script_block(node, source) {
|
||||
return make_container_chunk_from(
|
||||
node,
|
||||
node,
|
||||
ChunkKind::Block,
|
||||
Some(name),
|
||||
source,
|
||||
Some(recurse_self(script_block, ChunkContext::FunctionBody)),
|
||||
);
|
||||
}
|
||||
|
||||
if child_by_kind(node, &["assignment_expression"]).is_some() {
|
||||
group_candidate(node, ChunkKind::Declarations, source)
|
||||
} else {
|
||||
group_candidate(node, ChunkKind::Statements, source)
|
||||
}
|
||||
}
|
||||
|
||||
fn assigned_script_block<'t>(node: Node<'t>, source: &str) -> Option<(String, Node<'t>)> {
|
||||
let assignment = child_by_kind(node, &["assignment_expression"])?;
|
||||
let lhs = child_by_kind(assignment, &["left_assignment_expression"])?;
|
||||
let name = sanitize_identifier(
|
||||
node_text(source, lhs.start_byte(), lhs.end_byte()).trim_start_matches('$'),
|
||||
)?;
|
||||
let script_block = named_children(assignment)
|
||||
.into_iter()
|
||||
.filter(|child| child.kind() != "left_assignment_expression")
|
||||
.find_map(find_script_block)?;
|
||||
Some((name, script_block))
|
||||
}
|
||||
|
||||
fn find_script_block(node: Node<'_>) -> Option<Node<'_>> {
|
||||
if node.kind() == "script_block" {
|
||||
return Some(node);
|
||||
}
|
||||
for child in named_children(node) {
|
||||
if let Some(script_block) = find_script_block(child) {
|
||||
return Some(script_block);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn block_kind_for_parent(node: Node<'_>) -> ChunkKind {
|
||||
match node.parent().map(|parent| parent.kind()) {
|
||||
Some("function_statement" | "class_method_definition") => ChunkKind::Body,
|
||||
_ => ChunkKind::Block,
|
||||
}
|
||||
}
|
||||
|
||||
fn powershell_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
find_named_text(node, source, &[
|
||||
"function_name",
|
||||
"simple_name",
|
||||
"member_name",
|
||||
"type_identifier",
|
||||
"variable",
|
||||
])
|
||||
.and_then(|text| sanitize_identifier(text.trim_start_matches('$')))
|
||||
}
|
||||
|
||||
fn powershell_command_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
find_named_text(node, source, &["command_name"]).and_then(sanitize_identifier)
|
||||
}
|
||||
|
||||
fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> {
|
||||
if kinds.iter().any(|kind| node.kind() == *kind) {
|
||||
return Some(node_text(source, node.start_byte(), node.end_byte()));
|
||||
}
|
||||
|
||||
for child in named_children(node) {
|
||||
if let Some(text) = find_named_text(child, source, kinds) {
|
||||
return Some(text);
|
||||
}
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
@@ -1,193 +0,0 @@
|
||||
//! Chunk classifier for Protocol Buffers.
|
||||
//!
|
||||
//! Mirror the grammar's declaration structure directly: the root owns headers,
|
||||
//! imports, options, messages, enums, and services; message bodies own fields,
|
||||
//! oneofs, and nested messages/enums; services own rpc declarations and service
|
||||
//! options; rpc blocks may contain rpc-scoped options.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct ProtoClassifier;
|
||||
|
||||
const PROTO_ROOT_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"syntax",
|
||||
ChunkKind::Headers,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"package",
|
||||
ChunkKind::Headers,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"import",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"option",
|
||||
ChunkKind::Options,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const PROTO_CLASS_RULES: &[super::classify::SemanticRule] = &[semantic_rule(
|
||||
"option",
|
||||
ChunkKind::Options,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
)];
|
||||
|
||||
const PROTO_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: PROTO_ROOT_RULES,
|
||||
class: PROTO_CLASS_RULES,
|
||||
function: &[],
|
||||
structural_overrides: super::classify::StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
impl LangClassifier for ProtoClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&PROTO_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_proto_root(node, source),
|
||||
ChunkContext::ClassBody => classify_proto_class(node, source),
|
||||
ChunkContext::FunctionBody => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_proto_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"message" => make_named_proto_chunk(
|
||||
node,
|
||||
ChunkKind::Type,
|
||||
format!("msg_{}", proto_name(node, source)?),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["message_body"]),
|
||||
),
|
||||
"enum" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Enum,
|
||||
proto_name(node, source),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["enum_body"]),
|
||||
),
|
||||
"service" => make_named_proto_chunk(
|
||||
node,
|
||||
ChunkKind::Interface,
|
||||
format!("service_{}", proto_name(node, source)?),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
),
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_proto_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"field" if is_proto_message_field(node) => {
|
||||
make_kind_chunk(node, ChunkKind::Field, proto_name(node, source), source, None)
|
||||
},
|
||||
"oneof" => make_named_proto_chunk(
|
||||
node,
|
||||
ChunkKind::Either,
|
||||
format!("oneof_{}", proto_name(node, source)?),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
),
|
||||
"oneof_field" => {
|
||||
make_kind_chunk(node, ChunkKind::Field, proto_name(node, source), source, None)
|
||||
},
|
||||
"message" => make_named_proto_chunk(
|
||||
node,
|
||||
ChunkKind::Type,
|
||||
format!("msg_{}", proto_name(node, source)?),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["message_body"]),
|
||||
),
|
||||
"enum" => make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Enum,
|
||||
proto_name(node, source),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["enum_body"]),
|
||||
),
|
||||
"enum_field" => {
|
||||
make_kind_chunk(node, ChunkKind::Variant, proto_name(node, source), source, None)
|
||||
},
|
||||
"rpc" => make_named_proto_chunk(
|
||||
node,
|
||||
ChunkKind::Proc,
|
||||
format!("rpc_{}", proto_name(node, source)?),
|
||||
source,
|
||||
proto_rpc_recurse(node),
|
||||
),
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn make_named_proto_chunk<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
identifier: impl Into<Option<String>>,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'t>>,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
make_candidate(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse,
|
||||
source,
|
||||
)
|
||||
}
|
||||
|
||||
fn is_proto_message_field(node: Node<'_>) -> bool {
|
||||
node
|
||||
.parent()
|
||||
.is_some_and(|parent| parent.kind() == "message_body")
|
||||
}
|
||||
|
||||
fn proto_rpc_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
let has_nested_option = named_children(node)
|
||||
.into_iter()
|
||||
.any(|child| child.kind() == "option");
|
||||
if has_nested_option {
|
||||
Some(recurse_self(node, ChunkContext::ClassBody))
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
fn proto_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["message_name", "enum_name", "service_name", "rpc_name", "identifier"])
|
||||
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
|
||||
}
|
||||
@@ -1,244 +0,0 @@
|
||||
//! Language-specific chunk classifiers for Python and Starlark.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, WrapperSignature,
|
||||
WrapperTransform, promote_wrapper_candidate, semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct PythonClassifier;
|
||||
|
||||
const ROOT_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"import_statement",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"import_from_statement",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"assignment",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"function_definition",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"class_definition",
|
||||
ChunkKind::Class,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"if_statement",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"for_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"while_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"try_statement",
|
||||
ChunkKind::Try,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"with_statement",
|
||||
ChunkKind::Block,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"expression_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"global_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const CLASS_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"expression_statement",
|
||||
ChunkKind::Fields,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"assignment",
|
||||
ChunkKind::Fields,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"type_alias_statement",
|
||||
ChunkKind::Type,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const FUNCTION_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"if_statement",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"for_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"while_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"try_statement",
|
||||
ChunkKind::Try,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"with_statement",
|
||||
ChunkKind::Block,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"elif_clause",
|
||||
ChunkKind::Elif,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"except_clause",
|
||||
ChunkKind::Except,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"match_statement",
|
||||
ChunkKind::Match,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const PYTHON_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: ROOT_RULES,
|
||||
class: CLASS_RULES,
|
||||
function: FUNCTION_RULES,
|
||||
structural_overrides: super::classify::StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
impl LangClassifier for PythonClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&PYTHON_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root | ChunkContext::ClassBody if node.kind() == "decorated_definition" => {
|
||||
promote_wrapper_candidate(self, context, node, source, WrapperTransform {
|
||||
signature: WrapperSignature::Wrapper,
|
||||
..WrapperTransform::default()
|
||||
})
|
||||
.or_else(|| Some(positional_candidate(node, ChunkKind::Block, source)))
|
||||
},
|
||||
ChunkContext::ClassBody if node.kind() == "function_definition" => {
|
||||
Some(classify_class_method(node, source))
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
let _ = source;
|
||||
Some(group_candidate(node, ChunkKind::Statements, source))
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_class_method<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
let kind = if name == "__init__" || name == "__new__" {
|
||||
ChunkKind::Constructor
|
||||
} else {
|
||||
ChunkKind::Function
|
||||
};
|
||||
let identifier = if kind == ChunkKind::Constructor {
|
||||
None
|
||||
} else {
|
||||
Some(name)
|
||||
};
|
||||
make_kind_chunk(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
source,
|
||||
resolve_recurse(node, ChunkContext::FunctionBody),
|
||||
)
|
||||
}
|
||||
@@ -1,173 +0,0 @@
|
||||
//! R-specific chunk classifier.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct RClassifier;
|
||||
|
||||
static R_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: super::classify::StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
impl LangClassifier for RClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&R_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_root_custom(node, source),
|
||||
ChunkContext::FunctionBody => classify_function_custom(node, source),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
// ── Imports ──
|
||||
"call" if is_import_call(node, source) => group_candidate(node, ChunkKind::Imports, source),
|
||||
"call" => group_candidate(node, ChunkKind::Statements, source),
|
||||
|
||||
// ── Function / value assignments ──
|
||||
"binary_operator" => classify_assignment(node, source, ChunkScope::Root)?,
|
||||
|
||||
// ── Control flow at script scope ──
|
||||
"if_statement" => control_candidate(node, ChunkKind::If, source, recurse_if(node)),
|
||||
"for_statement" | "while_statement" | "repeat_statement" => {
|
||||
control_candidate(node, ChunkKind::Loop, source, recurse_loop(node))
|
||||
},
|
||||
|
||||
// ── Bare expressions ──
|
||||
"identifier" | "subset" | "subset2" | "extract_operator" => {
|
||||
group_candidate(node, ChunkKind::Statements, source)
|
||||
},
|
||||
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
// ── Local assignments ──
|
||||
"binary_operator" => classify_assignment(node, source, ChunkScope::Function)?,
|
||||
|
||||
// ── Control flow ──
|
||||
"if_statement" => control_candidate(node, ChunkKind::If, source, recurse_if(node)),
|
||||
"for_statement" | "while_statement" | "repeat_statement" => {
|
||||
control_candidate(node, ChunkKind::Loop, source, recurse_loop(node))
|
||||
},
|
||||
|
||||
// ── Calls / bare expressions ──
|
||||
"call" | "identifier" | "subset" | "subset2" | "extract_operator" | "break" | "next"
|
||||
| "return" => group_candidate(node, ChunkKind::Statements, source),
|
||||
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
enum ChunkScope {
|
||||
Root,
|
||||
Function,
|
||||
}
|
||||
|
||||
fn classify_assignment<'t>(
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
scope: ChunkScope,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
let (lhs, rhs) = assignment_sides(node, source)?;
|
||||
|
||||
if rhs.kind() == "function_definition" {
|
||||
let name = simple_lhs_name(lhs, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
return Some(make_kind_chunk_from(
|
||||
node,
|
||||
rhs,
|
||||
ChunkKind::Function,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_body(rhs, ChunkContext::FunctionBody),
|
||||
));
|
||||
}
|
||||
|
||||
match (scope, simple_lhs_name(lhs, source)) {
|
||||
(ChunkScope::Root, Some(name)) => {
|
||||
Some(make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None))
|
||||
},
|
||||
(ChunkScope::Function, Some(name)) if spans_multiple_lines(node) => {
|
||||
Some(make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None))
|
||||
},
|
||||
_ => Some(group_candidate(
|
||||
node,
|
||||
match scope {
|
||||
ChunkScope::Root => ChunkKind::Declarations,
|
||||
ChunkScope::Function => ChunkKind::Statements,
|
||||
},
|
||||
source,
|
||||
)),
|
||||
}
|
||||
}
|
||||
|
||||
fn assignment_sides<'t>(node: Node<'t>, source: &str) -> Option<(Node<'t>, Node<'t>)> {
|
||||
if node.kind() != "binary_operator" {
|
||||
return None;
|
||||
}
|
||||
|
||||
let operator = node.child_by_field_name("operator")?;
|
||||
let operator_text = node_text(source, operator.start_byte(), operator.end_byte());
|
||||
if !matches!(operator_text, "<-" | "<<-" | "=") {
|
||||
return None;
|
||||
}
|
||||
|
||||
Some((node.child_by_field_name("lhs")?, node.child_by_field_name("rhs")?))
|
||||
}
|
||||
|
||||
fn simple_lhs_name(lhs: Node<'_>, source: &str) -> Option<String> {
|
||||
(lhs.kind() == "identifier")
|
||||
.then(|| extract_identifier(lhs, source))
|
||||
.flatten()
|
||||
}
|
||||
|
||||
fn is_import_call(node: Node<'_>, source: &str) -> bool {
|
||||
matches!(
|
||||
extract_identifier(node, source).as_deref(),
|
||||
Some("library" | "require" | "requireNamespace" | "source")
|
||||
)
|
||||
}
|
||||
|
||||
fn recurse_if(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
recurse_into(node, ChunkContext::FunctionBody, &["consequence", "alternative"], &[
|
||||
"braced_expression",
|
||||
])
|
||||
}
|
||||
|
||||
fn recurse_loop(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
recurse_into(node, ChunkContext::FunctionBody, &["body"], &["braced_expression"])
|
||||
}
|
||||
|
||||
fn control_candidate<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'t>>,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
make_candidate(node, kind, None, NameStyle::Named, None, recurse, source)
|
||||
}
|
||||
|
||||
fn spans_multiple_lines(node: Node<'_>) -> bool {
|
||||
node.start_position().row != node.end_position().row
|
||||
}
|
||||
@@ -1,253 +0,0 @@
|
||||
//! Language-specific chunk classifiers for Ruby and Lua.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct RubyLuaClassifier;
|
||||
|
||||
const RUBY_LUA_ROOT_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"function_definition",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"method",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"singleton_method",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"class",
|
||||
ChunkKind::Class,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"module",
|
||||
ChunkKind::Module,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"if_statement",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"unless",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"while_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"for_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"assignment",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"expression_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"function_call",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const RUBY_LUA_CLASS_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"class",
|
||||
ChunkKind::Class,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"module",
|
||||
ChunkKind::Module,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"assignment",
|
||||
ChunkKind::Fields,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"call",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"command",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"identifier",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const RUBY_LUA_FUNCTION_RULES: &[super::classify::SemanticRule] = &[
|
||||
semantic_rule(
|
||||
"if_statement",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"unless",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"case_statement",
|
||||
ChunkKind::Switch,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"case_match",
|
||||
ChunkKind::Switch,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"while_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"for_statement",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"assignment",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const RUBY_LUA_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: RUBY_LUA_ROOT_RULES,
|
||||
class: RUBY_LUA_CLASS_RULES,
|
||||
function: RUBY_LUA_FUNCTION_RULES,
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &[],
|
||||
preserved_root_wrappers: &["module"],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
|
||||
impl LangClassifier for RubyLuaClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&RUBY_LUA_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match (context, node.kind()) {
|
||||
(ChunkContext::Root, "command" | "call") => {
|
||||
let target = extract_identifier(node, source);
|
||||
Some(match target.as_deref() {
|
||||
Some("require" | "require_relative" | "load" | "autoload") => {
|
||||
group_candidate(node, ChunkKind::Imports, source)
|
||||
},
|
||||
_ => group_candidate(node, ChunkKind::Statements, source),
|
||||
})
|
||||
},
|
||||
(ChunkContext::ClassBody, "method" | "singleton_method") => {
|
||||
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
let kind = if name == "initialize" {
|
||||
ChunkKind::Constructor
|
||||
} else {
|
||||
ChunkKind::Function
|
||||
};
|
||||
let identifier = (kind != ChunkKind::Constructor).then_some(name);
|
||||
Some(make_kind_chunk(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
source,
|
||||
resolve_recurse(node, ChunkContext::FunctionBody),
|
||||
))
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,404 +0,0 @@
|
||||
//! Rust-specific chunk classifier.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct RustClassifier;
|
||||
|
||||
const ROOT_RULES: &[super::classify::SemanticRule] = &[
|
||||
// ── Imports ──
|
||||
semantic_rule(
|
||||
"use_declaration",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"extern_crate_declaration",
|
||||
ChunkKind::Imports,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
// ── Functions ──
|
||||
semantic_rule(
|
||||
"function_item",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"function_definition",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
// ── Containers ──
|
||||
semantic_rule(
|
||||
"struct_item",
|
||||
ChunkKind::Struct,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"enum_item",
|
||||
ChunkKind::Enum,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"trait_item",
|
||||
ChunkKind::Trait,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"mod_item",
|
||||
ChunkKind::Module,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"foreign_block",
|
||||
ChunkKind::Module,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
// ── Types ──
|
||||
semantic_rule(
|
||||
"type_item",
|
||||
ChunkKind::Type,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::ClassBody),
|
||||
),
|
||||
// ── Macros ──
|
||||
semantic_rule(
|
||||
"macro_definition",
|
||||
ChunkKind::Macro,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"macro_rule",
|
||||
ChunkKind::Macro,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
// ── Statics / consts ──
|
||||
semantic_rule(
|
||||
"static_item",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"const_item",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
// ── Attributes ──
|
||||
semantic_rule(
|
||||
"inner_attribute_item",
|
||||
ChunkKind::Attrs,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
// ── Expression statements ──
|
||||
semantic_rule(
|
||||
"expression_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const CLASS_RULES: &[super::classify::SemanticRule] = &[
|
||||
// ── Functions (methods in impl/trait) ──
|
||||
semantic_rule(
|
||||
"function_item",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
semantic_rule(
|
||||
"function_definition",
|
||||
ChunkKind::Function,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::Auto(ChunkContext::FunctionBody),
|
||||
),
|
||||
// ── Types ──
|
||||
semantic_rule(
|
||||
"type_item",
|
||||
ChunkKind::Type,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"type_alias",
|
||||
ChunkKind::Type,
|
||||
RuleStyle::Named,
|
||||
NamingMode::AutoIdentifier,
|
||||
RecurseMode::None,
|
||||
),
|
||||
// ── Consts / macros in class body ──
|
||||
semantic_rule(
|
||||
"const_item",
|
||||
ChunkKind::Fields,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"macro_invocation",
|
||||
ChunkKind::Fields,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const FUNCTION_RULES: &[super::classify::SemanticRule] = &[
|
||||
// ── Control flow ──
|
||||
semantic_rule(
|
||||
"match_expression",
|
||||
ChunkKind::Match,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"loop_expression",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"while_expression",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"for_expression",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
// ── Expression statements ──
|
||||
semantic_rule(
|
||||
"expression_statement",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
];
|
||||
|
||||
const RUST_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: ROOT_RULES,
|
||||
class: CLASS_RULES,
|
||||
function: FUNCTION_RULES,
|
||||
structural_overrides: StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
impl LangClassifier for RustClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&RUST_TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_root_custom(node, source),
|
||||
ChunkContext::ClassBody => classify_class_custom(node, source),
|
||||
ChunkContext::FunctionBody => classify_function_custom(node, source),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// ── Impl blocks (custom name extraction) ──
|
||||
"impl_item" => {
|
||||
let name = extract_impl_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Impl,
|
||||
Some(name),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &["body"], &["declaration_list"]),
|
||||
))
|
||||
},
|
||||
|
||||
// ── Variables (conditional auto-id vs group) ──
|
||||
"let_declaration" => Some(match extract_identifier(node, source) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Declarations, source),
|
||||
}),
|
||||
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
// ── Fields (conditional auto-id vs group) ──
|
||||
"field_declaration" => Some(match extract_identifier(node, source) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Fields, source),
|
||||
}),
|
||||
|
||||
// ── Enum variants (conditional auto-id vs group) ──
|
||||
"enum_variant" => Some(match extract_identifier(node, source) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Variants, source),
|
||||
}),
|
||||
|
||||
// ── Attributes (explicitly return None — absorbed by framework) ──
|
||||
"attribute_item" => None,
|
||||
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
|
||||
match node.kind() {
|
||||
// ── Control flow with recurse ──
|
||||
"if_expression" => Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::If,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
None,
|
||||
fn_recurse(),
|
||||
source,
|
||||
)),
|
||||
|
||||
// ── Blocks ──
|
||||
"unsafe_block" | "async_block" | "const_block" | "block_expression" => Some(make_candidate(
|
||||
node,
|
||||
ChunkKind::Block,
|
||||
None,
|
||||
NameStyle::Named,
|
||||
None,
|
||||
fn_recurse(),
|
||||
source,
|
||||
)),
|
||||
|
||||
// ── Variables (conditional line span) ──
|
||||
"let_declaration" => {
|
||||
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
|
||||
Some(if span > 1 {
|
||||
match extract_identifier(node, source) {
|
||||
Some(name) => make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None),
|
||||
None => group_candidate(node, ChunkKind::Let, source),
|
||||
}
|
||||
} else {
|
||||
group_candidate(node, ChunkKind::Let, source)
|
||||
})
|
||||
},
|
||||
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract the name for an `impl` block.
|
||||
///
|
||||
/// - Plain impl: `impl Foo` → `"Foo"`
|
||||
/// - Trait impl: `impl Trait for Foo` → `"Trait_for_Foo"`
|
||||
/// - Scoped trait: `impl fmt::Display for Foo` → `"Display_for_Foo"`
|
||||
fn extract_impl_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
// Collect ALL children (including anonymous keywords like `for`).
|
||||
let all_children: Vec<Node<'_>> = (0..node.child_count())
|
||||
.filter_map(|i| node.child(i))
|
||||
.collect();
|
||||
|
||||
// Find the `for` keyword position.
|
||||
let for_index = all_children
|
||||
.iter()
|
||||
.position(|c| node_text(source, c.start_byte(), c.end_byte()) == "for");
|
||||
|
||||
if let Some(fi) = for_index {
|
||||
// Trait impl: trait name before `for`, type name after `for`.
|
||||
let trait_node = all_children[..fi].iter().rev().find(|c| {
|
||||
matches!(c.kind(), "type_identifier" | "scoped_type_identifier" | "generic_type")
|
||||
});
|
||||
let type_node = all_children[fi + 1..].iter().find(|c| {
|
||||
matches!(c.kind(), "type_identifier" | "scoped_type_identifier" | "generic_type")
|
||||
});
|
||||
|
||||
if let (Some(tn), Some(ty)) = (trait_node, type_node) {
|
||||
let trait_name = extract_last_type_identifier(*tn, source)
|
||||
.or_else(|| sanitize_identifier(node_text(source, tn.start_byte(), tn.end_byte())))?;
|
||||
let type_name = extract_last_type_identifier(*ty, source)
|
||||
.or_else(|| sanitize_identifier(node_text(source, ty.start_byte(), ty.end_byte())))?;
|
||||
return Some(format!("{trait_name}_for_{type_name}"));
|
||||
}
|
||||
}
|
||||
|
||||
// Plain impl: take the last type_identifier.
|
||||
let type_ids: Vec<Node<'_>> = named_children(node)
|
||||
.into_iter()
|
||||
.filter(|c| c.kind() == "type_identifier")
|
||||
.collect();
|
||||
type_ids
|
||||
.last()
|
||||
.and_then(|n| sanitize_identifier(node_text(source, n.start_byte(), n.end_byte())))
|
||||
}
|
||||
|
||||
/// Recursively find the innermost `type_identifier` from a type node.
|
||||
///
|
||||
/// Handles scoped types like `fmt::Display` by traversing into
|
||||
/// `scoped_type_identifier` and `generic_type` children.
|
||||
fn extract_last_type_identifier(node: Node<'_>, source: &str) -> Option<String> {
|
||||
if node.kind() == "type_identifier" {
|
||||
return sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()));
|
||||
}
|
||||
|
||||
let mut result = None;
|
||||
for child in named_children(node) {
|
||||
if child.kind() == "type_identifier" {
|
||||
result = sanitize_identifier(node_text(source, child.start_byte(), child.end_byte()));
|
||||
} else if matches!(child.kind(), "scoped_type_identifier" | "generic_type")
|
||||
&& let Some(inner) = extract_last_type_identifier(child, source)
|
||||
{
|
||||
result = Some(inner);
|
||||
}
|
||||
}
|
||||
result
|
||||
}
|
||||
@@ -1,282 +0,0 @@
|
||||
//! SQL-specific chunk classifier.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
|
||||
pub struct SqlClassifier;
|
||||
|
||||
impl LangClassifier for SqlClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &["empty_statement", "dollar_quote", "keyword_from"],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &[],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_sql_root(node, source),
|
||||
ChunkContext::ClassBody => classify_sql_class(node, source),
|
||||
ChunkContext::FunctionBody => classify_sql_function(node, source),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_sql_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
if node.kind() == "statement" {
|
||||
return classify_sql_statement_root(node, source);
|
||||
}
|
||||
|
||||
classify_sql_root_node(node, node, source).or_else(|| classify_sql_query_node(node, source))
|
||||
}
|
||||
|
||||
fn classify_sql_statement_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
let children = named_children(node);
|
||||
if children.len() == 1 {
|
||||
return classify_sql_root_node(node, children[0], source)
|
||||
.or_else(|| classify_sql_query_node(children[0], source));
|
||||
}
|
||||
|
||||
if children.iter().any(|child| is_sql_query_kind(child.kind())) {
|
||||
return Some(make_named_sql_chunk(
|
||||
node,
|
||||
ChunkKind::Query,
|
||||
None,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
));
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
|
||||
fn classify_sql_root_node<'t>(
|
||||
range_node: Node<'t>,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"create_schema" => make_kind_chunk_from(
|
||||
range_node,
|
||||
node,
|
||||
ChunkKind::Schema,
|
||||
extract_sql_identifier(node, source),
|
||||
source,
|
||||
None,
|
||||
),
|
||||
"create_table" => make_container_chunk_from(
|
||||
range_node,
|
||||
node,
|
||||
ChunkKind::Table,
|
||||
extract_sql_object_name(node, source),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::ClassBody, &[], &["column_definitions"]),
|
||||
),
|
||||
"create_view" => make_named_sql_chunk_from(
|
||||
range_node,
|
||||
node,
|
||||
ChunkKind::Query,
|
||||
format!(
|
||||
"view_{}",
|
||||
extract_sql_object_name(node, source).unwrap_or_else(|| "anonymous".to_string())
|
||||
),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["create_query"]),
|
||||
),
|
||||
"create_materialized_view" => make_named_sql_chunk_from(
|
||||
range_node,
|
||||
node,
|
||||
ChunkKind::Query,
|
||||
format!(
|
||||
"matview_{}",
|
||||
extract_sql_object_name(node, source).unwrap_or_else(|| "anonymous".to_string())
|
||||
),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["create_query"]),
|
||||
),
|
||||
"create_function" => make_container_chunk_from(
|
||||
range_node,
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
extract_sql_object_name(node, source),
|
||||
source,
|
||||
recurse_sql_function_query(node),
|
||||
),
|
||||
"create_trigger" => make_named_sql_chunk_from(
|
||||
range_node,
|
||||
node,
|
||||
ChunkKind::Function,
|
||||
format!(
|
||||
"trigger_{}",
|
||||
extract_sql_object_name(node, source).unwrap_or_else(|| "anonymous".to_string())
|
||||
),
|
||||
source,
|
||||
None,
|
||||
),
|
||||
"create_index" => make_named_sql_chunk_from(
|
||||
range_node,
|
||||
node,
|
||||
ChunkKind::Key,
|
||||
format!(
|
||||
"index_{}",
|
||||
extract_sql_identifier(node, source).unwrap_or_else(|| "anonymous".to_string())
|
||||
),
|
||||
source,
|
||||
None,
|
||||
),
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_sql_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"column_definition" => {
|
||||
make_kind_chunk(node, ChunkKind::Field, extract_sql_identifier(node, source), source, None)
|
||||
},
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn classify_sql_function<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
if node.kind() == "statement" {
|
||||
return Some(make_named_sql_chunk(
|
||||
node,
|
||||
ChunkKind::Query,
|
||||
None,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
));
|
||||
}
|
||||
|
||||
classify_sql_query_node(node, source)
|
||||
}
|
||||
|
||||
fn classify_sql_query_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
Some(match node.kind() {
|
||||
"insert" => group_candidate(node, ChunkKind::Statements, source),
|
||||
"keyword_with" => group_candidate(node, ChunkKind::With, source),
|
||||
"cte" => make_named_sql_chunk(
|
||||
node,
|
||||
ChunkKind::With,
|
||||
format!(
|
||||
"cte_{}",
|
||||
extract_sql_identifier(node, source).unwrap_or_else(|| "anonymous".to_string())
|
||||
),
|
||||
source,
|
||||
recurse_into(node, ChunkContext::FunctionBody, &[], &["statement"]),
|
||||
),
|
||||
"select" => positional_candidate(node, ChunkKind::Select, source),
|
||||
"from" => make_named_sql_chunk(
|
||||
node,
|
||||
ChunkKind::Query,
|
||||
"from".to_string(),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::FunctionBody)),
|
||||
),
|
||||
"relation" => group_candidate(node, ChunkKind::Relations, source),
|
||||
"join" => positional_candidate(node, ChunkKind::Join, source),
|
||||
"where" => positional_candidate(node, ChunkKind::Where, source),
|
||||
"group_by" => positional_candidate(node, ChunkKind::GroupBy, source),
|
||||
"order_by" => positional_candidate(node, ChunkKind::OrderBy, source),
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
fn make_named_sql_chunk<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
identifier: impl Into<Option<String>>,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'t>>,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
make_candidate(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse,
|
||||
source,
|
||||
)
|
||||
}
|
||||
|
||||
fn make_named_sql_chunk_from<'t>(
|
||||
range_node: Node<'t>,
|
||||
signature_node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
identifier: impl Into<Option<String>>,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'t>>,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
make_candidate(
|
||||
range_node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(signature_node, source),
|
||||
recurse,
|
||||
source,
|
||||
)
|
||||
}
|
||||
|
||||
fn recurse_sql_function_query(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
let body = child_by_kind(node, &["function_body"])?;
|
||||
recurse_into(body, ChunkContext::FunctionBody, &[], &["statement"])
|
||||
}
|
||||
|
||||
fn is_sql_query_kind(kind: &str) -> bool {
|
||||
matches!(
|
||||
kind,
|
||||
"insert"
|
||||
| "keyword_with"
|
||||
| "cte"
|
||||
| "select"
|
||||
| "from"
|
||||
| "where"
|
||||
| "group_by"
|
||||
| "order_by"
|
||||
| "join"
|
||||
)
|
||||
}
|
||||
|
||||
fn extract_sql_identifier(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["identifier"])
|
||||
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
|
||||
}
|
||||
|
||||
fn extract_sql_object_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["object_reference"]).and_then(|name| last_identifier(name, source))
|
||||
}
|
||||
|
||||
fn last_identifier(node: Node<'_>, source: &str) -> Option<String> {
|
||||
if node.kind() == "identifier" {
|
||||
return sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()));
|
||||
}
|
||||
|
||||
for child in named_children(node).into_iter().rev() {
|
||||
if let Some(identifier) = last_identifier(child, source) {
|
||||
return Some(identifier);
|
||||
}
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
@@ -1,299 +0,0 @@
|
||||
//! Language-specific chunk classifier for Svelte.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
use crate::language::SupportLang;
|
||||
|
||||
pub struct SvelteClassifier;
|
||||
|
||||
impl LangClassifier for SvelteClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &["document"],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
let include_plain_elements = matches!(context, ChunkContext::Root);
|
||||
classify_svelte_node(node, source, include_plain_elements)
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_svelte_node<'t>(
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
include_plain_elements: bool,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"script_element" => Some(classify_script_element(node, source)),
|
||||
"style_element" => Some(classify_style_element(node, source)),
|
||||
"snippet_statement" => Some(classify_snippet_statement(node, source)),
|
||||
"if_statement" => Some(classify_if_statement(node, source)),
|
||||
"else_if_statement" => Some(classify_else_if_statement(node, source)),
|
||||
"else_statement" => Some(classify_else_statement(node, source)),
|
||||
"each_statement" => Some(classify_each_statement(node, source)),
|
||||
"await_statement" => Some(classify_await_statement(node, source)),
|
||||
"then_statement" => Some(classify_then_statement(node, source)),
|
||||
"catch_statement" => Some(classify_catch_statement(node, source)),
|
||||
"render_expr" => Some(classify_render_expr(node, source)),
|
||||
"html_interpolation" => Some(group_candidate(node, ChunkKind::Html, source)),
|
||||
"interpolation" => Some(group_candidate(node, ChunkKind::Interpolation, source)),
|
||||
"expression" => Some(group_candidate(node, ChunkKind::Expression, source)),
|
||||
"element" if include_plain_elements || element_has_structure(node) => {
|
||||
classify_element(node, source)
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let kind = if has_attribute(node, "module", source)
|
||||
|| attribute_value(node, "context", source).as_deref() == Some("module")
|
||||
{
|
||||
ChunkKind::ScriptModule
|
||||
} else {
|
||||
ChunkKind::Script
|
||||
};
|
||||
classify_raw_text_block(node, kind, source, SupportLang::TypeScript)
|
||||
}
|
||||
|
||||
fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let kind = if has_attribute(node, "scoped", source) {
|
||||
ChunkKind::StyleScoped
|
||||
} else {
|
||||
ChunkKind::Style
|
||||
};
|
||||
classify_raw_text_block(node, kind, source, SupportLang::Css)
|
||||
}
|
||||
|
||||
fn classify_snippet_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let identifier = child_by_kind(node, &["snippet_start_expr"])
|
||||
.and_then(|start| child_by_kind(start, &["snippet_name"]))
|
||||
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())));
|
||||
force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Snippet,
|
||||
identifier,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
))
|
||||
}
|
||||
|
||||
fn classify_if_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let identifier = block_expr_identifier(node, source, "if_start_expr", &["raw_text_expr"]);
|
||||
force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::If,
|
||||
identifier,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
))
|
||||
}
|
||||
|
||||
fn classify_else_if_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let identifier = block_expr_identifier(node, source, "else_if_expr", &["raw_text_expr"])
|
||||
.map_or_else(|| "if".to_string(), |expr| format!("if_{expr}"));
|
||||
make_named_container_chunk(node, ChunkKind::Else, identifier, source)
|
||||
}
|
||||
|
||||
fn classify_else_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Else,
|
||||
None,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
))
|
||||
}
|
||||
|
||||
fn classify_each_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let expr = block_expr_identifier(node, source, "each_start_expr", &["raw_text_each"]);
|
||||
let id = expr.map_or_else(|| "each".to_string(), |expr| format!("each_{expr}"));
|
||||
make_named_container_chunk(node, ChunkKind::Loop, id, source)
|
||||
}
|
||||
|
||||
fn classify_await_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let expr = block_expr_identifier(node, source, "await_start_expr", &["raw_text_expr"]);
|
||||
let id = expr.map_or_else(|| "await".to_string(), |expr| format!("await_{expr}"));
|
||||
make_named_container_chunk(node, ChunkKind::With, id, source)
|
||||
}
|
||||
|
||||
fn classify_then_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let expr = block_expr_identifier(node, source, "then_expr", &["raw_text_expr"]);
|
||||
let id = expr.map_or_else(|| "then".to_string(), |expr| format!("then_{expr}"));
|
||||
make_named_container_chunk(node, ChunkKind::After, id, source)
|
||||
}
|
||||
|
||||
fn classify_catch_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let identifier = block_expr_identifier(node, source, "catch_expr", &["raw_text_expr"]);
|
||||
force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Catch,
|
||||
identifier,
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
))
|
||||
}
|
||||
|
||||
fn classify_render_expr<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let identifier = child_by_kind(node, &["snippet_name"])
|
||||
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())));
|
||||
make_kind_chunk(node, ChunkKind::Render, identifier, source, None)
|
||||
}
|
||||
|
||||
fn classify_element<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
let tag_name = extract_markup_tag_name(node, source)?;
|
||||
Some(force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Tag,
|
||||
Some(tag_name),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
)))
|
||||
}
|
||||
|
||||
fn make_named_container_chunk<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
identifier: impl Into<Option<String>>,
|
||||
source: &str,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
force_container(make_candidate(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
source,
|
||||
))
|
||||
}
|
||||
|
||||
const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> {
|
||||
candidate.force_recurse = true;
|
||||
candidate
|
||||
}
|
||||
|
||||
fn classify_raw_text_block<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
source: &str,
|
||||
default_language: SupportLang,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
let Some(content_node) = child_by_kind(node, &["raw_text"]) else {
|
||||
return positional_candidate(node, kind, source);
|
||||
};
|
||||
let candidate = with_region_node(positional_candidate(node, kind, source), Some(content_node));
|
||||
match resolve_embedded_language(node, source, default_language) {
|
||||
Some(language) => with_injected_subtree(candidate, language, content_node),
|
||||
None => candidate,
|
||||
}
|
||||
}
|
||||
|
||||
fn block_expr_identifier(
|
||||
node: Node<'_>,
|
||||
source: &str,
|
||||
header_kind: &str,
|
||||
expr_kinds: &[&str],
|
||||
) -> Option<String> {
|
||||
child_by_kind(node, &[header_kind])
|
||||
.and_then(|header| child_by_kind(header, expr_kinds))
|
||||
.and_then(|expr| sanitize_identifier(node_text(source, expr.start_byte(), expr.end_byte())))
|
||||
}
|
||||
|
||||
fn element_has_structure(node: Node<'_>) -> bool {
|
||||
named_children(node).into_iter().any(|child| {
|
||||
matches!(
|
||||
child.kind(),
|
||||
"snippet_statement"
|
||||
| "if_statement"
|
||||
| "else_if_statement"
|
||||
| "else_statement"
|
||||
| "each_statement"
|
||||
| "await_statement"
|
||||
| "then_statement"
|
||||
| "catch_statement"
|
||||
| "render_expr"
|
||||
| "html_interpolation"
|
||||
| "interpolation"
|
||||
| "expression"
|
||||
| "element"
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
fn extract_markup_tag_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
start_like(node)
|
||||
.and_then(|start| child_by_kind(start, &["tag_name"]))
|
||||
.and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte())))
|
||||
}
|
||||
|
||||
fn has_attribute(node: Node<'_>, name: &str, source: &str) -> bool {
|
||||
start_like(node)
|
||||
.into_iter()
|
||||
.flat_map(named_children)
|
||||
.filter(|child| child.kind() == "attribute")
|
||||
.filter_map(|attr| extract_attribute_name(attr, source))
|
||||
.any(|attr_name| attr_name == name)
|
||||
}
|
||||
|
||||
fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option<String> {
|
||||
let start = start_like(node)?;
|
||||
for child in named_children(start) {
|
||||
if child.kind() != "attribute" {
|
||||
continue;
|
||||
}
|
||||
if extract_attribute_name(child, source).as_deref() != Some(name) {
|
||||
continue;
|
||||
}
|
||||
if let Some(value) = child_by_kind(child, &["attribute_value", "quoted_attribute_value"]) {
|
||||
return sanitize_identifier(&unquote_text(node_text(
|
||||
source,
|
||||
value.start_byte(),
|
||||
value.end_byte(),
|
||||
)));
|
||||
}
|
||||
return Some(name.to_string());
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn resolve_embedded_language(
|
||||
node: Node<'_>,
|
||||
source: &str,
|
||||
default_language: SupportLang,
|
||||
) -> Option<SupportLang> {
|
||||
if let Some(language) = attribute_value(node, "lang", source) {
|
||||
return SupportLang::from_alias(language.as_str());
|
||||
}
|
||||
Some(default_language)
|
||||
}
|
||||
|
||||
fn extract_attribute_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["attribute_name"])
|
||||
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
|
||||
}
|
||||
|
||||
fn start_like(node: Node<'_>) -> Option<Node<'_>> {
|
||||
child_by_kind(node, &["start_tag", "self_closing_tag"])
|
||||
}
|
||||
@@ -1,387 +0,0 @@
|
||||
//! Language-specific chunk classification for TLA+ / `PlusCal`.
|
||||
//!
|
||||
//! The shared defaults are too noisy for TLA+: the parser exposes a top-level
|
||||
//! `module` wrapper, `PlusCal` algorithms live inside block comments, and the
|
||||
//! generated translation section introduces operator definitions we do not want
|
||||
//! to surface in chunked read/edit views.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{
|
||||
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
|
||||
semantic_rule,
|
||||
},
|
||||
common::{
|
||||
ChunkContext, RawChunkCandidate, RecurseSpec, child_by_kind, extract_identifier,
|
||||
make_container_chunk, make_container_chunk_from, make_kind_chunk, recurse_self,
|
||||
sanitize_identifier,
|
||||
},
|
||||
kind::ChunkKind,
|
||||
types::ChunkNode,
|
||||
};
|
||||
|
||||
pub struct TlaplusClassifier;
|
||||
|
||||
impl LangClassifier for TlaplusClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[
|
||||
semantic_rule(
|
||||
"variable_declaration",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"constant_declaration",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"recursive_declaration",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
],
|
||||
class: &[semantic_rule(
|
||||
"pcal_var_decls",
|
||||
ChunkKind::Declarations,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
)],
|
||||
function: &[
|
||||
semantic_rule(
|
||||
"pcal_if",
|
||||
ChunkKind::If,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"pcal_while",
|
||||
ChunkKind::Loop,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"pcal_either",
|
||||
ChunkKind::Either,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"pcal_with",
|
||||
ChunkKind::With,
|
||||
RuleStyle::Positional,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
semantic_rule(
|
||||
"pcal_assign",
|
||||
ChunkKind::Statements,
|
||||
RuleStyle::Group,
|
||||
NamingMode::None,
|
||||
RecurseMode::None,
|
||||
),
|
||||
],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[
|
||||
"header_line",
|
||||
"double_line",
|
||||
"extends",
|
||||
"pcal_algorithm_start",
|
||||
],
|
||||
preserved_trivia: &["block_comment"],
|
||||
extra_root_wrappers: &[],
|
||||
preserved_root_wrappers: &["module"],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_root_custom(node, source),
|
||||
ChunkContext::ClassBody => classify_class_custom(node, source),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn preserve_children(
|
||||
&self,
|
||||
parent: &RawChunkCandidate<'_>,
|
||||
_children: &[RawChunkCandidate<'_>],
|
||||
) -> bool {
|
||||
matches!(
|
||||
parent.kind,
|
||||
ChunkKind::Module | ChunkKind::Algo | ChunkKind::Proc | ChunkKind::Process
|
||||
)
|
||||
}
|
||||
|
||||
fn post_process(
|
||||
&self,
|
||||
chunks: &mut Vec<ChunkNode>,
|
||||
root_children: &mut Vec<String>,
|
||||
source: &str,
|
||||
) {
|
||||
let ranges = translation_ranges(source);
|
||||
if ranges.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
let removed_by_range = ranges
|
||||
.iter()
|
||||
.map(|range| {
|
||||
chunks
|
||||
.iter()
|
||||
.filter(|chunk| {
|
||||
!chunk.path.is_empty() && chunk_wholly_inside_translation_fence(chunk, *range)
|
||||
})
|
||||
.cloned()
|
||||
.collect::<Vec<_>>()
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
let removed_paths = removed_by_range
|
||||
.iter()
|
||||
.flatten()
|
||||
.map(|chunk| chunk.path.clone())
|
||||
.collect::<Vec<_>>();
|
||||
if removed_paths.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
chunks.retain(|chunk| !removed_paths.iter().any(|removed| removed == &chunk.path));
|
||||
for chunk in chunks.iter_mut() {
|
||||
chunk
|
||||
.children
|
||||
.retain(|child| !removed_paths.iter().any(|removed| removed == child));
|
||||
}
|
||||
root_children.retain(|child| !removed_paths.iter().any(|removed| removed == child));
|
||||
|
||||
for (index, range) in ranges.iter().copied().enumerate() {
|
||||
let removed_chunks = &removed_by_range[index];
|
||||
if removed_chunks.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let synthetic = translation_chunk(range, removed_chunks[0].parent_path.clone(), source);
|
||||
let synthetic_path = synthetic.path.clone();
|
||||
if let Some(parent_path) = synthetic.parent_path.as_ref() {
|
||||
if let Some(parent) = chunks.iter_mut().find(|chunk| chunk.path == *parent_path) {
|
||||
parent.children.push(synthetic_path);
|
||||
}
|
||||
} else {
|
||||
root_children.push(synthetic_path);
|
||||
}
|
||||
chunks.push(synthetic);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"module" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Module,
|
||||
tla_identifier(node, source),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::Root)),
|
||||
)),
|
||||
"operator_definition" => Some(make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Operator,
|
||||
tla_identifier(node, source),
|
||||
source,
|
||||
None,
|
||||
)),
|
||||
"module_definition" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Module,
|
||||
tla_identifier(node, source),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::Root)),
|
||||
)),
|
||||
"pcal_algorithm" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Algo,
|
||||
tla_identifier(node, source),
|
||||
source,
|
||||
recurse_child(node, "pcal_algorithm_body", ChunkContext::ClassBody),
|
||||
)),
|
||||
"block_comment" => child_by_kind(node, &["pcal_algorithm"]).map(|algorithm| {
|
||||
make_container_chunk_from(
|
||||
node,
|
||||
algorithm,
|
||||
ChunkKind::Algo,
|
||||
tla_identifier(algorithm, source),
|
||||
source,
|
||||
recurse_child(algorithm, "pcal_algorithm_body", ChunkContext::ClassBody),
|
||||
)
|
||||
}),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"pcal_procedure" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Proc,
|
||||
tla_identifier(node, source),
|
||||
source,
|
||||
recurse_child(node, "pcal_algorithm_body", ChunkContext::ClassBody),
|
||||
)),
|
||||
"pcal_process" => Some(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Process,
|
||||
tla_identifier(node, source),
|
||||
source,
|
||||
recurse_child(node, "pcal_algorithm_body", ChunkContext::ClassBody),
|
||||
)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn tla_identifier(node: Node<'_>, source: &str) -> Option<String> {
|
||||
extract_identifier(node, source).or_else(|| {
|
||||
child_by_kind(node, &["identifier"])
|
||||
.and_then(|child| sanitize_identifier(child.utf8_text(source.as_bytes()).ok()?))
|
||||
})
|
||||
}
|
||||
|
||||
fn recurse_child<'tree>(
|
||||
node: Node<'tree>,
|
||||
kind: &'static str,
|
||||
context: ChunkContext,
|
||||
) -> Option<RecurseSpec<'tree>> {
|
||||
child_by_kind(node, &[kind]).map(|child| RecurseSpec { node: child, context })
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
struct TranslationRange {
|
||||
start_line: u32,
|
||||
end_line: u32,
|
||||
}
|
||||
|
||||
fn translation_ranges(source: &str) -> Vec<TranslationRange> {
|
||||
let lines = source.split('\n').collect::<Vec<_>>();
|
||||
let mut ranges = Vec::new();
|
||||
let mut current_start: Option<u32> = None;
|
||||
|
||||
for (index, line) in lines.iter().enumerate() {
|
||||
let line_no = index as u32 + 1;
|
||||
let trimmed = line.trim();
|
||||
if trimmed == r"\* BEGIN TRANSLATION" {
|
||||
current_start = Some(line_no);
|
||||
continue;
|
||||
}
|
||||
if trimmed == r"\* END TRANSLATION"
|
||||
&& let Some(start_line) = current_start.take()
|
||||
{
|
||||
push_translation_range(&mut ranges, start_line, line_no, lines.len() as u32);
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(start_line) = current_start {
|
||||
push_translation_range(&mut ranges, start_line, lines.len() as u32, lines.len() as u32);
|
||||
}
|
||||
|
||||
ranges
|
||||
}
|
||||
|
||||
fn push_translation_range(
|
||||
ranges: &mut Vec<TranslationRange>,
|
||||
start_line: u32,
|
||||
end_line: u32,
|
||||
total_lines: u32,
|
||||
) {
|
||||
let clamped_end = end_line.min(total_lines.max(start_line));
|
||||
ranges.push(TranslationRange { start_line, end_line: clamped_end });
|
||||
}
|
||||
|
||||
/// True when the chunk's span lies entirely inside the `\* BEGIN` … `\* END`
|
||||
/// translation fence. We must not treat broad containers (e.g. the `module`
|
||||
/// chunk spanning the whole file) as translation-only, or the module node is
|
||||
/// removed and children become orphaned.
|
||||
const fn chunk_wholly_inside_translation_fence(chunk: &ChunkNode, range: TranslationRange) -> bool {
|
||||
chunk.start_line >= range.start_line && chunk.end_line <= range.end_line
|
||||
}
|
||||
|
||||
fn translation_chunk(
|
||||
range: TranslationRange,
|
||||
parent_path: Option<String>,
|
||||
source: &str,
|
||||
) -> ChunkNode {
|
||||
let path = match &parent_path {
|
||||
Some(parent) => format!("{parent}.translation_{}", range.start_line),
|
||||
None => format!("translation_{}", range.start_line),
|
||||
};
|
||||
let (start_byte, end_byte) = byte_range_for_lines(source, range.start_line, range.end_line);
|
||||
let checksum = super::chunk_checksum(&source.as_bytes()[start_byte as usize..end_byte as usize]);
|
||||
ChunkNode {
|
||||
path,
|
||||
identifier: Some(range.start_line.to_string()),
|
||||
kind: ChunkKind::Translation,
|
||||
leaf: true,
|
||||
virtual_content: None,
|
||||
parent_path,
|
||||
children: Vec::new(),
|
||||
signature: Some("translation block".to_string()),
|
||||
start_line: range.start_line,
|
||||
end_line: range.end_line,
|
||||
line_count: range.end_line.saturating_sub(range.start_line) + 1,
|
||||
start_byte,
|
||||
end_byte,
|
||||
checksum_start_byte: start_byte,
|
||||
prologue_end_byte: None,
|
||||
epilogue_start_byte: None,
|
||||
checksum,
|
||||
error: false,
|
||||
indent: 0,
|
||||
indent_char: String::new(),
|
||||
group: false,
|
||||
}
|
||||
}
|
||||
|
||||
fn byte_range_for_lines(source: &str, start_line: u32, end_line: u32) -> (u32, u32) {
|
||||
let mut start_byte = 0usize;
|
||||
let mut current_line = 1u32;
|
||||
for (byte_index, byte) in source.bytes().enumerate() {
|
||||
if current_line == start_line {
|
||||
start_byte = byte_index;
|
||||
break;
|
||||
}
|
||||
if byte == b'\n' {
|
||||
current_line += 1;
|
||||
start_byte = byte_index + 1;
|
||||
}
|
||||
}
|
||||
|
||||
let mut end_byte = source.len();
|
||||
current_line = 1;
|
||||
for (byte_index, byte) in source.bytes().enumerate() {
|
||||
if current_line > end_line {
|
||||
end_byte = byte_index;
|
||||
break;
|
||||
}
|
||||
if byte == b'\n' {
|
||||
current_line += 1;
|
||||
}
|
||||
}
|
||||
|
||||
(start_byte as u32, end_byte as u32)
|
||||
}
|
||||
@@ -1,318 +0,0 @@
|
||||
//! Language-specific chunk classifier for Vue single-file components.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
|
||||
common::*,
|
||||
kind::ChunkKind,
|
||||
};
|
||||
use crate::language::SupportLang;
|
||||
|
||||
pub struct VueClassifier;
|
||||
|
||||
impl LangClassifier for VueClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
static TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides {
|
||||
extra_trivia: &[],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &["document"],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
},
|
||||
};
|
||||
&TABLES
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
context: ChunkContext,
|
||||
node: Node<'t>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
match context {
|
||||
ChunkContext::Root => classify_root_node(node, source),
|
||||
ChunkContext::ClassBody | ChunkContext::FunctionBody => classify_nested_node(node, source),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_root_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"template_element" => Some(classify_template_element(node, source)),
|
||||
"script_element" => Some(classify_script_element(node, source)),
|
||||
"style_element" => Some(classify_style_element(node, source)),
|
||||
// Vue custom blocks (for example <i18n>) currently parse as plain `element`
|
||||
// nodes at the document root, so infer custom-block semantics from position.
|
||||
"element" => Some(classify_custom_block(node, source)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_nested_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
match node.kind() {
|
||||
"template_element" => Some(classify_template_element(node, source)),
|
||||
"element" => classify_element(node, source),
|
||||
"start_tag" => classify_start_tag(node, source),
|
||||
"directive_attribute" => Some(classify_directive_attribute(node, source)),
|
||||
"attribute" => Some(classify_attribute(node, source)),
|
||||
"interpolation" => Some(make_kind_chunk(node, ChunkKind::Expression, None, source, None)),
|
||||
"text" => Some(group_candidate(node, ChunkKind::Text, source)),
|
||||
"raw_text" => Some(group_candidate(node, ChunkKind::Text, source)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_template_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let recurse = Some(recurse_self(node, ChunkContext::ClassBody));
|
||||
if let Some(slot_name) = extract_slot_name(node, source) {
|
||||
force_container(make_container_chunk(node, ChunkKind::Slot, Some(slot_name), source, recurse))
|
||||
} else {
|
||||
force_container(make_container_chunk(node, ChunkKind::Template, None, source, recurse))
|
||||
}
|
||||
}
|
||||
|
||||
fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let kind = if has_attribute(node, "setup", source) {
|
||||
ChunkKind::ScriptSetup
|
||||
} else if attribute_value(node, "context", source).as_deref() == Some("module") {
|
||||
ChunkKind::ScriptModule
|
||||
} else {
|
||||
ChunkKind::Script
|
||||
};
|
||||
classify_raw_text_block(node, kind, source, SupportLang::JavaScript)
|
||||
}
|
||||
|
||||
fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let kind = if has_attribute(node, "scoped", source) {
|
||||
ChunkKind::StyleScoped
|
||||
} else {
|
||||
ChunkKind::Style
|
||||
};
|
||||
classify_raw_text_block(node, kind, source, SupportLang::Css)
|
||||
}
|
||||
|
||||
fn classify_custom_block<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let tag_name = extract_markup_tag_name(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
make_named_container_chunk(node, ChunkKind::Custom, tag_name, source)
|
||||
}
|
||||
|
||||
fn classify_element<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
let tag_name = extract_markup_tag_name(node, source)?;
|
||||
Some(force_container(make_container_chunk(
|
||||
node,
|
||||
ChunkKind::Tag,
|
||||
Some(tag_name),
|
||||
source,
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
)))
|
||||
}
|
||||
|
||||
fn classify_start_tag<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
if !named_children(node)
|
||||
.into_iter()
|
||||
.any(|child| matches!(child.kind(), "attribute" | "directive_attribute"))
|
||||
{
|
||||
return None;
|
||||
}
|
||||
let tag_name = child_by_kind(node, &["tag_name"])
|
||||
.and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte())))
|
||||
.unwrap_or_else(|| "anonymous".to_string());
|
||||
Some(make_named_container_chunk(node, ChunkKind::Attrs, tag_name, source))
|
||||
}
|
||||
|
||||
fn classify_attribute<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let name = child_by_kind(node, &["attribute_name"])
|
||||
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
|
||||
.unwrap_or_else(|| "attr".to_string());
|
||||
make_kind_chunk(node, ChunkKind::Attr, Some(name), source, None)
|
||||
}
|
||||
|
||||
fn classify_directive_attribute<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
|
||||
let raw = node_text(source, node.start_byte(), node.end_byte()).trim();
|
||||
let directive_name =
|
||||
extract_directive_name(node, source).unwrap_or_else(|| "directive".to_string());
|
||||
let modifier_suffix = extract_directive_modifiers(node, source)
|
||||
.filter(|mods| !mods.is_empty())
|
||||
.map(|mods| format!("_{mods}"))
|
||||
.unwrap_or_default();
|
||||
if raw.starts_with('@') {
|
||||
make_named_leaf_chunk(
|
||||
node,
|
||||
ChunkKind::Directive,
|
||||
format!("on_{directive_name}{modifier_suffix}"),
|
||||
source,
|
||||
)
|
||||
} else if raw.starts_with(':') {
|
||||
make_named_leaf_chunk(
|
||||
node,
|
||||
ChunkKind::Directive,
|
||||
format!("bind_{directive_name}{modifier_suffix}"),
|
||||
source,
|
||||
)
|
||||
} else if raw.starts_with('#') {
|
||||
make_kind_chunk(
|
||||
node,
|
||||
ChunkKind::Slot,
|
||||
Some(format!("{directive_name}{modifier_suffix}")),
|
||||
source,
|
||||
None,
|
||||
)
|
||||
} else {
|
||||
make_named_leaf_chunk(
|
||||
node,
|
||||
ChunkKind::Directive,
|
||||
format!("dir_{directive_name}{modifier_suffix}"),
|
||||
source,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
fn make_named_leaf_chunk<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
identifier: impl Into<Option<String>>,
|
||||
source: &str,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
make_candidate(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
None,
|
||||
source,
|
||||
)
|
||||
}
|
||||
|
||||
fn make_named_container_chunk<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
identifier: impl Into<Option<String>>,
|
||||
source: &str,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
force_container(make_candidate(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
Some(recurse_self(node, ChunkContext::ClassBody)),
|
||||
source,
|
||||
))
|
||||
}
|
||||
|
||||
const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> {
|
||||
candidate.force_recurse = true;
|
||||
candidate
|
||||
}
|
||||
|
||||
fn classify_raw_text_block<'t>(
|
||||
node: Node<'t>,
|
||||
kind: ChunkKind,
|
||||
source: &str,
|
||||
default_language: SupportLang,
|
||||
) -> RawChunkCandidate<'t> {
|
||||
let Some(content_node) = child_by_kind(node, &["raw_text"]) else {
|
||||
return positional_candidate(node, kind, source);
|
||||
};
|
||||
let candidate = with_region_node(positional_candidate(node, kind, source), Some(content_node));
|
||||
match resolve_embedded_language(node, source, default_language) {
|
||||
Some(language) => with_injected_subtree(candidate, language, content_node),
|
||||
None => candidate,
|
||||
}
|
||||
}
|
||||
|
||||
fn extract_markup_tag_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
start_like(node)
|
||||
.and_then(|start| child_by_kind(start, &["tag_name"]))
|
||||
.and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte())))
|
||||
}
|
||||
|
||||
fn extract_slot_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let start = start_like(node)?;
|
||||
named_children(start)
|
||||
.into_iter()
|
||||
.find(|child| {
|
||||
node_text(source, child.start_byte(), child.end_byte())
|
||||
.trim()
|
||||
.starts_with('#')
|
||||
})
|
||||
.and_then(|child| extract_directive_name(child, source))
|
||||
}
|
||||
|
||||
fn extract_directive_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["directive_name", "directive_value"])
|
||||
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
|
||||
}
|
||||
|
||||
fn extract_directive_modifiers(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["directive_modifiers"])
|
||||
.and_then(|mods| sanitize_identifier(node_text(source, mods.start_byte(), mods.end_byte())))
|
||||
}
|
||||
|
||||
fn has_attribute(node: Node<'_>, name: &str, source: &str) -> bool {
|
||||
start_like(node)
|
||||
.into_iter()
|
||||
.flat_map(named_children)
|
||||
.filter(|child| matches!(child.kind(), "attribute" | "directive_attribute"))
|
||||
.filter_map(|attr| extract_attribute_name(attr, source))
|
||||
.any(|attr_name| attr_name == name)
|
||||
}
|
||||
|
||||
fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option<String> {
|
||||
let start = start_like(node)?;
|
||||
for child in named_children(start) {
|
||||
if !matches!(child.kind(), "attribute" | "directive_attribute") {
|
||||
continue;
|
||||
}
|
||||
if extract_attribute_name(child, source).as_deref() != Some(name) {
|
||||
continue;
|
||||
}
|
||||
if let Some(value) =
|
||||
child_by_kind(child, &["attribute_value", "quoted_attribute_value", "directive_value"])
|
||||
{
|
||||
return sanitize_identifier(&unquote_text(node_text(
|
||||
source,
|
||||
value.start_byte(),
|
||||
value.end_byte(),
|
||||
)));
|
||||
}
|
||||
return Some(name.to_string());
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn resolve_embedded_language(
|
||||
node: Node<'_>,
|
||||
source: &str,
|
||||
default_language: SupportLang,
|
||||
) -> Option<SupportLang> {
|
||||
if let Some(language) = attribute_value(node, "lang", source) {
|
||||
return SupportLang::from_alias(language.as_str());
|
||||
}
|
||||
Some(default_language)
|
||||
}
|
||||
|
||||
fn extract_attribute_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
child_by_kind(node, &["attribute_name", "directive_name"])
|
||||
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
|
||||
.or_else(|| {
|
||||
if node_text(source, node.start_byte(), node.end_byte())
|
||||
.trim()
|
||||
.starts_with('#')
|
||||
{
|
||||
extract_directive_name(node, source)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
fn start_like(node: Node<'_>) -> Option<Node<'_>> {
|
||||
child_by_kind(node, &["start_tag", "self_closing_tag"])
|
||||
}
|
||||
@@ -1,145 +0,0 @@
|
||||
use std::collections::{HashMap, HashSet};
|
||||
|
||||
use super::schema;
|
||||
|
||||
type AtomSet = HashSet<&'static str>;
|
||||
|
||||
static ATOM_NODES: std::sync::LazyLock<HashMap<&'static str, AtomSet>> =
|
||||
std::sync::LazyLock::new(|| {
|
||||
HashMap::from([
|
||||
("astro", HashSet::from(["frontmatter"])),
|
||||
("bash", HashSet::from(["string", "raw_string", "heredoc_body", "simple_expansion"])),
|
||||
("c", HashSet::from(["string_literal", "char_literal"])),
|
||||
("clojure", HashSet::from(["kwd_lit", "regex_lit"])),
|
||||
("cmake", HashSet::from(["argument"])),
|
||||
("cpp", HashSet::from(["string_literal", "char_literal"])),
|
||||
(
|
||||
"csharp",
|
||||
HashSet::from([
|
||||
"string_literal",
|
||||
"verbatim_string_literal",
|
||||
"character_literal",
|
||||
"modifier",
|
||||
]),
|
||||
),
|
||||
("css", HashSet::from(["integer_value", "float_value", "color_value", "string_value"])),
|
||||
("elixir", HashSet::from(["string_constant_expr"])),
|
||||
("go", HashSet::from(["interpreted_string_literal", "raw_string_literal"])),
|
||||
(
|
||||
"haskell",
|
||||
HashSet::from([
|
||||
"qualified_variable",
|
||||
"qualified_module",
|
||||
"qualified_constructor",
|
||||
"strict_type",
|
||||
]),
|
||||
),
|
||||
("hcl", HashSet::from(["string_lit", "heredoc_template"])),
|
||||
(
|
||||
"html",
|
||||
HashSet::from(["doctype", "quoted_attribute_value", "raw_text", "tag_name", "text"]),
|
||||
),
|
||||
(
|
||||
"java",
|
||||
HashSet::from([
|
||||
"string_literal",
|
||||
"boolean_type",
|
||||
"integral_type",
|
||||
"floating_point_type",
|
||||
"void_type",
|
||||
]),
|
||||
),
|
||||
("json", HashSet::from(["string"])),
|
||||
(
|
||||
"julia",
|
||||
HashSet::from([
|
||||
"string_literal",
|
||||
"prefixed_string_literal",
|
||||
"command_literal",
|
||||
"character_literal",
|
||||
]),
|
||||
),
|
||||
(
|
||||
"kotlin",
|
||||
HashSet::from([
|
||||
"nullable_type",
|
||||
"string_literal",
|
||||
"line_string_literal",
|
||||
"character_literal",
|
||||
]),
|
||||
),
|
||||
("lua", HashSet::from(["string"])),
|
||||
("make", HashSet::from(["shell_text", "text"])),
|
||||
("nix", HashSet::from(["string_expression", "indented_string_expression"])),
|
||||
("objc", HashSet::from(["string_literal"])),
|
||||
(
|
||||
"perl",
|
||||
HashSet::from([
|
||||
"string_single_quoted",
|
||||
"string_double_quoted",
|
||||
"comments",
|
||||
"command_qx_quoted",
|
||||
"pattern_matcher_m",
|
||||
"regex_pattern_qr",
|
||||
"transliteration_tr_or_y",
|
||||
"substitution_pattern_s",
|
||||
"scalar_variable",
|
||||
"array_variable",
|
||||
"hash_variable",
|
||||
"hash_access_variable",
|
||||
]),
|
||||
),
|
||||
("php", HashSet::from(["string", "encapsed_string"])),
|
||||
("protobuf", HashSet::from(["string"])),
|
||||
("python", HashSet::from(["string"])),
|
||||
("r", HashSet::from(["string", "special"])),
|
||||
("ruby", HashSet::from(["string", "heredoc_body", "regex"])),
|
||||
("rust", HashSet::from(["char_literal", "string_literal", "raw_string_literal"])),
|
||||
("scala", HashSet::from(["string", "template_string", "interpolated_string_expression"])),
|
||||
("solidity", HashSet::from(["string", "hex_string_literal", "unicode_string_literal"])),
|
||||
("sql", HashSet::from(["string", "identifier"])),
|
||||
("swift", HashSet::from(["line_string_literal"])),
|
||||
("toml", HashSet::from(["string", "quoted_key"])),
|
||||
("tsx", HashSet::from(["string", "template_string"])),
|
||||
("typescript", HashSet::from(["string", "template_string", "regex", "predefined_type"])),
|
||||
("xml", HashSet::from(["AttValue", "XMLDecl"])),
|
||||
(
|
||||
"yaml",
|
||||
HashSet::from([
|
||||
"string_scalar",
|
||||
"double_quote_scalar",
|
||||
"single_quote_scalar",
|
||||
"block_scalar",
|
||||
]),
|
||||
),
|
||||
("verilog", HashSet::from(["integral_number"])),
|
||||
("zig", HashSet::from(["string"])),
|
||||
])
|
||||
});
|
||||
|
||||
pub fn is_atom_node(language: &str, kind: &str) -> bool {
|
||||
ATOM_NODES
|
||||
.get(language)
|
||||
.is_some_and(|atom_nodes| atom_nodes.contains(kind))
|
||||
}
|
||||
|
||||
pub fn is_atom_node_current(kind: &str) -> bool {
|
||||
schema::current_language().is_some_and(|language| is_atom_node(language, kind))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::is_atom_node;
|
||||
|
||||
#[test]
|
||||
fn nix_binding_set_is_not_an_atom() {
|
||||
assert!(!is_atom_node("nix", "binding_set"));
|
||||
assert!(is_atom_node("nix", "string_expression"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn typescript_predefined_types_stay_atomic() {
|
||||
assert!(is_atom_node("typescript", "predefined_type"));
|
||||
assert!(!is_atom_node("typescript", "class_declaration"));
|
||||
}
|
||||
}
|
||||
@@ -1,509 +0,0 @@
|
||||
//! Per-language chunk classification trait.
|
||||
//!
|
||||
//! Languages now provide semantic tables plus a narrow override hook for
|
||||
//! genuinely custom behavior.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
common::{
|
||||
ChunkContext, NameStyle, RawChunkCandidate, extract_identifier, is_absorbable_attribute,
|
||||
is_trivia_node, make_candidate, named_children, recurse_self, resolve_recurse,
|
||||
resolve_value_container, sanitize_node_kind, signature_for_node,
|
||||
try_promote_call_with_callback,
|
||||
},
|
||||
defaults,
|
||||
kind::ChunkKind,
|
||||
schema,
|
||||
};
|
||||
use crate::chunk::types::ChunkNode;
|
||||
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub enum RuleStyle {
|
||||
Named,
|
||||
Group,
|
||||
Positional,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub enum NamingMode {
|
||||
AutoIdentifier,
|
||||
None,
|
||||
SanitizedKind,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub enum RecurseMode {
|
||||
None,
|
||||
Auto(ChunkContext),
|
||||
SelfNode(ChunkContext),
|
||||
ValueContainer,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
|
||||
pub enum WrapperSignature {
|
||||
#[default]
|
||||
Child,
|
||||
Wrapper,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug, Default)]
|
||||
pub struct WrapperTransform {
|
||||
pub kind: Option<ChunkKind>,
|
||||
pub name_style: Option<NameStyle>,
|
||||
pub clear_identifier: bool,
|
||||
pub signature: WrapperSignature,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub struct SemanticRule {
|
||||
pub ts_kind: &'static str,
|
||||
pub chunk_kind: ChunkKind,
|
||||
pub style: RuleStyle,
|
||||
pub naming: NamingMode,
|
||||
pub recurse: RecurseMode,
|
||||
}
|
||||
|
||||
pub const fn semantic_rule(
|
||||
ts_kind: &'static str,
|
||||
chunk_kind: ChunkKind,
|
||||
style: RuleStyle,
|
||||
naming: NamingMode,
|
||||
recurse: RecurseMode,
|
||||
) -> SemanticRule {
|
||||
SemanticRule { ts_kind, chunk_kind, style, naming, recurse }
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub struct StructuralOverrides {
|
||||
pub extra_trivia: &'static [&'static str],
|
||||
pub preserved_trivia: &'static [&'static str],
|
||||
pub extra_root_wrappers: &'static [&'static str],
|
||||
pub preserved_root_wrappers: &'static [&'static str],
|
||||
pub absorbable_attrs: &'static [&'static str],
|
||||
}
|
||||
|
||||
impl StructuralOverrides {
|
||||
pub const EMPTY: Self = Self {
|
||||
extra_trivia: &[],
|
||||
preserved_trivia: &[],
|
||||
extra_root_wrappers: &[],
|
||||
preserved_root_wrappers: &[],
|
||||
absorbable_attrs: &[],
|
||||
};
|
||||
|
||||
pub fn is_extra_trivia(&self, kind: &str) -> bool {
|
||||
self.extra_trivia.contains(&kind)
|
||||
}
|
||||
|
||||
pub fn preserves_trivia(&self, kind: &str) -> bool {
|
||||
self.preserved_trivia.contains(&kind)
|
||||
}
|
||||
|
||||
pub fn is_extra_root_wrapper(&self, kind: &str) -> bool {
|
||||
self.extra_root_wrappers.contains(&kind)
|
||||
}
|
||||
|
||||
pub fn preserves_root_wrapper(&self, kind: &str) -> bool {
|
||||
self.preserved_root_wrappers.contains(&kind)
|
||||
}
|
||||
|
||||
pub fn is_absorbable_attr(&self, kind: &str) -> bool {
|
||||
self.absorbable_attrs.contains(&kind)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub struct ClassifierTables {
|
||||
pub root: &'static [SemanticRule],
|
||||
pub class: &'static [SemanticRule],
|
||||
pub function: &'static [SemanticRule],
|
||||
pub structural_overrides: StructuralOverrides,
|
||||
}
|
||||
|
||||
pub const EMPTY_CLASSIFIER_TABLES: ClassifierTables = ClassifierTables {
|
||||
root: &[],
|
||||
class: &[],
|
||||
function: &[],
|
||||
structural_overrides: StructuralOverrides::EMPTY,
|
||||
};
|
||||
|
||||
pub trait LangClassifier {
|
||||
fn tables(&self) -> &'static ClassifierTables {
|
||||
&EMPTY_CLASSIFIER_TABLES
|
||||
}
|
||||
|
||||
fn classify_root<'t>(&self, _node: Node<'t>, _source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
None
|
||||
}
|
||||
|
||||
fn classify_class<'t>(&self, _node: Node<'t>, _source: &str) -> Option<RawChunkCandidate<'t>> {
|
||||
None
|
||||
}
|
||||
|
||||
fn classify_function<'t>(
|
||||
&self,
|
||||
_node: Node<'t>,
|
||||
_source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
None
|
||||
}
|
||||
|
||||
fn is_root_wrapper(&self, _kind: &str) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
fn preserve_root_wrapper(&self, _kind: &str) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
fn preserve_trivia(&self, _kind: &str) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
fn is_trivia(&self, _kind: &str) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
fn is_absorbable_attr(&self, _kind: &str) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
/// Return true to drop a named child from `collect_children_for_context`
|
||||
/// without turning it into a chunk and without absorbing its byte range
|
||||
/// into the next sibling via `attach_leading_trivia`.
|
||||
///
|
||||
/// Use this for structural framing nodes (JSX opening/closing elements,
|
||||
/// framework fragment markers, etc.) that have no meaningful chunk of
|
||||
/// their own and should NOT extend the following chunk's span backward.
|
||||
/// Prefer `is_trivia` for comment-like nodes that should be absorbed as
|
||||
/// leading context of the next chunk.
|
||||
fn should_skip_child(&self, _kind: &str) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
fn classify_override<'t>(
|
||||
&self,
|
||||
_context: ChunkContext,
|
||||
_node: Node<'t>,
|
||||
_source: &str,
|
||||
) -> Option<RawChunkCandidate<'t>> {
|
||||
None
|
||||
}
|
||||
|
||||
fn preserve_children(
|
||||
&self,
|
||||
_parent: &RawChunkCandidate<'_>,
|
||||
_children: &[RawChunkCandidate<'_>],
|
||||
) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
fn post_process(
|
||||
&self,
|
||||
_chunks: &mut Vec<ChunkNode>,
|
||||
_root_children: &mut Vec<String>,
|
||||
_source: &str,
|
||||
) {
|
||||
}
|
||||
}
|
||||
|
||||
pub fn structural_overrides(classifier: &dyn LangClassifier) -> StructuralOverrides {
|
||||
classifier.tables().structural_overrides
|
||||
}
|
||||
|
||||
pub fn classify_with_tables<'tree>(
|
||||
classifier: &dyn LangClassifier,
|
||||
context: ChunkContext,
|
||||
node: Node<'tree>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'tree>> {
|
||||
if let Some(candidate) = classifier.classify_override(context, node, source) {
|
||||
return Some(candidate);
|
||||
}
|
||||
|
||||
find_rule(classifier.tables(), context, node.kind())
|
||||
.map(|rule| build_candidate_from_rule(node, source, *rule))
|
||||
.or_else(|| match context {
|
||||
ChunkContext::Root => classifier.classify_root(node, source),
|
||||
ChunkContext::ClassBody => classifier.classify_class(node, source),
|
||||
ChunkContext::FunctionBody => classifier.classify_function(node, source),
|
||||
})
|
||||
}
|
||||
|
||||
pub fn classify_with_defaults<'tree>(
|
||||
classifier: &dyn LangClassifier,
|
||||
context: ChunkContext,
|
||||
node: Node<'tree>,
|
||||
source: &str,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
if node.is_error() || node.kind() == "ERROR" {
|
||||
return make_candidate(node, ChunkKind::Error, None, NameStyle::Error, None, None, source);
|
||||
}
|
||||
|
||||
let candidate = match context {
|
||||
ChunkContext::Root => classify_with_tables(classifier, context, node, source)
|
||||
.unwrap_or_else(|| defaults::classify_root_default(node, source)),
|
||||
ChunkContext::ClassBody => classify_with_tables(classifier, context, node, source)
|
||||
.unwrap_or_else(|| defaults::classify_class_default(node, source)),
|
||||
ChunkContext::FunctionBody => classify_with_tables(classifier, context, node, source)
|
||||
.unwrap_or_else(|| defaults::classify_function_default(node, source)),
|
||||
};
|
||||
|
||||
// If the classifier produced a groupable leaf (no recurse), try to
|
||||
// promote call-with-trailing-callback patterns into named container
|
||||
// chunks. This handles `describe(...)`, `t.Run(...)`, etc. across
|
||||
// all languages without per-language opt-in.
|
||||
if candidate.recurse.is_none()
|
||||
&& candidate.groupable
|
||||
&& let Some(promoted) = try_promote_call_with_callback(node, source)
|
||||
{
|
||||
return promoted;
|
||||
}
|
||||
|
||||
candidate
|
||||
}
|
||||
|
||||
pub fn first_wrapper_content_child<'tree>(
|
||||
classifier: &dyn LangClassifier,
|
||||
node: Node<'tree>,
|
||||
) -> Option<Node<'tree>> {
|
||||
if let Some(child) = schema_wrapper_child(node) {
|
||||
return Some(child);
|
||||
}
|
||||
|
||||
let overrides = structural_overrides(classifier);
|
||||
named_children(node)
|
||||
.into_iter()
|
||||
.find(|child| !is_wrapper_metadata_child(*child, classifier, overrides))
|
||||
}
|
||||
|
||||
pub fn promote_wrapper_candidate<'tree>(
|
||||
classifier: &dyn LangClassifier,
|
||||
context: ChunkContext,
|
||||
node: Node<'tree>,
|
||||
source: &str,
|
||||
transform: WrapperTransform,
|
||||
) -> Option<RawChunkCandidate<'tree>> {
|
||||
let (child, candidate) = promotable_wrapper_child(classifier, context, node, source)?;
|
||||
let signature_node = match transform.signature {
|
||||
WrapperSignature::Child => child,
|
||||
WrapperSignature::Wrapper => node,
|
||||
};
|
||||
let kind = transform.kind.unwrap_or(candidate.kind);
|
||||
let name_style = transform.name_style.unwrap_or(candidate.name_style);
|
||||
let identifier = if transform.clear_identifier {
|
||||
None
|
||||
} else {
|
||||
candidate.identifier
|
||||
};
|
||||
|
||||
Some(make_candidate(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
name_style,
|
||||
signature_for_node(signature_node, source),
|
||||
candidate.recurse,
|
||||
source,
|
||||
))
|
||||
}
|
||||
|
||||
pub fn build_candidate_from_rule<'tree>(
|
||||
node: Node<'tree>,
|
||||
source: &str,
|
||||
rule: SemanticRule,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
let identifier = match rule.naming {
|
||||
NamingMode::AutoIdentifier => extract_identifier(node, source),
|
||||
NamingMode::None => None,
|
||||
NamingMode::SanitizedKind => Some(sanitize_node_kind(node.kind()).to_string()),
|
||||
};
|
||||
|
||||
let recurse = match rule.recurse {
|
||||
RecurseMode::None => None,
|
||||
RecurseMode::Auto(context) => resolve_recurse(node, context),
|
||||
RecurseMode::SelfNode(context) => Some(recurse_self(node, context)),
|
||||
RecurseMode::ValueContainer => resolve_value_container(node),
|
||||
};
|
||||
|
||||
match rule.style {
|
||||
RuleStyle::Named => make_candidate(
|
||||
node,
|
||||
rule.chunk_kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse,
|
||||
source,
|
||||
),
|
||||
RuleStyle::Group => {
|
||||
make_candidate(node, rule.chunk_kind, identifier, NameStyle::Group, None, recurse, source)
|
||||
},
|
||||
RuleStyle::Positional => make_candidate(
|
||||
node,
|
||||
rule.chunk_kind,
|
||||
None::<String>,
|
||||
NameStyle::Named,
|
||||
None,
|
||||
recurse,
|
||||
source,
|
||||
),
|
||||
}
|
||||
}
|
||||
|
||||
fn find_rule(
|
||||
tables: &ClassifierTables,
|
||||
context: ChunkContext,
|
||||
kind: &str,
|
||||
) -> Option<&'static SemanticRule> {
|
||||
let rules = match context {
|
||||
ChunkContext::Root => tables.root,
|
||||
ChunkContext::ClassBody => tables.class,
|
||||
ChunkContext::FunctionBody => tables.function,
|
||||
};
|
||||
|
||||
rules.iter().find(|rule| rule.ts_kind == kind)
|
||||
}
|
||||
|
||||
fn promotable_wrapper_child<'tree>(
|
||||
classifier: &dyn LangClassifier,
|
||||
context: ChunkContext,
|
||||
node: Node<'tree>,
|
||||
source: &str,
|
||||
) -> Option<(Node<'tree>, RawChunkCandidate<'tree>)> {
|
||||
if let Some(child) = schema_wrapper_child(node) {
|
||||
let candidate = classify_with_defaults(classifier, context, child, source);
|
||||
if is_promotable_wrapper_candidate(child, &candidate) {
|
||||
return Some((child, candidate));
|
||||
}
|
||||
}
|
||||
|
||||
let overrides = structural_overrides(classifier);
|
||||
let mut promoted = named_children(node).into_iter().filter_map(|child| {
|
||||
if is_wrapper_metadata_child(child, classifier, overrides) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let candidate = classify_with_defaults(classifier, context, child, source);
|
||||
is_promotable_wrapper_candidate(child, &candidate).then_some((child, candidate))
|
||||
});
|
||||
|
||||
let promoted_child = promoted.next()?;
|
||||
if promoted.next().is_some() {
|
||||
return None;
|
||||
}
|
||||
Some(promoted_child)
|
||||
}
|
||||
|
||||
fn is_wrapper_metadata_child(
|
||||
node: Node<'_>,
|
||||
classifier: &dyn LangClassifier,
|
||||
overrides: StructuralOverrides,
|
||||
) -> bool {
|
||||
let kind = node.kind();
|
||||
((is_trivia_node(node) || classifier.is_trivia(kind))
|
||||
&& !overrides.preserves_trivia(kind)
|
||||
&& !classifier.preserve_trivia(kind))
|
||||
|| (overrides.is_extra_trivia(kind)
|
||||
&& !overrides.preserves_trivia(kind)
|
||||
&& !classifier.preserve_trivia(kind))
|
||||
|| is_absorbable_attribute(kind)
|
||||
|| overrides.is_absorbable_attr(kind)
|
||||
|| classifier.is_absorbable_attr(kind)
|
||||
}
|
||||
|
||||
fn is_promotable_wrapper_candidate(node: Node<'_>, candidate: &RawChunkCandidate<'_>) -> bool {
|
||||
if matches!(candidate.kind, ChunkKind::Error | ChunkKind::Chunk | ChunkKind::Statements) {
|
||||
return false;
|
||||
}
|
||||
|
||||
candidate.identifier.is_some()
|
||||
|| candidate.recurse.is_some()
|
||||
|| candidate.kind.traits().container
|
||||
|| node.kind().ends_with("_definition")
|
||||
|| node.kind().ends_with("_declaration")
|
||||
}
|
||||
|
||||
fn schema_wrapper_child(node: Node<'_>) -> Option<Node<'_>> {
|
||||
let schema = schema::schema_for_current(node.kind())?;
|
||||
for field in &schema.promotion_fields {
|
||||
if let Some(child) = node.child_by_field_name(field) {
|
||||
return Some(child);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
/// Resolve a [`LangClassifier`] for the given language.
|
||||
pub fn classifier_for(lang: &str) -> &'static dyn LangClassifier {
|
||||
match lang {
|
||||
"astro" => &super::ast_astro::AstroClassifier,
|
||||
// JS / TS family
|
||||
"javascript" | "js" | "jsx" | "typescript" | "ts" | "tsx" => {
|
||||
&super::ast_js_ts::JsTsClassifier
|
||||
},
|
||||
// Python / Starlark
|
||||
"python" | "starlark" => &super::ast_python::PythonClassifier,
|
||||
// Rust
|
||||
"rust" => &super::ast_rust::RustClassifier,
|
||||
// Go
|
||||
"go" | "golang" => &super::ast_go::GoClassifier,
|
||||
// C / C++ / Objective-C
|
||||
"c" | "cpp" | "c++" | "objc" | "objective-c" => &super::ast_c_cpp_objc::CCppClassifier,
|
||||
// C# / Java
|
||||
"csharp" | "java" => &super::ast_csharp_java::CSharpJavaClassifier,
|
||||
// Clojure
|
||||
"clojure" => &super::ast_clojure::ClojureClassifier,
|
||||
// CMake
|
||||
"cmake" => &super::ast_cmake::CMakeClassifier,
|
||||
// CSS
|
||||
"css" => &super::ast_css::CssClassifier,
|
||||
// Data formats
|
||||
"json" | "toml" | "yaml" => &super::ast_data_formats::DataFormatsClassifier,
|
||||
// Dockerfile
|
||||
"dockerfile" => &super::ast_dockerfile::DockerfileClassifier,
|
||||
// Elixir
|
||||
"elixir" => &super::ast_elixir::ElixirClassifier,
|
||||
// Erlang
|
||||
"erlang" => &super::ast_erlang::ErlangClassifier,
|
||||
// GraphQL
|
||||
"graphql" => &super::ast_graphql::GraphqlClassifier,
|
||||
// Haskell / Scala
|
||||
"haskell" | "scala" => &super::ast_haskell_scala::HaskellScalaClassifier,
|
||||
// HTML / XML
|
||||
"html" | "xml" => &super::ast_html_xml::HtmlXmlClassifier,
|
||||
// INI
|
||||
"ini" => &super::ast_ini::IniClassifier,
|
||||
// Just
|
||||
"just" => &super::ast_just::JustClassifier,
|
||||
// Markdown / Handlebars
|
||||
"markdown" | "handlebars" => &super::ast_markup::MarkupClassifier,
|
||||
// Nix / HCL
|
||||
"nix" | "hcl" => &super::ast_nix_hcl::NixHclClassifier,
|
||||
// OCaml
|
||||
"ocaml" => &super::ast_ocaml::OcamlClassifier,
|
||||
// Perl
|
||||
"perl" => &super::ast_perl::PerlClassifier,
|
||||
// PowerShell
|
||||
"powershell" => &super::ast_powershell::PowershellClassifier,
|
||||
// Protobuf
|
||||
"protobuf" | "proto" => &super::ast_proto::ProtoClassifier,
|
||||
// R
|
||||
"r" => &super::ast_r::RClassifier,
|
||||
// Ruby / Lua
|
||||
"ruby" | "lua" => &super::ast_ruby_lua::RubyLuaClassifier,
|
||||
// SQL
|
||||
"sql" => &super::ast_sql::SqlClassifier,
|
||||
// Svelte
|
||||
"svelte" => &super::ast_svelte::SvelteClassifier,
|
||||
// TLA+ / PlusCal
|
||||
"tlaplus" | "pluscal" | "pcal" | "tla" | "tla+" => &super::ast_tlaplus::TlaplusClassifier,
|
||||
// Bash / Make / Diff
|
||||
"bash" | "make" | "diff" => &super::ast_bash_make_diff::ShellBuildClassifier,
|
||||
// Vue
|
||||
"vue" => &super::ast_vue::VueClassifier,
|
||||
// Everything else (Kotlin, Swift, PHP, Solidity, etc.)
|
||||
_ => &super::ast_misc::MiscClassifier,
|
||||
}
|
||||
}
|
||||
@@ -1,903 +0,0 @@
|
||||
//! Shared helpers for chunk classification.
|
||||
//!
|
||||
//! These are the building blocks that per-language classifiers use to construct
|
||||
//! [`RawChunkCandidate`] values. They are also used by the default (shared)
|
||||
//! classification in [`super::defaults`].
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{
|
||||
kind::{ChunkKind, SummaryStyle},
|
||||
shape,
|
||||
types::ChunkNode,
|
||||
};
|
||||
use crate::{env_uint, language::SupportLang};
|
||||
|
||||
// ── Configuration (environment overrides) ────────────────────────────────
|
||||
env_uint! {
|
||||
// Configured leaf threshold.
|
||||
pub static LEAF_THRESHOLD: usize = "PI_CHUNK_LEAF_THRESHOLD" or 8 => [1, usize::MAX];
|
||||
// Configured max chunk lines.
|
||||
pub static MAX_CHUNK_LINES: usize = "PI_CHUNK_MAX_LINES" or 25 => [1, usize::MAX];
|
||||
// Configured min recurse savings.
|
||||
pub static MIN_RECURSE_SAVINGS: usize = "PI_CHUNK_MIN_SAVINGS" or 4 => [1, usize::MAX];
|
||||
}
|
||||
|
||||
// ── Internal types ───────────────────────────────────────────────────────
|
||||
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum ChunkContext {
|
||||
Root,
|
||||
ClassBody,
|
||||
FunctionBody,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum NameStyle {
|
||||
Named,
|
||||
Group,
|
||||
Error,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub struct RecurseSpec<'tree> {
|
||||
pub node: Node<'tree>,
|
||||
pub context: ChunkContext,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub struct InjectedChunkSpec<'tree> {
|
||||
pub language: SupportLang,
|
||||
pub content_node: Node<'tree>,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct RawChunkCandidate<'tree> {
|
||||
pub identifier: Option<String>,
|
||||
pub kind: ChunkKind,
|
||||
pub name_style: NameStyle,
|
||||
pub range_start_byte: usize,
|
||||
pub range_end_byte: usize,
|
||||
/// Start byte for `chunk_checksum`; stays at the primary node's start while
|
||||
/// `range_start_byte` may be extended backward to include leading
|
||||
/// attributes/comments.
|
||||
pub checksum_start_byte: usize,
|
||||
pub range_start_line: usize,
|
||||
pub range_end_line: usize,
|
||||
pub signature: Option<String>,
|
||||
pub error: bool,
|
||||
pub groupable: bool,
|
||||
pub has_leading_comment: bool,
|
||||
pub force_recurse: bool,
|
||||
pub region_node: Option<Node<'tree>>,
|
||||
pub injected: Option<InjectedChunkSpec<'tree>>,
|
||||
pub recurse: Option<RecurseSpec<'tree>>,
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct ChunkAccumulator {
|
||||
pub chunks: Vec<ChunkNode>,
|
||||
}
|
||||
|
||||
// ── Candidate constructors ───────────────────────────────────────────────
|
||||
|
||||
/// Convert a tree-sitter `end_position` into a 1-indexed line number.
|
||||
///
|
||||
/// Tree-sitter byte ranges are half-open, so `end_position` points to the byte
|
||||
/// immediately *after* the last byte of the node. When that byte lands at
|
||||
/// column 0 of a new row, the node's last byte actually sits on the previous
|
||||
/// row and the 1-indexed last line is exactly `end.row` — not `end.row + 1`.
|
||||
/// This matters for grammars whose container nodes terminate on the start of
|
||||
/// the next sibling (tree-sitter-markdown sections, tree-sitter-toml tables,
|
||||
/// etc.): the naive `end.row + 1` would claim the sibling's heading line and
|
||||
/// make `replace_range_by_lines` clobber it.
|
||||
const fn end_row_as_line(start: tree_sitter::Point, end: tree_sitter::Point) -> usize {
|
||||
if end.column == 0 && end.row > start.row {
|
||||
end.row
|
||||
} else {
|
||||
end.row + 1
|
||||
}
|
||||
}
|
||||
|
||||
pub fn make_candidate<'tree>(
|
||||
node: Node<'tree>,
|
||||
kind: ChunkKind,
|
||||
identifier: impl Into<Option<String>>,
|
||||
name_style: NameStyle,
|
||||
signature: Option<String>,
|
||||
recurse: Option<RecurseSpec<'tree>>,
|
||||
source: &str,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
let identifier = identifier.into();
|
||||
let start = node.start_position();
|
||||
let end = node.end_position();
|
||||
let summary = summary_for_node(node, kind, identifier.as_deref(), signature.as_deref(), source);
|
||||
let start_byte = node.start_byte();
|
||||
RawChunkCandidate {
|
||||
identifier,
|
||||
kind,
|
||||
name_style,
|
||||
range_start_byte: start_byte,
|
||||
range_end_byte: node.end_byte(),
|
||||
checksum_start_byte: start_byte,
|
||||
range_start_line: start.row + 1,
|
||||
range_end_line: end_row_as_line(start, end),
|
||||
signature: summary,
|
||||
error: kind == ChunkKind::Error,
|
||||
groupable: kind.traits().groupable,
|
||||
has_leading_comment: false,
|
||||
force_recurse: kind.traits().container || (recurse.is_some() && node.has_error()),
|
||||
region_node: recurse.map(|spec| spec.node),
|
||||
injected: None,
|
||||
recurse,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn group_candidate<'tree>(
|
||||
node: Node<'tree>,
|
||||
kind: ChunkKind,
|
||||
source: &str,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
make_candidate(node, kind, None, NameStyle::Group, None, None, source)
|
||||
}
|
||||
|
||||
pub fn positional_candidate<'tree>(
|
||||
node: Node<'tree>,
|
||||
kind: ChunkKind,
|
||||
source: &str,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
make_candidate(node, kind, None, NameStyle::Named, None, None, source)
|
||||
}
|
||||
|
||||
pub fn named_candidate<'tree>(
|
||||
node: Node<'tree>,
|
||||
kind: ChunkKind,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'tree>>,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
make_kind_chunk(node, kind, extract_identifier(node, source), source, recurse)
|
||||
}
|
||||
|
||||
pub fn container_candidate<'tree>(
|
||||
node: Node<'tree>,
|
||||
kind: ChunkKind,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'tree>>,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
make_kind_chunk(node, kind, extract_identifier(node, source), source, recurse)
|
||||
}
|
||||
|
||||
pub fn make_kind_chunk<'tree>(
|
||||
node: Node<'tree>,
|
||||
kind: ChunkKind,
|
||||
identifier: Option<String>,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'tree>>,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
make_candidate(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse,
|
||||
source,
|
||||
)
|
||||
}
|
||||
|
||||
pub fn make_kind_chunk_from<'tree>(
|
||||
range_node: Node<'tree>,
|
||||
signature_node: Node<'tree>,
|
||||
kind: ChunkKind,
|
||||
identifier: Option<String>,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'tree>>,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
make_candidate(
|
||||
range_node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(signature_node, source),
|
||||
recurse,
|
||||
source,
|
||||
)
|
||||
}
|
||||
|
||||
pub fn make_container_chunk<'tree>(
|
||||
node: Node<'tree>,
|
||||
kind: ChunkKind,
|
||||
identifier: Option<String>,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'tree>>,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
make_candidate(
|
||||
node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(node, source),
|
||||
recurse,
|
||||
source,
|
||||
)
|
||||
}
|
||||
|
||||
pub fn make_container_chunk_from<'tree>(
|
||||
range_node: Node<'tree>,
|
||||
signature_node: Node<'tree>,
|
||||
kind: ChunkKind,
|
||||
identifier: Option<String>,
|
||||
source: &str,
|
||||
recurse: Option<RecurseSpec<'tree>>,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
make_candidate(
|
||||
range_node,
|
||||
kind,
|
||||
identifier,
|
||||
NameStyle::Named,
|
||||
signature_for_node(signature_node, source),
|
||||
recurse,
|
||||
source,
|
||||
)
|
||||
}
|
||||
|
||||
/// Derive a "`prefix_identifier`" name from a node.
|
||||
pub fn prefixed_name(prefix: &str, node: Node<'_>, source: &str) -> String {
|
||||
let identifier = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
format!("{prefix}_{identifier}")
|
||||
}
|
||||
|
||||
pub const fn with_region_node<'tree>(
|
||||
mut candidate: RawChunkCandidate<'tree>,
|
||||
region_node: Option<Node<'tree>>,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
candidate.region_node = region_node;
|
||||
candidate
|
||||
}
|
||||
|
||||
pub const fn with_injected_subtree<'tree>(
|
||||
mut candidate: RawChunkCandidate<'tree>,
|
||||
language: SupportLang,
|
||||
content_node: Node<'tree>,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
candidate.injected = Some(InjectedChunkSpec { language, content_node });
|
||||
candidate
|
||||
}
|
||||
|
||||
pub const fn embedded_selector_token(language: SupportLang) -> &'static str {
|
||||
match language {
|
||||
SupportLang::Bash => "bash",
|
||||
SupportLang::C => "c",
|
||||
SupportLang::Cmake => "cmake",
|
||||
SupportLang::Cpp => "cpp",
|
||||
SupportLang::CSharp => "cs",
|
||||
SupportLang::Css => "css",
|
||||
SupportLang::Go => "go",
|
||||
SupportLang::Html => "html",
|
||||
SupportLang::Java => "java",
|
||||
SupportLang::JavaScript => "js",
|
||||
SupportLang::Json => "json",
|
||||
SupportLang::Kotlin => "kt",
|
||||
SupportLang::Lua => "lua",
|
||||
SupportLang::Markdown => "md",
|
||||
SupportLang::Php => "php",
|
||||
SupportLang::Python => "py",
|
||||
SupportLang::Ruby => "rb",
|
||||
SupportLang::Rust => "rust",
|
||||
SupportLang::Scala => "scala",
|
||||
SupportLang::Sql => "sql",
|
||||
SupportLang::Swift => "swift",
|
||||
SupportLang::Toml => "toml",
|
||||
SupportLang::Tsx => "tsx",
|
||||
SupportLang::TypeScript => "ts",
|
||||
SupportLang::Yaml => "yaml",
|
||||
_ => language.canonical_name(),
|
||||
}
|
||||
}
|
||||
|
||||
// ── Inferred / catch-all candidates ──────────────────────────────────────
|
||||
|
||||
/// Derive a semantic name from a node's kind and/or identifier.
|
||||
pub fn infer_named_candidate<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
|
||||
let kind_name = sanitize_node_kind(node.kind());
|
||||
let kind = ChunkKind::from_sanitized_kind(kind_name);
|
||||
auto_classify(node, kind, source)
|
||||
}
|
||||
|
||||
pub fn auto_classify<'tree>(
|
||||
node: Node<'tree>,
|
||||
kind: ChunkKind,
|
||||
source: &str,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
make_kind_chunk(
|
||||
node,
|
||||
kind,
|
||||
extract_identifier(node, source),
|
||||
source,
|
||||
auto_recurse_for_kind(node, kind),
|
||||
)
|
||||
}
|
||||
|
||||
// ── Tree navigation helpers ──────────────────────────────────────────────
|
||||
|
||||
pub fn named_children(node: Node<'_>) -> Vec<Node<'_>> {
|
||||
let mut children = Vec::new();
|
||||
for index in 0..node.child_count() {
|
||||
if let Some(child) = node.child(index)
|
||||
&& (child.is_named() || child.is_error() || child.kind() == "ERROR")
|
||||
{
|
||||
children.push(child);
|
||||
}
|
||||
}
|
||||
children
|
||||
}
|
||||
|
||||
pub fn child_by_kind<'tree>(node: Node<'tree>, kinds: &[&str]) -> Option<Node<'tree>> {
|
||||
named_children(node)
|
||||
.into_iter()
|
||||
.find(|child| kinds.iter().any(|kind| child.kind() == *kind))
|
||||
}
|
||||
|
||||
pub fn child_by_field_or_kind<'tree>(
|
||||
node: Node<'tree>,
|
||||
fields: &[&str],
|
||||
kinds: &[&str],
|
||||
) -> Option<Node<'tree>> {
|
||||
for field in fields {
|
||||
if let Some(child) = node.child_by_field_name(field) {
|
||||
return Some(child);
|
||||
}
|
||||
}
|
||||
child_by_kind(node, kinds)
|
||||
}
|
||||
|
||||
pub fn resolve_recurse(node: Node<'_>, context: ChunkContext) -> Option<RecurseSpec<'_>> {
|
||||
shape::recurse_target(node).map(|child| RecurseSpec { node: child, context })
|
||||
}
|
||||
|
||||
pub fn resolve_value_container(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
shape::value_container_target(node)
|
||||
.map(|child| RecurseSpec { node: child, context: ChunkContext::ClassBody })
|
||||
}
|
||||
|
||||
fn auto_recurse_for_kind(node: Node<'_>, kind: ChunkKind) -> Option<RecurseSpec<'_>> {
|
||||
let context = match kind {
|
||||
ChunkKind::Constructor
|
||||
| ChunkKind::Function
|
||||
| ChunkKind::Macro
|
||||
| ChunkKind::Method
|
||||
| ChunkKind::Proc
|
||||
| ChunkKind::Recipe => ChunkContext::FunctionBody,
|
||||
_ if kind.traits().container => ChunkContext::ClassBody,
|
||||
_ => return None,
|
||||
};
|
||||
|
||||
resolve_recurse(node, context)
|
||||
}
|
||||
|
||||
pub fn compute_body_inner_boundaries(
|
||||
source: &str,
|
||||
body_start: usize,
|
||||
body_end: usize,
|
||||
) -> (usize, usize) {
|
||||
let bounded_start = body_start.min(source.len());
|
||||
let bounded_end = body_end.min(source.len()).max(bounded_start);
|
||||
let slice = &source[bounded_start..bounded_end];
|
||||
|
||||
let Some((first_non_ws_rel, first_non_ws)) = slice
|
||||
.char_indices()
|
||||
.find(|(_, ch)| !matches!(ch, ' ' | '\t' | '\n' | '\r'))
|
||||
else {
|
||||
return (bounded_start, bounded_end);
|
||||
};
|
||||
let Some((last_non_ws_rel, last_non_ws)) = slice
|
||||
.char_indices()
|
||||
.rev()
|
||||
.find(|(_, ch)| !matches!(ch, ' ' | '\t' | '\n' | '\r'))
|
||||
else {
|
||||
return (bounded_start, bounded_end);
|
||||
};
|
||||
|
||||
let has_delimiters = matches!((first_non_ws, last_non_ws), ('{', '}') | ('(', ')') | ('[', ']'));
|
||||
if !has_delimiters {
|
||||
let line_start = source[..bounded_start].rfind('\n').map_or(0, |pos| pos + 1);
|
||||
let leading_indent = &source[line_start..bounded_start];
|
||||
if !leading_indent.is_empty() && leading_indent.chars().all(|ch| matches!(ch, ' ' | '\t')) {
|
||||
let mut inner_end = bounded_end;
|
||||
let trailing = &source[bounded_end..];
|
||||
if let Some(rel_newline) = trailing.find('\n') {
|
||||
if trailing[..rel_newline]
|
||||
.chars()
|
||||
.all(|ch| matches!(ch, ' ' | '\t' | '\r'))
|
||||
{
|
||||
inner_end = bounded_end + rel_newline + 1;
|
||||
}
|
||||
} else if trailing.chars().all(|ch| matches!(ch, ' ' | '\t' | '\r')) {
|
||||
inner_end = source.len();
|
||||
}
|
||||
return (line_start, inner_end);
|
||||
}
|
||||
return (bounded_start, bounded_end);
|
||||
}
|
||||
|
||||
let mut inner_start = bounded_start + first_non_ws_rel + first_non_ws.len_utf8();
|
||||
if source[inner_start..].starts_with("\r\n") {
|
||||
inner_start += 2;
|
||||
} else if source[inner_start..].starts_with('\n') {
|
||||
inner_start += 1;
|
||||
}
|
||||
|
||||
// Epilogue starts at the beginning of the line containing the closing
|
||||
// delimiter so that the closing line's indentation is part of the epilogue,
|
||||
// not the body.
|
||||
let close_abs = bounded_start + last_non_ws_rel;
|
||||
let inner_end = source[..close_abs]
|
||||
.rfind('\n')
|
||||
.map_or(close_abs, |nl| nl + 1);
|
||||
(inner_start.min(bounded_end), inner_end.max(inner_start).min(bounded_end))
|
||||
}
|
||||
|
||||
// ── Recurse helpers ──────────────────────────────────────────────────────
|
||||
|
||||
pub fn recurse_into<'tree>(
|
||||
node: Node<'tree>,
|
||||
context: ChunkContext,
|
||||
fields: &[&str],
|
||||
kinds: &[&str],
|
||||
) -> Option<RecurseSpec<'tree>> {
|
||||
child_by_field_or_kind(node, fields, kinds).map(|child| RecurseSpec { node: child, context })
|
||||
}
|
||||
|
||||
pub const fn recurse_self(node: Node<'_>, context: ChunkContext) -> RecurseSpec<'_> {
|
||||
RecurseSpec { node, context }
|
||||
}
|
||||
|
||||
pub fn recurse_body(node: Node<'_>, context: ChunkContext) -> Option<RecurseSpec<'_>> {
|
||||
resolve_recurse(node, context)
|
||||
}
|
||||
|
||||
pub fn recurse_class(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
resolve_recurse(node, ChunkContext::ClassBody)
|
||||
}
|
||||
|
||||
pub fn recurse_enum(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
resolve_recurse(node, ChunkContext::ClassBody)
|
||||
}
|
||||
|
||||
pub fn recurse_value_container(node: Node<'_>) -> Option<RecurseSpec<'_>> {
|
||||
resolve_value_container(node)
|
||||
}
|
||||
|
||||
/// Try to promote a node that wraps a call expression with a trailing
|
||||
/// callback/block argument. Returns a named chunk candidate with `recurse`
|
||||
/// pointing into the callback body.
|
||||
///
|
||||
/// This is language-agnostic: it uses structural shape detection to find
|
||||
/// call-with-callback patterns in any language (JS `describe(...)`, Go
|
||||
/// `t.Run(...)`, Rust `tokio::spawn(async { ... })`, etc.).
|
||||
pub fn try_promote_call_with_callback<'tree>(
|
||||
node: Node<'tree>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'tree>> {
|
||||
let (func_node, body) = shape::trailing_callback_body(node)?;
|
||||
|
||||
// Extract a name from the call target (e.g. `describe`, `describe.serial`,
|
||||
// `app.use`). Sanitize the raw source text (dots become underscores).
|
||||
let name = sanitize_identifier(node_text(source, func_node.start_byte(), func_node.end_byte()));
|
||||
|
||||
let recurse = Some(RecurseSpec { node: body, context: ChunkContext::FunctionBody });
|
||||
|
||||
Some(make_kind_chunk(node, ChunkKind::Expression, name, source, recurse))
|
||||
}
|
||||
|
||||
// ── Identifier extraction ────────────────────────────────────────────────
|
||||
|
||||
pub fn extract_identifier(node: Node<'_>, source: &str) -> Option<String> {
|
||||
if node.kind() == "constructor" {
|
||||
return Some("constructor".to_string());
|
||||
}
|
||||
|
||||
if let Some(name_node) = shape::identifier_node(node) {
|
||||
return sanitize_identifier(node_text(source, name_node.start_byte(), name_node.end_byte()));
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
|
||||
/// Extract the name of a single-declarator binding like `const FOO = ...`.
|
||||
pub fn extract_single_declarator_name(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let declarators: Vec<Node<'_>> = named_children(node)
|
||||
.into_iter()
|
||||
.filter(|c| c.kind() == "variable_declarator")
|
||||
.collect();
|
||||
if declarators.len() != 1 {
|
||||
return None;
|
||||
}
|
||||
extract_identifier(declarators[0], source)
|
||||
}
|
||||
|
||||
// ── Text helpers ─────────────────────────────────────────────────────────
|
||||
|
||||
pub fn node_text(source: &str, start_byte: usize, end_byte: usize) -> &str {
|
||||
source.get(start_byte..end_byte).unwrap_or("")
|
||||
}
|
||||
|
||||
pub fn sanitize_identifier(text: &str) -> Option<String> {
|
||||
let mut out = String::new();
|
||||
let mut previous_was_underscore = false;
|
||||
|
||||
for ch in text.chars() {
|
||||
if ch.is_alphanumeric() || ch == '_' || ch == '$' {
|
||||
out.push(ch);
|
||||
previous_was_underscore = false;
|
||||
continue;
|
||||
}
|
||||
|
||||
if !previous_was_underscore {
|
||||
out.push('_');
|
||||
previous_was_underscore = true;
|
||||
}
|
||||
}
|
||||
|
||||
let sanitized = out.trim_matches('_').to_string();
|
||||
if sanitized.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(sanitized)
|
||||
}
|
||||
}
|
||||
|
||||
pub fn unquote_text(text: &str) -> String {
|
||||
text.trim().trim_matches('"').trim_matches('\'').to_string()
|
||||
}
|
||||
|
||||
pub fn sanitize_node_kind(kind: &str) -> &str {
|
||||
let mut kind_stripped = kind;
|
||||
for suffix in [
|
||||
"_instruction",
|
||||
"_statement",
|
||||
"_declaration",
|
||||
"_definition",
|
||||
"_item",
|
||||
"ession", // _expression -> _expr
|
||||
] {
|
||||
if let Some(stripped) = kind_stripped.strip_suffix(suffix) {
|
||||
kind_stripped = stripped;
|
||||
}
|
||||
}
|
||||
if kind_stripped.is_empty() {
|
||||
kind
|
||||
} else {
|
||||
kind_stripped
|
||||
}
|
||||
}
|
||||
|
||||
pub fn normalized_header(source: &str, start_byte: usize, end_byte: usize) -> String {
|
||||
let slice = node_text(source, start_byte, end_byte);
|
||||
let mut header = String::new();
|
||||
|
||||
for line in slice.lines().take(4) {
|
||||
let trimmed = line.trim();
|
||||
if trimmed.is_empty() {
|
||||
continue;
|
||||
}
|
||||
if !header.is_empty() {
|
||||
header.push(' ');
|
||||
}
|
||||
header.push_str(trimmed);
|
||||
if trimmed.contains('{') || trimmed.ends_with(';') {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
collapse_whitespace(header.as_str())
|
||||
}
|
||||
|
||||
pub fn collapse_whitespace(text: &str) -> String {
|
||||
let mut out = String::new();
|
||||
let mut pending_space = false;
|
||||
for ch in text.chars() {
|
||||
if ch.is_whitespace() {
|
||||
pending_space = true;
|
||||
continue;
|
||||
}
|
||||
if pending_space && !out.is_empty() {
|
||||
out.push(' ');
|
||||
}
|
||||
out.push(ch);
|
||||
pending_space = false;
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
// ── Signature helpers ────────────────────────────────────────────────────
|
||||
|
||||
pub fn signature_for_node(node: Node<'_>, source: &str) -> Option<String> {
|
||||
let raw = if let Some(end_byte) = shape::signature_end_byte(node) {
|
||||
node_text(source, node.start_byte(), end_byte)
|
||||
} else {
|
||||
node_text(source, node.start_byte(), node.end_byte())
|
||||
};
|
||||
|
||||
let sig = collapse_whitespace(raw.trim());
|
||||
let sig = sig
|
||||
.trim_end_matches('{')
|
||||
.trim_end_matches(':')
|
||||
.trim_end_matches(';')
|
||||
.trim();
|
||||
if sig.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(sig.to_string())
|
||||
}
|
||||
}
|
||||
|
||||
// ── Summary / canonical naming ───────────────────────────────────────────
|
||||
|
||||
fn normalize_summary_text(summary: &str) -> Option<String> {
|
||||
let summary = collapse_whitespace(summary.trim())
|
||||
.trim_end_matches('{')
|
||||
.trim_end_matches(':')
|
||||
.trim_end_matches(';')
|
||||
.trim()
|
||||
.to_string();
|
||||
if summary.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(summary)
|
||||
}
|
||||
}
|
||||
|
||||
fn summarize_function_node(
|
||||
kind: ChunkKind,
|
||||
identifier: Option<&str>,
|
||||
raw_signature: &str,
|
||||
) -> String {
|
||||
let name = identifier.unwrap_or_else(|| kind.prefix());
|
||||
let tail = go_method_signature(raw_signature, identifier)
|
||||
.or_else(|| function_signature(raw_signature))
|
||||
.or_else(|| python_function_signature(raw_signature))
|
||||
.or_else(|| rust_function_signature(raw_signature))
|
||||
.unwrap_or_else(|| raw_signature.to_string());
|
||||
let tail = tail.replacen("): ", ") → ", 1);
|
||||
format!("{} {name}{tail}", kind.prefix())
|
||||
}
|
||||
|
||||
fn summarize_variable_node(
|
||||
node: Node<'_>,
|
||||
kind: ChunkKind,
|
||||
identifier: Option<&str>,
|
||||
source: &str,
|
||||
) -> Option<String> {
|
||||
let header = normalized_header(source, node.start_byte(), node.end_byte());
|
||||
let keyword = header.split_whitespace().next()?;
|
||||
let name = identifier.unwrap_or_else(|| kind.prefix());
|
||||
Some(format!("{keyword} {name}"))
|
||||
}
|
||||
|
||||
fn summarize_statement_node(node: Node<'_>, source: &str) -> Option<String> {
|
||||
normalize_summary_text(normalized_header(source, node.start_byte(), node.end_byte()).as_str())
|
||||
}
|
||||
|
||||
pub fn summary_for_node(
|
||||
node: Node<'_>,
|
||||
kind: ChunkKind,
|
||||
identifier: Option<&str>,
|
||||
raw_signature: Option<&str>,
|
||||
source: &str,
|
||||
) -> Option<String> {
|
||||
match kind.traits().summary {
|
||||
SummaryStyle::Imports => return Some("imports".to_string()),
|
||||
SummaryStyle::Function => {
|
||||
if let Some(signature) = raw_signature {
|
||||
return Some(summarize_function_node(kind, identifier, signature));
|
||||
}
|
||||
},
|
||||
SummaryStyle::Variable => {
|
||||
return summarize_variable_node(node, kind, identifier, source);
|
||||
},
|
||||
SummaryStyle::Default => {},
|
||||
}
|
||||
if matches!(
|
||||
node.kind(),
|
||||
"for_statement"
|
||||
| "for_in_statement"
|
||||
| "for_of_statement"
|
||||
| "if_statement"
|
||||
| "return_statement"
|
||||
| "expression_statement"
|
||||
| "call_expression"
|
||||
| "call"
|
||||
| "function_call"
|
||||
) {
|
||||
return summarize_statement_node(node, source);
|
||||
}
|
||||
raw_signature
|
||||
.and_then(normalize_summary_text)
|
||||
.or_else(|| summarize_statement_node(node, source))
|
||||
}
|
||||
|
||||
fn function_signature(header: &str) -> Option<String> {
|
||||
let start = header.find('(')?;
|
||||
let end = header.rfind('{').unwrap_or(header.len());
|
||||
let signature = header.get(start..end)?.trim().trim_end_matches(';').trim();
|
||||
if signature.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(signature.to_string())
|
||||
}
|
||||
}
|
||||
|
||||
fn go_method_signature(header: &str, identifier: Option<&str>) -> Option<String> {
|
||||
let name = identifier?;
|
||||
let declaration = header
|
||||
.trim()
|
||||
.trim_end_matches('{')
|
||||
.trim_end_matches(';')
|
||||
.trim();
|
||||
let rest = declaration.strip_prefix("func")?.trim_start();
|
||||
if !rest.starts_with('(') {
|
||||
return None;
|
||||
}
|
||||
let receiver_end = find_matching_paren(rest, 0)?;
|
||||
let after_receiver = rest.get(receiver_end + 1..)?.trim_start();
|
||||
let after_name = after_receiver.strip_prefix(name)?.trim_start();
|
||||
if !after_name.starts_with('(') {
|
||||
return None;
|
||||
}
|
||||
Some(after_name.to_string())
|
||||
}
|
||||
|
||||
fn find_matching_paren(text: &str, open_index: usize) -> Option<usize> {
|
||||
let mut depth = 0usize;
|
||||
for (index, ch) in text
|
||||
.char_indices()
|
||||
.skip_while(|(index, _)| *index < open_index)
|
||||
{
|
||||
match ch {
|
||||
'(' => depth = depth.saturating_add(1),
|
||||
')' => {
|
||||
depth = depth.checked_sub(1)?;
|
||||
if depth == 0 {
|
||||
return Some(index);
|
||||
}
|
||||
},
|
||||
_ => {},
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn python_function_signature(header: &str) -> Option<String> {
|
||||
let start = header.find('(')?;
|
||||
let end = header.rfind(':').unwrap_or(header.len());
|
||||
let signature = header.get(start..end)?.trim();
|
||||
if signature.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(signature.to_string())
|
||||
}
|
||||
}
|
||||
|
||||
fn rust_function_signature(header: &str) -> Option<String> {
|
||||
let start = header.find('(')?;
|
||||
let end = header.rfind('{').unwrap_or(header.len());
|
||||
let mut sig = header.get(start..end)?.trim();
|
||||
if let Some(idx) = sig.find(" where ") {
|
||||
sig = sig.get(..idx)?.trim();
|
||||
}
|
||||
if sig.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(sig.to_string())
|
||||
}
|
||||
}
|
||||
|
||||
// ── Trivia and attribute detection ───────────────────────────────────────
|
||||
|
||||
pub fn is_trivia_node(node: Node<'_>) -> bool {
|
||||
shape::is_generic_trivia(node)
|
||||
}
|
||||
|
||||
pub fn is_absorbable_attribute(kind: &str) -> bool {
|
||||
shape::is_generic_absorbable_attr(kind)
|
||||
}
|
||||
|
||||
// ── Other helpers ────────────────────────────────────────────────────────
|
||||
|
||||
pub fn looks_like_python_statement(node: Node<'_>, source: &str) -> bool {
|
||||
let header = normalized_header(source, node.start_byte(), node.end_byte());
|
||||
header.contains(':') && !header.contains('{')
|
||||
}
|
||||
|
||||
pub fn detect_indent(source: &str, start_byte: usize) -> (u32, String) {
|
||||
let line_start = source.as_bytes()[..start_byte]
|
||||
.iter()
|
||||
.rposition(|&b| b == b'\n')
|
||||
.map_or(0, |pos| pos + 1);
|
||||
let line_prefix = &source[line_start..start_byte];
|
||||
let mut cols = 0u32;
|
||||
let mut ch = String::new();
|
||||
for byte in line_prefix.bytes() {
|
||||
match byte {
|
||||
b'\t' => {
|
||||
cols += 1;
|
||||
if ch.is_empty() {
|
||||
ch = "\t".to_string();
|
||||
}
|
||||
},
|
||||
b' ' => {
|
||||
cols += 1;
|
||||
if ch.is_empty() {
|
||||
ch = " ".to_string();
|
||||
}
|
||||
},
|
||||
_ => break,
|
||||
}
|
||||
}
|
||||
(cols, ch)
|
||||
}
|
||||
|
||||
pub fn is_root_wrapper_node(node: Node<'_>) -> bool {
|
||||
shape::is_root_wrapper_node(node)
|
||||
}
|
||||
|
||||
pub const fn line_span(start_line: usize, end_line: usize) -> usize {
|
||||
end_line.saturating_sub(start_line) + 1
|
||||
}
|
||||
|
||||
pub fn total_line_count(source: &str) -> usize {
|
||||
if source.is_empty() {
|
||||
0
|
||||
} else {
|
||||
source.bytes().filter(|byte| *byte == b'\n').count() + 1
|
||||
}
|
||||
}
|
||||
|
||||
pub fn first_scalar_child(node: Node<'_>) -> Option<Node<'_>> {
|
||||
named_children(node).into_iter().find(|child| {
|
||||
!matches!(
|
||||
child.kind(),
|
||||
"block_node"
|
||||
| "flow_node"
|
||||
| "block_mapping"
|
||||
| "flow_mapping"
|
||||
| "block_sequence"
|
||||
| "flow_sequence"
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::compute_body_inner_boundaries;
|
||||
|
||||
#[test]
|
||||
fn compute_body_inner_boundaries_handles_brace_and_indent_bodies() {
|
||||
let ts = "function main() {\n\treturn 1;\n}\n";
|
||||
let ts_start = ts.find('{').expect("open brace");
|
||||
let ts_end = ts.rfind('}').expect("close brace") + 1;
|
||||
let (ts_inner_start, ts_inner_end) = compute_body_inner_boundaries(ts, ts_start, ts_end);
|
||||
assert_eq!(&ts[ts_inner_start..ts_inner_end], "\treturn 1;\n");
|
||||
|
||||
let rust = "fn main() {\n println!(\"hi\");\n}\n";
|
||||
let rust_start = rust.find('{').expect("open brace");
|
||||
let rust_end = rust.rfind('}').expect("close brace") + 1;
|
||||
let (rust_inner_start, rust_inner_end) =
|
||||
compute_body_inner_boundaries(rust, rust_start, rust_end);
|
||||
assert_eq!(&rust[rust_inner_start..rust_inner_end], " println!(\"hi\");\n");
|
||||
|
||||
let go = "func main() {\n\treturn\n}\n";
|
||||
let go_start = go.find('{').expect("open brace");
|
||||
let go_end = go.rfind('}').expect("close brace") + 1;
|
||||
let (go_inner_start, go_inner_end) = compute_body_inner_boundaries(go, go_start, go_end);
|
||||
assert_eq!(&go[go_inner_start..go_inner_end], "\treturn\n");
|
||||
|
||||
let py = "def main():\n return 1\n";
|
||||
let py_body_start = py.find(" return 1").expect("body start");
|
||||
let py_body_end = py_body_start + " return 1".len();
|
||||
let (py_inner_start, py_inner_end) =
|
||||
compute_body_inner_boundaries(py, py_body_start, py_body_end);
|
||||
assert_eq!(&py[py_inner_start..py_inner_end], " return 1");
|
||||
}
|
||||
}
|
||||
@@ -1,690 +0,0 @@
|
||||
use std::collections::{HashMap, HashSet};
|
||||
|
||||
use super::{
|
||||
chunk_checksum,
|
||||
common::{detect_indent, total_line_count},
|
||||
kind::ChunkKind,
|
||||
line_start_offsets,
|
||||
state::ConflictMeta,
|
||||
types::{ChunkNode, ChunkTree},
|
||||
};
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct ConflictRegion {
|
||||
pub ours_start_line: usize,
|
||||
pub ours_end_line: usize,
|
||||
pub theirs_start_line: usize,
|
||||
pub theirs_end_line: usize,
|
||||
pub marker_start_line: usize,
|
||||
pub marker_end_line: usize,
|
||||
pub ours_content: String,
|
||||
pub theirs_content: String,
|
||||
pub base_content: Option<String>,
|
||||
pub base_label: Option<String>,
|
||||
pub ours_label: String,
|
||||
pub theirs_label: String,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct CleanResult {
|
||||
pub source: String,
|
||||
pub conflicts: Vec<ConflictRegion>,
|
||||
pub ours_byte_ranges: Vec<(usize, usize)>,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
struct PendingConflict {
|
||||
path: Option<String>,
|
||||
parent_path: Option<String>,
|
||||
ours_start_byte: usize,
|
||||
ours_end_byte: usize,
|
||||
ours_content: String,
|
||||
theirs_content: String,
|
||||
base_content: Option<String>,
|
||||
base_label: Option<String>,
|
||||
ours_label: String,
|
||||
theirs_label: String,
|
||||
}
|
||||
|
||||
pub fn has_conflict_markers(source: &str) -> bool {
|
||||
source.contains("<<<<<<<") && source.contains("=======") && source.contains(">>>>>>>")
|
||||
}
|
||||
|
||||
pub fn detect_conflicts(source: &str) -> Vec<ConflictRegion> {
|
||||
let lines = source_lines(source);
|
||||
let mut conflicts = Vec::new();
|
||||
let mut index = 0usize;
|
||||
|
||||
while index < lines.len() {
|
||||
let Some(ours_label) = marker_label(lines[index], "<<<<<<<") else {
|
||||
index += 1;
|
||||
continue;
|
||||
};
|
||||
let marker_start_line = index + 1;
|
||||
let ours_start_index = index + 1;
|
||||
let mut separator_index = None;
|
||||
let mut base_marker_index = None;
|
||||
let mut base_label = None;
|
||||
let mut cursor = ours_start_index;
|
||||
|
||||
while cursor < lines.len() {
|
||||
if let Some(label) = marker_label(lines[cursor], "|||||||") {
|
||||
base_marker_index = Some(cursor);
|
||||
base_label = Some(label);
|
||||
cursor += 1;
|
||||
while cursor < lines.len() {
|
||||
if is_marker_line(lines[cursor], "=======") {
|
||||
separator_index = Some(cursor);
|
||||
break;
|
||||
}
|
||||
cursor += 1;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if is_marker_line(lines[cursor], "=======") {
|
||||
separator_index = Some(cursor);
|
||||
break;
|
||||
}
|
||||
cursor += 1;
|
||||
}
|
||||
|
||||
let Some(separator_index) = separator_index else {
|
||||
index += 1;
|
||||
continue;
|
||||
};
|
||||
|
||||
let theirs_start_index = separator_index + 1;
|
||||
let mut end_index = theirs_start_index;
|
||||
while end_index < lines.len() && marker_label(lines[end_index], ">>>>>>>").is_none() {
|
||||
end_index += 1;
|
||||
}
|
||||
let Some(theirs_label) = lines
|
||||
.get(end_index)
|
||||
.and_then(|line| marker_label(line, ">>>>>>>"))
|
||||
else {
|
||||
index += 1;
|
||||
continue;
|
||||
};
|
||||
|
||||
let ours_end_index = base_marker_index.unwrap_or(separator_index);
|
||||
let base_start_index = base_marker_index.map_or(separator_index, |value| value + 1);
|
||||
let base_end_index = separator_index;
|
||||
|
||||
let (ours_start_line, ours_end_line) = content_line_range(ours_start_index, ours_end_index);
|
||||
let (theirs_start_line, theirs_end_line) = content_line_range(theirs_start_index, end_index);
|
||||
|
||||
conflicts.push(ConflictRegion {
|
||||
ours_start_line,
|
||||
ours_end_line,
|
||||
theirs_start_line,
|
||||
theirs_end_line,
|
||||
marker_start_line,
|
||||
marker_end_line: end_index + 1,
|
||||
ours_content: join_lines(&lines[ours_start_index..ours_end_index]),
|
||||
theirs_content: join_lines(&lines[theirs_start_index..end_index]),
|
||||
base_content: base_marker_index
|
||||
.map(|_| join_lines(&lines[base_start_index..base_end_index])),
|
||||
base_label,
|
||||
ours_label,
|
||||
theirs_label,
|
||||
});
|
||||
index = end_index + 1;
|
||||
}
|
||||
|
||||
conflicts
|
||||
}
|
||||
|
||||
pub fn accept_ours(source: &str, conflicts: &[ConflictRegion]) -> CleanResult {
|
||||
if conflicts.is_empty() {
|
||||
return CleanResult {
|
||||
source: source.to_owned(),
|
||||
conflicts: Vec::new(),
|
||||
ours_byte_ranges: Vec::new(),
|
||||
};
|
||||
}
|
||||
|
||||
let lines = source_lines(source);
|
||||
let mut clean_source = String::with_capacity(source.len());
|
||||
let mut ours_byte_ranges = Vec::with_capacity(conflicts.len());
|
||||
let mut cursor_line = 1usize;
|
||||
|
||||
for conflict in conflicts {
|
||||
let marker_start = conflict.marker_start_line.saturating_sub(1);
|
||||
for line in lines
|
||||
.iter()
|
||||
.take(marker_start)
|
||||
.skip(cursor_line.saturating_sub(1))
|
||||
{
|
||||
clean_source.push_str(line);
|
||||
}
|
||||
let ours_start_byte = clean_source.len();
|
||||
let ours_start_index = conflict.ours_start_line.saturating_sub(1);
|
||||
let ours_end_index = if conflict.ours_start_line <= conflict.ours_end_line {
|
||||
conflict.ours_end_line
|
||||
} else {
|
||||
ours_start_index
|
||||
};
|
||||
for line in lines.iter().take(ours_end_index).skip(ours_start_index) {
|
||||
clean_source.push_str(line);
|
||||
}
|
||||
ours_byte_ranges.push((ours_start_byte, clean_source.len()));
|
||||
cursor_line = conflict.marker_end_line + 1;
|
||||
}
|
||||
|
||||
for line in lines.iter().skip(cursor_line.saturating_sub(1)) {
|
||||
clean_source.push_str(line);
|
||||
}
|
||||
|
||||
CleanResult { source: clean_source, conflicts: conflicts.to_vec(), ours_byte_ranges }
|
||||
}
|
||||
|
||||
pub fn reconstruct_markers(
|
||||
clean_source: &str,
|
||||
conflict_meta: &HashMap<String, ConflictMeta>,
|
||||
) -> String {
|
||||
if conflict_meta.is_empty() {
|
||||
return clean_source.to_owned();
|
||||
}
|
||||
|
||||
let mut conflicts = conflict_meta.iter().collect::<Vec<_>>();
|
||||
conflicts.sort_unstable_by_key(|(_, meta)| meta.ours_start_byte);
|
||||
|
||||
let mut rendered = String::with_capacity(clean_source.len() + conflict_meta.len() * 64);
|
||||
let mut cursor = 0usize;
|
||||
for (_, meta) in conflicts {
|
||||
if meta.ours_start_byte > clean_source.len()
|
||||
|| meta.ours_end_byte > clean_source.len()
|
||||
|| meta.ours_start_byte > meta.ours_end_byte
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
rendered.push_str(&clean_source[cursor..meta.ours_start_byte]);
|
||||
push_marker_line(&mut rendered, "<<<<<<<", Some(meta.ours_label.as_str()));
|
||||
push_conflict_content(&mut rendered, &clean_source[meta.ours_start_byte..meta.ours_end_byte]);
|
||||
if let Some(base_content) = meta.base_content.as_deref() {
|
||||
push_marker_line(&mut rendered, "|||||||", meta.base_label.as_deref());
|
||||
push_conflict_content(&mut rendered, base_content);
|
||||
}
|
||||
push_marker_line(&mut rendered, "=======", None);
|
||||
push_conflict_content(&mut rendered, meta.theirs_content.as_str());
|
||||
push_marker_line(&mut rendered, ">>>>>>>", Some(meta.theirs_label.as_str()));
|
||||
cursor = meta.ours_end_byte;
|
||||
}
|
||||
rendered.push_str(&clean_source[cursor..]);
|
||||
rendered
|
||||
}
|
||||
|
||||
pub fn inject_conflict_chunks(
|
||||
tree: &mut ChunkTree,
|
||||
source: &str,
|
||||
clean_result: &CleanResult,
|
||||
) -> HashMap<String, ConflictMeta> {
|
||||
let mut pending = Vec::with_capacity(clean_result.conflicts.len());
|
||||
for (conflict, (ours_start_byte, ours_end_byte)) in clean_result
|
||||
.conflicts
|
||||
.iter()
|
||||
.zip(clean_result.ours_byte_ranges.iter().copied())
|
||||
{
|
||||
pending.push(PendingConflict {
|
||||
path: None,
|
||||
parent_path: None,
|
||||
ours_start_byte,
|
||||
ours_end_byte,
|
||||
ours_content: conflict.ours_content.clone(),
|
||||
theirs_content: conflict.theirs_content.clone(),
|
||||
base_content: conflict.base_content.clone(),
|
||||
base_label: conflict.base_label.clone(),
|
||||
ours_label: conflict.ours_label.clone(),
|
||||
theirs_label: conflict.theirs_label.clone(),
|
||||
});
|
||||
}
|
||||
inject_pending_conflicts(tree, source, pending)
|
||||
}
|
||||
|
||||
pub fn reinject_conflict_chunks(
|
||||
tree: &mut ChunkTree,
|
||||
source: &str,
|
||||
conflict_meta: &HashMap<String, ConflictMeta>,
|
||||
) -> HashMap<String, ConflictMeta> {
|
||||
let mut pending = conflict_meta
|
||||
.iter()
|
||||
.map(|(path, meta)| PendingConflict {
|
||||
path: Some(path.clone()),
|
||||
parent_path: conflict_parent_path(path),
|
||||
ours_start_byte: meta.ours_start_byte,
|
||||
ours_end_byte: meta.ours_end_byte,
|
||||
ours_content: source
|
||||
.get(meta.ours_start_byte..meta.ours_end_byte)
|
||||
.unwrap_or_default()
|
||||
.to_owned(),
|
||||
theirs_content: meta.theirs_content.clone(),
|
||||
base_content: meta.base_content.clone(),
|
||||
base_label: meta.base_label.clone(),
|
||||
ours_label: meta.ours_label.clone(),
|
||||
theirs_label: meta.theirs_label.clone(),
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
pending.sort_unstable_by_key(|conflict| conflict.ours_start_byte);
|
||||
inject_pending_conflicts(tree, source, pending)
|
||||
}
|
||||
|
||||
fn inject_pending_conflicts(
|
||||
tree: &mut ChunkTree,
|
||||
source: &str,
|
||||
pending: Vec<PendingConflict>,
|
||||
) -> HashMap<String, ConflictMeta> {
|
||||
let mut conflict_meta = HashMap::new();
|
||||
let mut existing_paths = tree
|
||||
.chunks
|
||||
.iter()
|
||||
.map(|chunk| chunk.path.clone())
|
||||
.collect::<HashSet<_>>();
|
||||
let line_starts = line_start_offsets(source);
|
||||
let mut counters = HashMap::<String, usize>::new();
|
||||
|
||||
for pending_conflict in pending {
|
||||
if pending_conflict.ours_start_byte > pending_conflict.ours_end_byte
|
||||
|| pending_conflict.ours_end_byte > source.len()
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
let parent_path = match pending_conflict.parent_path.as_deref() {
|
||||
Some(parent_path) => {
|
||||
if tree.chunks.iter().any(|chunk| chunk.path == parent_path) {
|
||||
parent_path.to_owned()
|
||||
} else {
|
||||
continue;
|
||||
}
|
||||
},
|
||||
None => find_innermost_parent_path(
|
||||
tree,
|
||||
pending_conflict.ours_start_byte,
|
||||
pending_conflict.ours_end_byte,
|
||||
)
|
||||
.unwrap_or_default(),
|
||||
};
|
||||
|
||||
let conflict_path = pending_conflict.path.unwrap_or_else(|| {
|
||||
next_conflict_path(parent_path.as_str(), &mut counters, &existing_paths)
|
||||
});
|
||||
let ours_path = format!("{conflict_path}.ours");
|
||||
let theirs_path = format!("{conflict_path}.theirs");
|
||||
existing_paths.insert(conflict_path.clone());
|
||||
existing_paths.insert(ours_path.clone());
|
||||
existing_paths.insert(theirs_path.clone());
|
||||
|
||||
let (start_line, end_line, line_count) = real_line_stats(
|
||||
&line_starts,
|
||||
pending_conflict.ours_start_byte,
|
||||
pending_conflict.ours_end_byte,
|
||||
);
|
||||
let theirs_line_count = display_line_count(pending_conflict.theirs_content.as_str()) as u32;
|
||||
let (indent, indent_char) =
|
||||
detect_indent(source, pending_conflict.ours_start_byte.min(source.len()));
|
||||
let conflict_identifier = conflict_path
|
||||
.rsplit('.')
|
||||
.next()
|
||||
.and_then(|leaf| leaf.strip_prefix("conflict_"))
|
||||
.map(ToOwned::to_owned);
|
||||
let conflict_checksum = chunk_checksum(
|
||||
format!(
|
||||
"{}\0{}\0{}\0{}\0{}",
|
||||
pending_conflict.ours_label,
|
||||
pending_conflict.theirs_label,
|
||||
pending_conflict.ours_content,
|
||||
pending_conflict.theirs_content,
|
||||
pending_conflict.base_content.as_deref().unwrap_or_default(),
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
let theirs_start_line = start_line;
|
||||
let theirs_end_line = if theirs_line_count == 0 {
|
||||
theirs_start_line
|
||||
} else {
|
||||
theirs_start_line + theirs_line_count - 1
|
||||
};
|
||||
let conflict_end_line = end_line.max(theirs_end_line);
|
||||
let conflict_line_count = if conflict_end_line >= start_line {
|
||||
conflict_end_line - start_line + 1
|
||||
} else {
|
||||
0
|
||||
};
|
||||
|
||||
tree.chunks.push(ChunkNode {
|
||||
path: conflict_path.clone(),
|
||||
identifier: conflict_identifier,
|
||||
kind: ChunkKind::Conflict,
|
||||
leaf: false,
|
||||
virtual_content: None,
|
||||
parent_path: Some(parent_path.clone()),
|
||||
children: vec![ours_path.clone(), theirs_path.clone()],
|
||||
signature: None,
|
||||
start_line,
|
||||
end_line: conflict_end_line,
|
||||
line_count: conflict_line_count,
|
||||
start_byte: pending_conflict.ours_start_byte as u32,
|
||||
end_byte: pending_conflict.ours_end_byte as u32,
|
||||
checksum_start_byte: pending_conflict.ours_start_byte as u32,
|
||||
prologue_end_byte: None,
|
||||
epilogue_start_byte: None,
|
||||
checksum: conflict_checksum,
|
||||
error: false,
|
||||
indent,
|
||||
indent_char: indent_char.clone(),
|
||||
group: false,
|
||||
});
|
||||
tree.chunks.push(ChunkNode {
|
||||
path: ours_path.clone(),
|
||||
identifier: None,
|
||||
kind: ChunkKind::Ours,
|
||||
leaf: true,
|
||||
virtual_content: (pending_conflict.ours_start_byte == pending_conflict.ours_end_byte)
|
||||
.then(String::new),
|
||||
parent_path: Some(conflict_path.clone()),
|
||||
children: Vec::new(),
|
||||
signature: None,
|
||||
start_line,
|
||||
end_line,
|
||||
line_count,
|
||||
start_byte: pending_conflict.ours_start_byte as u32,
|
||||
end_byte: pending_conflict.ours_end_byte as u32,
|
||||
checksum_start_byte: pending_conflict.ours_start_byte as u32,
|
||||
prologue_end_byte: None,
|
||||
epilogue_start_byte: None,
|
||||
checksum: chunk_checksum(
|
||||
source
|
||||
.as_bytes()
|
||||
.get(pending_conflict.ours_start_byte..pending_conflict.ours_end_byte)
|
||||
.unwrap_or_default(),
|
||||
),
|
||||
error: false,
|
||||
indent,
|
||||
indent_char: indent_char.clone(),
|
||||
group: false,
|
||||
});
|
||||
tree.chunks.push(ChunkNode {
|
||||
path: theirs_path.clone(),
|
||||
identifier: None,
|
||||
kind: ChunkKind::Theirs,
|
||||
leaf: true,
|
||||
virtual_content: Some(pending_conflict.theirs_content.clone()),
|
||||
parent_path: Some(conflict_path.clone()),
|
||||
children: Vec::new(),
|
||||
signature: None,
|
||||
start_line: theirs_start_line,
|
||||
end_line: theirs_end_line,
|
||||
line_count: theirs_line_count,
|
||||
start_byte: pending_conflict.ours_start_byte as u32,
|
||||
end_byte: pending_conflict.ours_start_byte as u32,
|
||||
checksum_start_byte: pending_conflict.ours_start_byte as u32,
|
||||
prologue_end_byte: None,
|
||||
epilogue_start_byte: None,
|
||||
checksum: chunk_checksum(pending_conflict.theirs_content.as_bytes()),
|
||||
error: false,
|
||||
indent,
|
||||
indent_char,
|
||||
group: false,
|
||||
});
|
||||
|
||||
insert_conflict_child(
|
||||
tree,
|
||||
parent_path.as_str(),
|
||||
conflict_path.as_str(),
|
||||
pending_conflict.ours_start_byte,
|
||||
pending_conflict.ours_end_byte,
|
||||
);
|
||||
|
||||
conflict_meta.insert(conflict_path, ConflictMeta {
|
||||
theirs_content: pending_conflict.theirs_content,
|
||||
ours_label: pending_conflict.ours_label,
|
||||
theirs_label: pending_conflict.theirs_label,
|
||||
base_content: pending_conflict.base_content,
|
||||
base_label: pending_conflict.base_label,
|
||||
ours_start_byte: pending_conflict.ours_start_byte,
|
||||
ours_end_byte: pending_conflict.ours_end_byte,
|
||||
});
|
||||
}
|
||||
|
||||
conflict_meta
|
||||
}
|
||||
|
||||
fn source_lines(source: &str) -> Vec<&str> {
|
||||
if source.is_empty() {
|
||||
Vec::new()
|
||||
} else {
|
||||
source.split_inclusive('\n').collect()
|
||||
}
|
||||
}
|
||||
|
||||
fn strip_line_ending(line: &str) -> &str {
|
||||
line.trim_end_matches(['\n', '\r'])
|
||||
}
|
||||
|
||||
fn marker_label(line: &str, prefix: &str) -> Option<String> {
|
||||
let stripped = strip_line_ending(line);
|
||||
let remainder = stripped.strip_prefix(prefix)?;
|
||||
Some(remainder.trim_start().to_owned())
|
||||
}
|
||||
|
||||
fn is_marker_line(line: &str, prefix: &str) -> bool {
|
||||
strip_line_ending(line).starts_with(prefix)
|
||||
}
|
||||
|
||||
fn join_lines(lines: &[&str]) -> String {
|
||||
let mut joined = String::new();
|
||||
for line in lines {
|
||||
joined.push_str(line);
|
||||
}
|
||||
joined
|
||||
}
|
||||
|
||||
const fn content_line_range(start_index: usize, end_index: usize) -> (usize, usize) {
|
||||
if start_index < end_index {
|
||||
(start_index + 1, end_index)
|
||||
} else {
|
||||
(start_index + 1, start_index)
|
||||
}
|
||||
}
|
||||
|
||||
fn push_marker_line(out: &mut String, marker: &str, label: Option<&str>) {
|
||||
out.push_str(marker);
|
||||
if let Some(label) = label
|
||||
&& !label.is_empty()
|
||||
{
|
||||
out.push(' ');
|
||||
out.push_str(label);
|
||||
}
|
||||
out.push('\n');
|
||||
}
|
||||
|
||||
fn push_conflict_content(out: &mut String, content: &str) {
|
||||
out.push_str(content);
|
||||
if !content.is_empty() && !content.ends_with('\n') {
|
||||
out.push('\n');
|
||||
}
|
||||
}
|
||||
|
||||
fn find_innermost_parent_path(tree: &ChunkTree, start: usize, end: usize) -> Option<String> {
|
||||
tree
|
||||
.chunks
|
||||
.iter()
|
||||
.filter(|chunk| {
|
||||
(chunk.start_byte as usize) <= start
|
||||
&& end <= (chunk.end_byte as usize)
|
||||
&& (chunk.end_byte as usize).saturating_sub(chunk.start_byte as usize)
|
||||
>= end.saturating_sub(start)
|
||||
})
|
||||
.min_by_key(|chunk| {
|
||||
(
|
||||
(chunk.end_byte as usize).saturating_sub(chunk.start_byte as usize),
|
||||
chunk.path.split('.').count(),
|
||||
)
|
||||
})
|
||||
.map(|chunk| chunk.path.clone())
|
||||
}
|
||||
|
||||
fn next_conflict_path(
|
||||
parent_path: &str,
|
||||
counters: &mut HashMap<String, usize>,
|
||||
existing_paths: &HashSet<String>,
|
||||
) -> String {
|
||||
let key = parent_path.to_owned();
|
||||
let next = counters.entry(key).or_insert(1);
|
||||
loop {
|
||||
let leaf = format!("conflict_{next}");
|
||||
let path = if parent_path.is_empty() {
|
||||
leaf
|
||||
} else {
|
||||
format!("{parent_path}.{leaf}")
|
||||
};
|
||||
*next += 1;
|
||||
if !existing_paths.contains(path.as_str()) {
|
||||
return path;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn insert_conflict_child(
|
||||
tree: &mut ChunkTree,
|
||||
parent_path: &str,
|
||||
conflict_path: &str,
|
||||
ours_start_byte: usize,
|
||||
ours_end_byte: usize,
|
||||
) {
|
||||
let Some(parent_index) = tree
|
||||
.chunks
|
||||
.iter()
|
||||
.position(|chunk| chunk.path == parent_path)
|
||||
else {
|
||||
return;
|
||||
};
|
||||
let current_children = tree.chunks[parent_index].children.clone();
|
||||
let mut updated_children = Vec::with_capacity(current_children.len() + 1);
|
||||
let mut inserted = false;
|
||||
|
||||
for child_path in current_children {
|
||||
let Some(child) = tree.chunks.iter().find(|chunk| chunk.path == child_path) else {
|
||||
continue;
|
||||
};
|
||||
let child_start = child.start_byte as usize;
|
||||
let child_end = child.end_byte as usize;
|
||||
let overlaps = child_start < ours_end_byte && ours_start_byte < child_end;
|
||||
if overlaps {
|
||||
continue;
|
||||
}
|
||||
if !inserted && child_start > ours_start_byte {
|
||||
updated_children.push(conflict_path.to_owned());
|
||||
inserted = true;
|
||||
}
|
||||
updated_children.push(child_path);
|
||||
}
|
||||
|
||||
if !inserted {
|
||||
updated_children.push(conflict_path.to_owned());
|
||||
}
|
||||
|
||||
tree.chunks[parent_index]
|
||||
.children
|
||||
.clone_from(&updated_children);
|
||||
if parent_path.is_empty() {
|
||||
tree.root_children = updated_children;
|
||||
}
|
||||
}
|
||||
|
||||
fn byte_to_line(line_starts: &[usize], byte: usize) -> u32 {
|
||||
if line_starts.is_empty() {
|
||||
return 0;
|
||||
}
|
||||
line_starts.partition_point(|offset| *offset <= byte) as u32
|
||||
}
|
||||
|
||||
fn real_line_stats(line_starts: &[usize], start: usize, end: usize) -> (u32, u32, u32) {
|
||||
if start == end {
|
||||
let line = byte_to_line(line_starts, start);
|
||||
return (line, line, 0);
|
||||
}
|
||||
let start_line = byte_to_line(line_starts, start);
|
||||
let end_line = byte_to_line(line_starts, end.saturating_sub(1));
|
||||
let line_count = if end_line >= start_line {
|
||||
end_line - start_line + 1
|
||||
} else {
|
||||
0
|
||||
};
|
||||
(start_line, end_line, line_count)
|
||||
}
|
||||
|
||||
fn display_line_count(content: &str) -> usize {
|
||||
if content.is_empty() {
|
||||
0
|
||||
} else if content.ends_with('\n') {
|
||||
content.split_terminator('\n').count()
|
||||
} else {
|
||||
total_line_count(content)
|
||||
}
|
||||
}
|
||||
|
||||
fn conflict_parent_path(path: &str) -> Option<String> {
|
||||
match path.rsplit_once('.') {
|
||||
Some((parent, _)) => Some(parent.to_owned()),
|
||||
None if !path.is_empty() => Some(String::new()),
|
||||
None => None,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::*;
|
||||
use crate::chunk::state::ConflictMeta;
|
||||
|
||||
#[test]
|
||||
fn detects_standard_and_diff3_conflicts() {
|
||||
let source = "\
|
||||
one\n<<<<<<< HEAD\nours\n||||||| base\nbase\n=======\ntheirs\n>>>>>>> topic\ntwo\n<<<<<<< \
|
||||
HEAD\nx\n=======\ny\n>>>>>>> other\n";
|
||||
let conflicts = detect_conflicts(source);
|
||||
assert_eq!(conflicts.len(), 2);
|
||||
assert_eq!(conflicts[0].ours_content, "ours\n");
|
||||
assert_eq!(conflicts[0].theirs_content, "theirs\n");
|
||||
assert_eq!(conflicts[0].base_content.as_deref(), Some("base\n"));
|
||||
assert_eq!(conflicts[0].base_label.as_deref(), Some("base"));
|
||||
assert_eq!(conflicts[1].ours_content, "x\n");
|
||||
assert_eq!(conflicts[1].theirs_content, "y\n");
|
||||
assert!(conflicts[1].base_content.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn accept_ours_returns_clean_source_and_byte_ranges() {
|
||||
let source = "\
|
||||
fn a() {\n<<<<<<< HEAD\n\treturn foo();\n=======\n\treturn bar();\n>>>>>>> topic\n}\n";
|
||||
let conflicts = detect_conflicts(source);
|
||||
let clean = accept_ours(source, &conflicts);
|
||||
assert_eq!(clean.source, "fn a() {\n\treturn foo();\n}\n");
|
||||
assert_eq!(clean.ours_byte_ranges, vec![(9, 24)]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reconstructs_conflict_markers_from_clean_source() {
|
||||
let clean_source = "fn a() {\n\treturn foo();\n}\n";
|
||||
let mut conflict_meta = HashMap::new();
|
||||
conflict_meta.insert("fn_a.conflict_1".to_owned(), ConflictMeta {
|
||||
theirs_content: "\treturn bar();\n".to_owned(),
|
||||
ours_label: "HEAD".to_owned(),
|
||||
theirs_label: "topic".to_owned(),
|
||||
base_content: Some("\treturn baz();\n".to_owned()),
|
||||
base_label: Some("base".to_owned()),
|
||||
ours_start_byte: 9,
|
||||
ours_end_byte: 24,
|
||||
});
|
||||
|
||||
let reconstructed = reconstruct_markers(clean_source, &conflict_meta);
|
||||
assert_eq!(
|
||||
reconstructed,
|
||||
"fn a() {\n<<<<<<< HEAD\n\treturn foo();\n||||||| base\n\treturn \
|
||||
baz();\n=======\n\treturn bar();\n>>>>>>> topic\n}\n",
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1,85 +0,0 @@
|
||||
//! Default (shared) classification logic.
|
||||
//!
|
||||
//! These are minimal catch-all fallbacks for node kinds not handled by any
|
||||
//! per-language classifier. The real classification lives in the `ast_*`
|
||||
//! modules; these defaults only fire for truly unrecognized node kinds.
|
||||
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{common::*, kind::ChunkKind};
|
||||
|
||||
pub fn classify_root_default<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
|
||||
infer_named_candidate(node, source)
|
||||
}
|
||||
|
||||
pub fn classify_class_default<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
|
||||
infer_named_candidate(node, source)
|
||||
}
|
||||
|
||||
pub fn classify_function_default<'tree>(
|
||||
node: Node<'tree>,
|
||||
source: &str,
|
||||
) -> RawChunkCandidate<'tree> {
|
||||
let kind_name = sanitize_node_kind(node.kind());
|
||||
let kind = ChunkKind::from_sanitized_kind(kind_name);
|
||||
let candidate = auto_classify(node, kind, source);
|
||||
if candidate.recurse.is_some() || candidate.identifier.is_some() {
|
||||
candidate
|
||||
} else {
|
||||
group_candidate(node, kind, source)
|
||||
}
|
||||
}
|
||||
|
||||
pub fn classify_var_decl<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
|
||||
if let Some(candidate) = promote_assigned_expression(node, node, source) {
|
||||
return candidate;
|
||||
}
|
||||
if let Some(name) = extract_single_declarator_name(node, source) {
|
||||
return make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None);
|
||||
}
|
||||
group_candidate(node, ChunkKind::Declarations, source)
|
||||
}
|
||||
|
||||
pub fn promote_assigned_expression<'tree>(
|
||||
range_node: Node<'tree>,
|
||||
declaration_node: Node<'tree>,
|
||||
source: &str,
|
||||
) -> Option<RawChunkCandidate<'tree>> {
|
||||
let declarators: Vec<Node<'tree>> = named_children(declaration_node)
|
||||
.into_iter()
|
||||
.filter(|c| c.kind() == "variable_declarator")
|
||||
.collect();
|
||||
if declarators.len() != 1 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let decl = declarators[0];
|
||||
let value = decl.child_by_field_name("value")?;
|
||||
let name = extract_identifier(decl, source).unwrap_or_else(|| "anonymous".to_string());
|
||||
|
||||
match value.kind() {
|
||||
"arrow_function" | "function_expression" | "function" => {
|
||||
let recurse = recurse_body(value, ChunkContext::FunctionBody);
|
||||
Some(make_kind_chunk_from(
|
||||
range_node,
|
||||
value,
|
||||
ChunkKind::Function,
|
||||
Some(name),
|
||||
source,
|
||||
recurse,
|
||||
))
|
||||
},
|
||||
"class" | "class_expression" => {
|
||||
let recurse = recurse_class(value);
|
||||
Some(make_container_chunk_from(
|
||||
range_node,
|
||||
value,
|
||||
ChunkKind::Class,
|
||||
Some(name),
|
||||
source,
|
||||
recurse,
|
||||
))
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,618 +0,0 @@
|
||||
use crate::chunk::{HASHLINE_BIGRAMS, types::ChunkTree};
|
||||
|
||||
const DEFAULT_SPACE_INDENT_STEP: usize = 4;
|
||||
const MAX_REASONABLE_INDENT_STEP: usize = 8;
|
||||
|
||||
pub fn dedent_python_style(text: &str) -> String {
|
||||
let mut margin: Option<&str> = None;
|
||||
for line in text.split('\n') {
|
||||
if line.trim().is_empty() {
|
||||
continue;
|
||||
}
|
||||
let indent = leading_whitespace(line);
|
||||
margin = Some(match margin {
|
||||
None => indent,
|
||||
Some(current) if indent.starts_with(current) => current,
|
||||
Some(current) if current.starts_with(indent) => indent,
|
||||
Some(current) => common_prefix(current, indent),
|
||||
});
|
||||
}
|
||||
|
||||
let Some(margin) = margin else {
|
||||
return text.to_owned();
|
||||
};
|
||||
if margin.is_empty() {
|
||||
return text.to_owned();
|
||||
}
|
||||
|
||||
text
|
||||
.split('\n')
|
||||
.map(|line| line.strip_prefix(margin).unwrap_or(line))
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n")
|
||||
}
|
||||
|
||||
pub fn indent_non_empty_lines(text: &str, prefix: &str) -> String {
|
||||
if prefix.is_empty() {
|
||||
return text.to_owned();
|
||||
}
|
||||
text
|
||||
.split('\n')
|
||||
.map(|line| {
|
||||
if line.trim().is_empty() {
|
||||
line.to_owned()
|
||||
} else {
|
||||
format!("{prefix}{line}")
|
||||
}
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n")
|
||||
}
|
||||
|
||||
pub fn detect_space_indent_step(text: &str) -> usize {
|
||||
let mut min = usize::MAX;
|
||||
for line in text.split('\n') {
|
||||
if line.trim().is_empty() {
|
||||
continue;
|
||||
}
|
||||
let count = line.chars().take_while(|ch| *ch == ' ').count();
|
||||
if count > 0 {
|
||||
min = min.min(count);
|
||||
}
|
||||
}
|
||||
if min == usize::MAX || min == 0 || min > MAX_REASONABLE_INDENT_STEP {
|
||||
DEFAULT_SPACE_INDENT_STEP
|
||||
} else {
|
||||
min
|
||||
}
|
||||
}
|
||||
|
||||
pub fn count_indent_columns(whitespace: &str, space_step: usize) -> usize {
|
||||
whitespace
|
||||
.chars()
|
||||
.map(|ch| if ch == '\t' { space_step } else { 1 })
|
||||
.sum()
|
||||
}
|
||||
|
||||
pub fn normalize_to_tabs(line: &str, indent_char: char, indent_step: usize) -> String {
|
||||
if indent_char == '\t' {
|
||||
return line.to_owned();
|
||||
}
|
||||
|
||||
let whitespace = leading_whitespace(line);
|
||||
if whitespace.is_empty() {
|
||||
return line.to_owned();
|
||||
}
|
||||
|
||||
let step = indent_step.max(1);
|
||||
let total_columns = count_indent_columns(whitespace, step);
|
||||
let tabs = total_columns / step;
|
||||
let remainder = total_columns % step;
|
||||
format!("{}{}{}", "\t".repeat(tabs), " ".repeat(remainder), &line[whitespace.len()..])
|
||||
}
|
||||
|
||||
pub fn denormalize_from_tabs(
|
||||
line: &str,
|
||||
file_indent_char: char,
|
||||
file_indent_step: usize,
|
||||
) -> String {
|
||||
if file_indent_char != ' ' && file_indent_char != '\t' {
|
||||
return line.to_owned();
|
||||
}
|
||||
|
||||
let whitespace = leading_whitespace(line);
|
||||
if whitespace.is_empty() {
|
||||
return line.to_owned();
|
||||
}
|
||||
|
||||
let step = file_indent_step.max(1);
|
||||
let mut converted = String::with_capacity(whitespace.len() * step.max(1));
|
||||
for ch in whitespace.chars() {
|
||||
match ch {
|
||||
'\t' if file_indent_char == '\t' => converted.push('\t'),
|
||||
'\t' => converted.push_str(&file_indent_char.to_string().repeat(step)),
|
||||
' ' => converted.push(' '),
|
||||
_ => converted.push(ch),
|
||||
}
|
||||
}
|
||||
format!("{converted}{}", &line[whitespace.len()..])
|
||||
}
|
||||
|
||||
pub fn normalize_target_indent(target_indent: &str, sample_text: &str) -> String {
|
||||
if target_indent.is_empty() {
|
||||
return String::new();
|
||||
}
|
||||
let has_tabs = target_indent.contains('\t');
|
||||
let has_spaces = target_indent.contains(' ');
|
||||
if !has_tabs || !has_spaces {
|
||||
return target_indent.to_owned();
|
||||
}
|
||||
|
||||
let space_step = detect_space_indent_step(sample_text);
|
||||
let total_columns = count_indent_columns(target_indent, space_step);
|
||||
let normalized_levels = round_to_nearest_step(total_columns, space_step) / space_step;
|
||||
if normalized_levels == 0 {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
match target_indent.chars().next().unwrap_or(' ') {
|
||||
'\t' => "\t".repeat(normalized_levels),
|
||||
_ => " ".repeat(normalized_levels * space_step),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn normalize_leading_whitespace_char(
|
||||
text: &str,
|
||||
target_char: char,
|
||||
file_indent_step: Option<usize>,
|
||||
) -> String {
|
||||
if target_char != ' ' && target_char != '\t' {
|
||||
return text.to_owned();
|
||||
}
|
||||
|
||||
let other_char = if target_char == ' ' { '\t' } else { ' ' };
|
||||
let mut needs_conversion = false;
|
||||
for line in text.split('\n') {
|
||||
if line.trim().is_empty() {
|
||||
continue;
|
||||
}
|
||||
let ws = leading_whitespace(line);
|
||||
if ws.is_empty() {
|
||||
continue;
|
||||
}
|
||||
needs_conversion |= ws.contains(other_char);
|
||||
}
|
||||
|
||||
if !needs_conversion {
|
||||
return text.to_owned();
|
||||
}
|
||||
|
||||
let space_step = if target_char == ' ' {
|
||||
file_indent_step
|
||||
.filter(|step| *step > 1)
|
||||
.unwrap_or_else(|| detect_space_indent_step(text))
|
||||
} else {
|
||||
file_indent_step.unwrap_or_else(|| detect_space_indent_step(text))
|
||||
};
|
||||
|
||||
text
|
||||
.split('\n')
|
||||
.map(|line| {
|
||||
let ws = leading_whitespace(line);
|
||||
if ws.is_empty() {
|
||||
return line.to_owned();
|
||||
}
|
||||
let rest = &line[ws.len()..];
|
||||
let total_spaces = count_indent_columns(ws, space_step);
|
||||
if target_char == ' ' {
|
||||
format!("{}{}", " ".repeat(total_spaces), rest)
|
||||
} else {
|
||||
let tabs = total_spaces / space_step;
|
||||
let remainder = total_spaces % space_step;
|
||||
format!("{}{}{}", "\t".repeat(tabs), " ".repeat(remainder), rest)
|
||||
}
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n")
|
||||
}
|
||||
|
||||
pub fn reindent_inserted_block(
|
||||
content: &str,
|
||||
target_indent: &str,
|
||||
file_indent_step: Option<usize>,
|
||||
) -> String {
|
||||
if content.is_empty() {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
let normalized_target_indent = normalize_target_indent(target_indent, content);
|
||||
let mut dedented = dedent_python_style(content);
|
||||
|
||||
if let Some(target_char) = normalized_target_indent.chars().next() {
|
||||
let target_step = if target_char == ' ' {
|
||||
file_indent_step.unwrap_or_else(|| normalized_target_indent.chars().count())
|
||||
} else {
|
||||
file_indent_step.unwrap_or_else(|| detect_space_indent_step(&dedented))
|
||||
};
|
||||
dedented = normalize_leading_whitespace_char(&dedented, target_char, Some(target_step));
|
||||
}
|
||||
|
||||
indent_non_empty_lines(&dedented, &normalized_target_indent)
|
||||
}
|
||||
|
||||
/// Detect the file's indent character.
|
||||
/// Prefer chunk metadata, then fall back to scanning source lines.
|
||||
/// Returns `' '` when the file provides no indentation signal.
|
||||
pub fn detect_file_indent_char(source: &str, tree: &ChunkTree) -> char {
|
||||
for chunk in &tree.chunks {
|
||||
if chunk.indent > 0 && !chunk.indent_char.is_empty() {
|
||||
return chunk.indent_char.chars().next().unwrap_or(' ');
|
||||
}
|
||||
}
|
||||
|
||||
for line in source.split('\n') {
|
||||
if line.trim().is_empty() {
|
||||
continue;
|
||||
}
|
||||
if let Some(ch) = leading_whitespace(line).chars().next()
|
||||
&& matches!(ch, ' ' | '\t')
|
||||
{
|
||||
return ch;
|
||||
}
|
||||
}
|
||||
|
||||
' '
|
||||
}
|
||||
|
||||
/// Detect spaces-per-indent-level from parent→child indent differences.
|
||||
/// Falls back to scanning `source` for the minimum indented-line width,
|
||||
/// then to `DEFAULT_SPACE_INDENT_STEP` when neither gives a signal.
|
||||
pub fn detect_file_indent_step(source: &str, tree: &ChunkTree) -> u32 {
|
||||
for chunk in &tree.chunks {
|
||||
if chunk.children.is_empty() {
|
||||
continue;
|
||||
}
|
||||
for child_path in &chunk.children {
|
||||
let Some(child) = tree
|
||||
.chunks
|
||||
.iter()
|
||||
.find(|candidate| &candidate.path == child_path)
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
if child.indent <= chunk.indent || child.indent_char != " " {
|
||||
continue;
|
||||
}
|
||||
let step = child.indent - chunk.indent;
|
||||
if step > 0 && step <= MAX_REASONABLE_INDENT_STEP as u32 {
|
||||
return step;
|
||||
}
|
||||
}
|
||||
}
|
||||
detect_space_indent_step(source) as u32
|
||||
}
|
||||
|
||||
pub fn strip_content_prefixes(content: &str) -> String {
|
||||
let lines = content.split('\n').collect::<Vec<_>>();
|
||||
let mut line_num_count = 0usize;
|
||||
let mut non_empty = 0usize;
|
||||
for line in &lines {
|
||||
if line.trim().is_empty() {
|
||||
continue;
|
||||
}
|
||||
non_empty += 1;
|
||||
if parse_chunk_gutter_code_row(line).is_some() {
|
||||
line_num_count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
if non_empty == 0 {
|
||||
return content.to_owned();
|
||||
}
|
||||
|
||||
let without_line_numbers = if line_num_count * 10 > non_empty * 6 {
|
||||
lines
|
||||
.iter()
|
||||
.map(|line| strip_chunk_gutter_line(line))
|
||||
.collect::<Vec<_>>()
|
||||
} else {
|
||||
lines
|
||||
.iter()
|
||||
.map(|line| (*line).to_owned())
|
||||
.collect::<Vec<_>>()
|
||||
};
|
||||
|
||||
strip_new_line_prefixes(&without_line_numbers).join("\n")
|
||||
}
|
||||
|
||||
fn leading_whitespace(line: &str) -> &str {
|
||||
let end = line
|
||||
.char_indices()
|
||||
.find_map(|(index, ch)| (!matches!(ch, ' ' | '\t')).then_some(index))
|
||||
.unwrap_or(line.len());
|
||||
&line[..end]
|
||||
}
|
||||
|
||||
fn common_prefix<'a>(left: &'a str, right: &'a str) -> &'a str {
|
||||
let mut matched = 0usize;
|
||||
for ((left_index, left_char), (_, right_char)) in left.char_indices().zip(right.char_indices()) {
|
||||
if left_char != right_char {
|
||||
break;
|
||||
}
|
||||
matched = left_index + left_char.len_utf8();
|
||||
}
|
||||
&left[..matched]
|
||||
}
|
||||
|
||||
const fn round_to_nearest_step(value: usize, step: usize) -> usize {
|
||||
if step == 0 {
|
||||
return value;
|
||||
}
|
||||
((value + (step / 2)) / step) * step
|
||||
}
|
||||
|
||||
fn parse_chunk_gutter_code_row(line: &str) -> Option<&str> {
|
||||
let trimmed = line.trim_start_matches([' ', '\t']);
|
||||
let digits = trimmed.chars().take_while(|ch| ch.is_ascii_digit()).count();
|
||||
if digits == 0 {
|
||||
return None;
|
||||
}
|
||||
let after_digits = &trimmed[digits..];
|
||||
let after_spaces = after_digits.trim_start_matches([' ', '\t']);
|
||||
let after_pipe = after_spaces
|
||||
.strip_prefix('|')
|
||||
.or_else(|| after_spaces.strip_prefix('│'))?;
|
||||
Some(
|
||||
after_pipe
|
||||
.strip_prefix(' ')
|
||||
.or_else(|| after_pipe.strip_prefix('\t'))
|
||||
.unwrap_or(after_pipe),
|
||||
)
|
||||
}
|
||||
|
||||
fn strip_chunk_gutter_line(line: &str) -> String {
|
||||
if let Some(rest) = parse_chunk_gutter_code_row(line) {
|
||||
return rest.to_owned();
|
||||
}
|
||||
let trimmed = line.trim_start_matches([' ', '\t']);
|
||||
if trimmed.starts_with('|') || trimmed.starts_with('│') {
|
||||
return String::new();
|
||||
}
|
||||
line.to_owned()
|
||||
}
|
||||
|
||||
fn strip_new_line_prefixes(lines: &[String]) -> Vec<String> {
|
||||
let non_empty = lines.iter().filter(|line| !line.trim().is_empty()).count();
|
||||
if non_empty == 0 {
|
||||
return lines.to_vec();
|
||||
}
|
||||
|
||||
let hash_prefixed = lines
|
||||
.iter()
|
||||
.filter(|line| !line.trim().is_empty())
|
||||
.filter(|line| hashline_prefix_len(line).is_some())
|
||||
.count();
|
||||
if hash_prefixed == non_empty {
|
||||
return lines
|
||||
.iter()
|
||||
.map(|line| match hashline_prefix_len(line) {
|
||||
Some(prefix_len) => line[prefix_len..].to_owned(),
|
||||
None => line.clone(),
|
||||
})
|
||||
.collect();
|
||||
}
|
||||
|
||||
lines.to_vec()
|
||||
}
|
||||
|
||||
fn hashline_prefix_len(line: &str) -> Option<usize> {
|
||||
let mut offset = line.len() - line.trim_start_matches([' ', '\t']).len();
|
||||
let mut remainder = &line[offset..];
|
||||
|
||||
if let Some(stripped) = remainder.strip_prefix(">>>") {
|
||||
offset += 3;
|
||||
remainder = stripped;
|
||||
} else if let Some(stripped) = remainder.strip_prefix(">>") {
|
||||
offset += 2;
|
||||
remainder = stripped;
|
||||
}
|
||||
|
||||
let ws = remainder.len() - remainder.trim_start_matches([' ', '\t']).len();
|
||||
offset += ws;
|
||||
remainder = &remainder[ws..];
|
||||
|
||||
if let Some(stripped) = remainder.strip_prefix('+') {
|
||||
offset += 1;
|
||||
remainder = stripped;
|
||||
let inner_ws = remainder.len() - remainder.trim_start_matches([' ', '\t']).len();
|
||||
offset += inner_ws;
|
||||
remainder = &remainder[inner_ws..];
|
||||
}
|
||||
|
||||
// Line number digits are mandatory (no `#`-only form in the new format).
|
||||
let digits = remainder
|
||||
.chars()
|
||||
.take_while(|ch| ch.is_ascii_digit())
|
||||
.count();
|
||||
if digits == 0 {
|
||||
return None;
|
||||
}
|
||||
offset += digits;
|
||||
remainder = &remainder[digits..];
|
||||
|
||||
// Match exactly one BPE bigram (2 ASCII chars) from HASHLINE_BIGRAMS,
|
||||
// directly adjacent to the line-number digits (no `#` separator).
|
||||
// Use char-boundary-safe slicing to avoid panicking on multi-byte content.
|
||||
let bigram_end = 2;
|
||||
if remainder.len() < bigram_end || !remainder.is_char_boundary(bigram_end) {
|
||||
return None;
|
||||
}
|
||||
let bigram = &remainder[..bigram_end];
|
||||
if !bigram.is_ascii() || !HASHLINE_BIGRAMS.contains(&bigram) {
|
||||
return None;
|
||||
}
|
||||
offset += bigram_end;
|
||||
remainder = &remainder[bigram_end..];
|
||||
|
||||
// Anchor terminator is a single colon character.
|
||||
remainder.strip_prefix(':').map(|_| offset + 1)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::chunk::{kind::ChunkKind, types::ChunkNode};
|
||||
|
||||
fn chunk(
|
||||
path: &str,
|
||||
parent_path: Option<&str>,
|
||||
children: &[&str],
|
||||
indent: u32,
|
||||
indent_char: &str,
|
||||
) -> ChunkNode {
|
||||
let kind = match path.split_once('_').map_or(path, |(prefix, _)| prefix) {
|
||||
"fn" => ChunkKind::Function,
|
||||
"class" => ChunkKind::Class,
|
||||
"stmts" => ChunkKind::Statements,
|
||||
_ => ChunkKind::Chunk,
|
||||
};
|
||||
ChunkNode {
|
||||
path: path.to_owned(),
|
||||
identifier: path
|
||||
.split_once('_')
|
||||
.and_then(|(_, identifier)| (!identifier.is_empty()).then_some(identifier.to_owned())),
|
||||
kind,
|
||||
leaf: children.is_empty(),
|
||||
virtual_content: None,
|
||||
parent_path: parent_path.map(str::to_owned),
|
||||
children: children.iter().map(|child| (*child).to_owned()).collect(),
|
||||
signature: None,
|
||||
start_line: 1,
|
||||
end_line: 1,
|
||||
line_count: 1,
|
||||
start_byte: 0,
|
||||
end_byte: 0,
|
||||
checksum_start_byte: 0,
|
||||
prologue_end_byte: None,
|
||||
epilogue_start_byte: None,
|
||||
checksum: "ABCD".to_owned(),
|
||||
error: false,
|
||||
indent,
|
||||
indent_char: indent_char.to_owned(),
|
||||
group: false,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dedent_preserves_mixed_common_margin() {
|
||||
let input = "\t foo\n\t bar\n\t baz";
|
||||
assert_eq!(dedent_python_style(input), "foo\n bar\nbaz");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn normalize_leading_whitespace_char_uses_file_step() {
|
||||
let input = "\t alpha\n\t beta";
|
||||
assert_eq!(
|
||||
normalize_leading_whitespace_char(input, ' ', Some(4)),
|
||||
" alpha\n beta"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn canonical_indent_round_trips_common_profiles() {
|
||||
let cases = [
|
||||
(" value()", ' ', 4, "\tvalue()", " value()"),
|
||||
(" value()", ' ', 3, "\t\tvalue()", " value()"),
|
||||
(" value()", ' ', 2, "\tvalue()", " value()"),
|
||||
("\tvalue()", '\t', 4, "\tvalue()", "\tvalue()"),
|
||||
(" \t value()", ' ', 4, "\t value()", " value()"),
|
||||
];
|
||||
|
||||
for (input, indent_char, indent_step, canonical, restored) in cases {
|
||||
let normalized = normalize_to_tabs(input, indent_char, indent_step);
|
||||
assert_eq!(normalized, canonical, "unexpected canonical indent for {input:?}");
|
||||
assert_eq!(
|
||||
denormalize_from_tabs(&normalized, indent_char, indent_step),
|
||||
restored,
|
||||
"unexpected restored indent for {input:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reindent_inserted_block_preserves_relative_indentation() {
|
||||
// Agent sends tab-based content: call(\n\talpha,\n\tbeta,\n)
|
||||
// After denormalize_from_tabs (4 spaces/tab): call(\n alpha,\n beta,\n)
|
||||
// Should add target indent to all lines, preserving relative offsets.
|
||||
let input = "call(\n alpha,\n beta,\n)";
|
||||
assert_eq!(
|
||||
reindent_inserted_block(input, " ", Some(4)),
|
||||
" call(\n alpha,\n beta,\n )"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strip_content_prefixes_removes_gutter_and_meta_rows() {
|
||||
let input = "10 | fn main() {\n │ <.fn_main#ABCD>\n11 | println!(\"hi\");\n12 | }";
|
||||
assert_eq!(strip_content_prefixes(input), "fn main() {\n\n println!(\"hi\");\n}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strip_content_prefixes_removes_colon_hashline_prefixes() {
|
||||
let input = "1th:fn main() {\n2er:\tprintln!(\"hi\");\n3in:}";
|
||||
assert_eq!(strip_content_prefixes(input), "fn main() {\n\tprintln!(\"hi\");\n}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn detect_file_indent_step_prefers_space_children() {
|
||||
let tree = ChunkTree {
|
||||
language: "rust".to_owned(),
|
||||
checksum: "ABCD".to_owned(),
|
||||
line_count: 1,
|
||||
parse_errors: 0,
|
||||
parse_error_lines: Vec::new(),
|
||||
fallback: false,
|
||||
root_path: String::new(),
|
||||
root_children: vec!["cls_A".to_owned()],
|
||||
chunks: vec![
|
||||
chunk("cls_A", Some(""), &["fn_b"], 0, " "),
|
||||
chunk("fn_b", Some("cls_A"), &[], 2, " "),
|
||||
],
|
||||
};
|
||||
assert_eq!(detect_file_indent_step("", &tree), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn detect_file_indent_step_falls_back_to_source_scan() {
|
||||
// When the chunk tree has no parent->child pairs (all leaves),
|
||||
// fall back to scanning source lines for the minimum indent width.
|
||||
let tree = ChunkTree {
|
||||
language: "yaml".to_owned(),
|
||||
checksum: "ABCD".to_owned(),
|
||||
line_count: 3,
|
||||
parse_errors: 0,
|
||||
parse_error_lines: Vec::new(),
|
||||
fallback: false,
|
||||
root_path: String::new(),
|
||||
root_children: vec!["key_ser".to_owned()],
|
||||
chunks: vec![chunk("key_ser", Some(""), &[], 0, " ")],
|
||||
};
|
||||
let source = "server:\n host: localhost\n port: 5432\n";
|
||||
assert_eq!(detect_file_indent_step(source, &tree), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn detect_file_indent_char_falls_back_to_source_lines() {
|
||||
let tree = ChunkTree {
|
||||
language: "rust".to_owned(),
|
||||
checksum: "ABCD".to_owned(),
|
||||
line_count: 3,
|
||||
parse_errors: 0,
|
||||
parse_error_lines: Vec::new(),
|
||||
fallback: false,
|
||||
root_path: String::new(),
|
||||
root_children: vec!["fn_mai".to_owned()],
|
||||
chunks: vec![chunk("fn_mai", Some(""), &[], 0, "")],
|
||||
};
|
||||
|
||||
let source = "fn main() {\n println!(\"hi\");\n}\n";
|
||||
assert_eq!(detect_file_indent_char(source, &tree), ' ');
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn detect_file_indent_char_defaults_to_spaces_without_signal() {
|
||||
let tree = ChunkTree {
|
||||
language: "rust".to_owned(),
|
||||
checksum: "ABCD".to_owned(),
|
||||
line_count: 0,
|
||||
parse_errors: 0,
|
||||
parse_error_lines: Vec::new(),
|
||||
fallback: false,
|
||||
root_path: String::new(),
|
||||
root_children: Vec::new(),
|
||||
chunks: Vec::new(),
|
||||
};
|
||||
|
||||
assert_eq!(detect_file_indent_char("", &tree), ' ');
|
||||
}
|
||||
}
|
||||
@@ -1,699 +0,0 @@
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
|
||||
pub enum ChunkKind {
|
||||
Add,
|
||||
After,
|
||||
Alias,
|
||||
Algo,
|
||||
Arg,
|
||||
Argv,
|
||||
Array,
|
||||
At,
|
||||
Attr,
|
||||
AttrExpr,
|
||||
Attrs,
|
||||
Block,
|
||||
BlockIf,
|
||||
BlockLocals,
|
||||
Body,
|
||||
Case,
|
||||
Cases,
|
||||
Catch,
|
||||
Cell,
|
||||
Class,
|
||||
Clause,
|
||||
Cmd,
|
||||
Code,
|
||||
Cond,
|
||||
Constructor,
|
||||
Contract,
|
||||
Copy,
|
||||
Custom,
|
||||
Declarations,
|
||||
Decl,
|
||||
DefaultExport,
|
||||
Define,
|
||||
Directive,
|
||||
Either,
|
||||
Elif,
|
||||
Else,
|
||||
Enum,
|
||||
Env,
|
||||
Error,
|
||||
Except,
|
||||
Exports,
|
||||
Expose,
|
||||
Expression,
|
||||
Field,
|
||||
Fields,
|
||||
File,
|
||||
Frame,
|
||||
Function,
|
||||
For,
|
||||
ForIn,
|
||||
ForOf,
|
||||
Form,
|
||||
Frontmatter,
|
||||
Group,
|
||||
GroupBy,
|
||||
Headers,
|
||||
Healthcheck,
|
||||
Html,
|
||||
Hunks,
|
||||
Hunk,
|
||||
If,
|
||||
Iface,
|
||||
Impl,
|
||||
Imports,
|
||||
Includes,
|
||||
InlineFragment,
|
||||
Install,
|
||||
Interface,
|
||||
Interpolation,
|
||||
Item,
|
||||
Join,
|
||||
Key,
|
||||
KeyScripts,
|
||||
Label,
|
||||
Let,
|
||||
List,
|
||||
Loop,
|
||||
Macro,
|
||||
Map,
|
||||
Markdown,
|
||||
Match,
|
||||
Method,
|
||||
Methods,
|
||||
Module,
|
||||
Mustache,
|
||||
Object,
|
||||
Operation,
|
||||
Operator,
|
||||
Option,
|
||||
Options,
|
||||
OrderBy,
|
||||
Parameters,
|
||||
Preamble,
|
||||
Proc,
|
||||
Process,
|
||||
Project,
|
||||
Proto,
|
||||
Py,
|
||||
Python,
|
||||
Query,
|
||||
Receive,
|
||||
Recipe,
|
||||
Relations,
|
||||
Render,
|
||||
Return,
|
||||
Row,
|
||||
Root,
|
||||
Rule,
|
||||
Schema,
|
||||
Script,
|
||||
ScriptModule,
|
||||
ScriptSetup,
|
||||
Section,
|
||||
Select,
|
||||
Setting,
|
||||
Shebang,
|
||||
Shell,
|
||||
Slot,
|
||||
Snippet,
|
||||
Source,
|
||||
Stage,
|
||||
StaticInit,
|
||||
Statements,
|
||||
Struct,
|
||||
Style,
|
||||
StyleScoped,
|
||||
Switch,
|
||||
Table,
|
||||
Tag,
|
||||
Target,
|
||||
Template,
|
||||
Text,
|
||||
Trait,
|
||||
Translation,
|
||||
Try,
|
||||
Ts,
|
||||
Type,
|
||||
Typescript,
|
||||
Union,
|
||||
User,
|
||||
Val,
|
||||
Variable,
|
||||
Variant,
|
||||
Variants,
|
||||
VersionGate,
|
||||
When,
|
||||
Where,
|
||||
While,
|
||||
With,
|
||||
Workdir,
|
||||
Conflict,
|
||||
Ours,
|
||||
Theirs,
|
||||
Chunk,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum SummaryStyle {
|
||||
Function,
|
||||
Variable,
|
||||
Imports,
|
||||
Default,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub struct ChunkTraits {
|
||||
pub groupable: bool,
|
||||
pub packed: bool,
|
||||
pub addressable_leaf: bool,
|
||||
pub always_preserve_children: bool,
|
||||
pub has_addressable_members: bool,
|
||||
pub summary: SummaryStyle,
|
||||
pub container: bool,
|
||||
}
|
||||
|
||||
const DEFAULT_TRAITS: ChunkTraits = ChunkTraits {
|
||||
groupable: false,
|
||||
packed: false,
|
||||
addressable_leaf: false,
|
||||
always_preserve_children: false,
|
||||
has_addressable_members: false,
|
||||
summary: SummaryStyle::Default,
|
||||
container: false,
|
||||
};
|
||||
|
||||
const GROUP_TRAITS: ChunkTraits = ChunkTraits { groupable: true, ..DEFAULT_TRAITS };
|
||||
|
||||
const CONTAINER_TRAITS: ChunkTraits = ChunkTraits { container: true, ..DEFAULT_TRAITS };
|
||||
|
||||
const ADDRESSABLE_CONTAINER_TRAITS: ChunkTraits =
|
||||
ChunkTraits { container: true, has_addressable_members: true, ..DEFAULT_TRAITS };
|
||||
|
||||
const PRESERVE_CHILDREN_TRAITS: ChunkTraits =
|
||||
ChunkTraits { container: true, always_preserve_children: true, ..DEFAULT_TRAITS };
|
||||
|
||||
const PACKED_LEAF_TRAITS: ChunkTraits =
|
||||
ChunkTraits { packed: true, addressable_leaf: true, ..DEFAULT_TRAITS };
|
||||
|
||||
const FUNCTION_TRAITS: ChunkTraits =
|
||||
ChunkTraits { summary: SummaryStyle::Function, ..DEFAULT_TRAITS };
|
||||
|
||||
const VARIABLE_TRAITS: ChunkTraits = ChunkTraits {
|
||||
packed: true,
|
||||
addressable_leaf: true,
|
||||
summary: SummaryStyle::Variable,
|
||||
..DEFAULT_TRAITS
|
||||
};
|
||||
|
||||
const IMPORTS_TRAITS: ChunkTraits =
|
||||
ChunkTraits { groupable: true, summary: SummaryStyle::Imports, ..DEFAULT_TRAITS };
|
||||
|
||||
impl ChunkKind {
|
||||
pub const fn prefix(self) -> &'static str {
|
||||
match self {
|
||||
Self::Add => "add",
|
||||
Self::After => "aft",
|
||||
Self::Alias => "al",
|
||||
Self::Algo => "algo",
|
||||
Self::Arg => "arg",
|
||||
Self::Argv => "argv",
|
||||
Self::Array => "ar",
|
||||
Self::At => "at",
|
||||
Self::Attr => "attr",
|
||||
Self::AttrExpr => "aex",
|
||||
Self::Attrs => "ats",
|
||||
Self::Block => "blk",
|
||||
Self::BlockIf => "bif",
|
||||
Self::BlockLocals => "blc",
|
||||
Self::Body => "b",
|
||||
Self::Case => "case",
|
||||
Self::Cases => "cs",
|
||||
Self::Catch => "ctc",
|
||||
Self::Cell => "cell",
|
||||
Self::Class => "cls",
|
||||
Self::Clause => "cla",
|
||||
Self::Cmd => "cmd",
|
||||
Self::Code => "code",
|
||||
Self::Cond => "cond",
|
||||
Self::Constructor => "ctor",
|
||||
Self::Contract => "ctr",
|
||||
Self::Copy => "copy",
|
||||
Self::Custom => "cus",
|
||||
Self::Declarations => "d",
|
||||
Self::Decl => "decl",
|
||||
Self::DefaultExport => "dex",
|
||||
Self::Define => "def",
|
||||
Self::Directive => "dir",
|
||||
Self::Either => "eth",
|
||||
Self::Elif => "elif",
|
||||
Self::Else => "else",
|
||||
Self::Enum => "en",
|
||||
Self::Env => "env",
|
||||
Self::Error => "err",
|
||||
Self::Except => "exc",
|
||||
Self::Exports => "exp",
|
||||
Self::Expose => "exo",
|
||||
Self::Expression => "ex",
|
||||
Self::Field => "fld",
|
||||
Self::Fields => "flds",
|
||||
Self::File => "file",
|
||||
Self::Frame => "fr",
|
||||
Self::Function => "fn",
|
||||
Self::For => "for",
|
||||
Self::ForIn => "fri",
|
||||
Self::ForOf => "fro",
|
||||
Self::Form => "form",
|
||||
Self::Frontmatter => "fm",
|
||||
Self::Group => "grp",
|
||||
Self::GroupBy => "gby",
|
||||
Self::Headers => "hdrs",
|
||||
Self::Healthcheck => "hck",
|
||||
Self::Html => "html",
|
||||
Self::Hunks => "hks",
|
||||
Self::Hunk => "hunk",
|
||||
Self::If => "if",
|
||||
Self::Iface => "ifc",
|
||||
Self::Impl => "ipl",
|
||||
Self::Imports => "imp",
|
||||
Self::Includes => "incl",
|
||||
Self::InlineFragment => "ifr",
|
||||
Self::Install => "inst",
|
||||
Self::Interface => "intf",
|
||||
Self::Interpolation => "itp",
|
||||
Self::Item => "item",
|
||||
Self::Join => "join",
|
||||
Self::Key => "key",
|
||||
Self::KeyScripts => "ksc",
|
||||
Self::Label => "lbl",
|
||||
Self::Let => "let",
|
||||
Self::List => "list",
|
||||
Self::Loop => "loop",
|
||||
Self::Macro => "mc",
|
||||
Self::Map => "map",
|
||||
Self::Markdown => "md",
|
||||
Self::Match => "mt",
|
||||
Self::Method => "m",
|
||||
Self::Methods => "ms",
|
||||
Self::Module => "mod",
|
||||
Self::Mustache => "mst",
|
||||
Self::Object => "obj",
|
||||
Self::Operation => "op",
|
||||
Self::Operator => "oper",
|
||||
Self::Option => "opt",
|
||||
Self::Options => "opts",
|
||||
Self::OrderBy => "oby",
|
||||
Self::Parameters => "p",
|
||||
Self::Preamble => "pre",
|
||||
Self::Proc => "proc",
|
||||
Self::Process => "prcs",
|
||||
Self::Project => "proj",
|
||||
Self::Proto => "pt",
|
||||
Self::Py => "py",
|
||||
Self::Python => "pyt",
|
||||
Self::Query => "qry",
|
||||
Self::Receive => "recv",
|
||||
Self::Recipe => "rcp",
|
||||
Self::Relations => "rels",
|
||||
Self::Render => "rnd",
|
||||
Self::Return => "ret",
|
||||
Self::Row => "row",
|
||||
Self::Root => "root",
|
||||
Self::Rule => "rule",
|
||||
Self::Schema => "sch",
|
||||
Self::Script => "scr",
|
||||
Self::ScriptModule => "smo",
|
||||
Self::ScriptSetup => "sse",
|
||||
Self::Section => "sct",
|
||||
Self::Select => "sel",
|
||||
Self::Setting => "stn",
|
||||
Self::Shebang => "shb",
|
||||
Self::Shell => "sh",
|
||||
Self::Slot => "slot",
|
||||
Self::Snippet => "snip",
|
||||
Self::Source => "src",
|
||||
Self::Stage => "stg",
|
||||
Self::StaticInit => "sni",
|
||||
Self::Statements => "st",
|
||||
Self::Struct => "stc",
|
||||
Self::Style => "sty",
|
||||
Self::StyleScoped => "sco",
|
||||
Self::Switch => "sw",
|
||||
Self::Table => "tbl",
|
||||
Self::Tag => "tag",
|
||||
Self::Target => "tgt",
|
||||
Self::Template => "tmpl",
|
||||
Self::Text => "text",
|
||||
Self::Trait => "tr",
|
||||
Self::Translation => "trs",
|
||||
Self::Try => "try",
|
||||
Self::Ts => "ts",
|
||||
Self::Type => "ty",
|
||||
Self::Typescript => "tysc",
|
||||
Self::Union => "u",
|
||||
Self::User => "user",
|
||||
Self::Val => "val",
|
||||
Self::Variable => "var",
|
||||
Self::Variant => "vr",
|
||||
Self::Variants => "vrs",
|
||||
Self::VersionGate => "vg",
|
||||
Self::When => "when",
|
||||
Self::Where => "wh",
|
||||
Self::While => "wl",
|
||||
Self::With => "with",
|
||||
Self::Workdir => "wd",
|
||||
Self::Conflict => "cfl",
|
||||
Self::Ours => "ours",
|
||||
Self::Theirs => "ths",
|
||||
Self::Chunk => "ch",
|
||||
}
|
||||
}
|
||||
|
||||
pub const fn traits(self) -> &'static ChunkTraits {
|
||||
match self {
|
||||
Self::Add => &GROUP_TRAITS,
|
||||
Self::After => &GROUP_TRAITS,
|
||||
Self::Alias => &DEFAULT_TRAITS,
|
||||
Self::Algo => &CONTAINER_TRAITS,
|
||||
Self::Arg => &DEFAULT_TRAITS,
|
||||
Self::Argv => &DEFAULT_TRAITS,
|
||||
Self::Array => &DEFAULT_TRAITS,
|
||||
Self::At => &DEFAULT_TRAITS,
|
||||
Self::Attr => &DEFAULT_TRAITS,
|
||||
Self::AttrExpr => &DEFAULT_TRAITS,
|
||||
Self::Attrs => &CONTAINER_TRAITS,
|
||||
Self::Block => &DEFAULT_TRAITS,
|
||||
Self::BlockIf => &DEFAULT_TRAITS,
|
||||
Self::BlockLocals => &DEFAULT_TRAITS,
|
||||
Self::Body => &DEFAULT_TRAITS,
|
||||
Self::Case => &DEFAULT_TRAITS,
|
||||
Self::Cases => &GROUP_TRAITS,
|
||||
Self::Catch => &DEFAULT_TRAITS,
|
||||
Self::Cell => &CONTAINER_TRAITS,
|
||||
Self::Class => &CONTAINER_TRAITS,
|
||||
Self::Clause => &DEFAULT_TRAITS,
|
||||
Self::Cmd => &GROUP_TRAITS,
|
||||
Self::Code => &GROUP_TRAITS,
|
||||
Self::Cond => &DEFAULT_TRAITS,
|
||||
Self::Constructor => &FUNCTION_TRAITS,
|
||||
Self::Contract => &DEFAULT_TRAITS,
|
||||
Self::Copy => &GROUP_TRAITS,
|
||||
Self::Custom => &DEFAULT_TRAITS,
|
||||
Self::Declarations => &GROUP_TRAITS,
|
||||
Self::Decl => &DEFAULT_TRAITS,
|
||||
Self::DefaultExport => &DEFAULT_TRAITS,
|
||||
Self::Define => &DEFAULT_TRAITS,
|
||||
Self::Directive => &DEFAULT_TRAITS,
|
||||
Self::Either => &DEFAULT_TRAITS,
|
||||
Self::Elif => &DEFAULT_TRAITS,
|
||||
Self::Else => &DEFAULT_TRAITS,
|
||||
Self::Enum => &ADDRESSABLE_CONTAINER_TRAITS,
|
||||
Self::Env => &DEFAULT_TRAITS,
|
||||
Self::Error => &DEFAULT_TRAITS,
|
||||
Self::Except => &DEFAULT_TRAITS,
|
||||
Self::Exports => &GROUP_TRAITS,
|
||||
Self::Expose => &DEFAULT_TRAITS,
|
||||
Self::Expression => &DEFAULT_TRAITS,
|
||||
Self::Field => &PACKED_LEAF_TRAITS,
|
||||
Self::Fields => &GROUP_TRAITS,
|
||||
Self::File => &DEFAULT_TRAITS,
|
||||
Self::Frame => &DEFAULT_TRAITS,
|
||||
Self::Function => &FUNCTION_TRAITS,
|
||||
Self::For => &DEFAULT_TRAITS,
|
||||
Self::ForIn => &DEFAULT_TRAITS,
|
||||
Self::ForOf => &DEFAULT_TRAITS,
|
||||
Self::Form => &DEFAULT_TRAITS,
|
||||
Self::Frontmatter => &DEFAULT_TRAITS,
|
||||
Self::Group => &DEFAULT_TRAITS,
|
||||
Self::GroupBy => &DEFAULT_TRAITS,
|
||||
Self::Headers => &GROUP_TRAITS,
|
||||
Self::Healthcheck => &DEFAULT_TRAITS,
|
||||
Self::Html => &GROUP_TRAITS,
|
||||
Self::Hunks => &GROUP_TRAITS,
|
||||
Self::Hunk => &DEFAULT_TRAITS,
|
||||
Self::If => &DEFAULT_TRAITS,
|
||||
Self::Iface => &PRESERVE_CHILDREN_TRAITS,
|
||||
Self::Impl => &CONTAINER_TRAITS,
|
||||
Self::Imports => &IMPORTS_TRAITS,
|
||||
Self::Includes => &GROUP_TRAITS,
|
||||
Self::InlineFragment => &DEFAULT_TRAITS,
|
||||
Self::Install => &DEFAULT_TRAITS,
|
||||
Self::Interface => &PRESERVE_CHILDREN_TRAITS,
|
||||
Self::Interpolation => &GROUP_TRAITS,
|
||||
Self::Item => &DEFAULT_TRAITS,
|
||||
Self::Join => &DEFAULT_TRAITS,
|
||||
Self::Key => &PACKED_LEAF_TRAITS,
|
||||
Self::KeyScripts => &DEFAULT_TRAITS,
|
||||
Self::Label => &DEFAULT_TRAITS,
|
||||
Self::Let => &DEFAULT_TRAITS,
|
||||
Self::List => &DEFAULT_TRAITS,
|
||||
Self::Loop => &DEFAULT_TRAITS,
|
||||
Self::Macro => &DEFAULT_TRAITS,
|
||||
Self::Map => &DEFAULT_TRAITS,
|
||||
Self::Markdown => &DEFAULT_TRAITS,
|
||||
Self::Match => &DEFAULT_TRAITS,
|
||||
Self::Method => &DEFAULT_TRAITS,
|
||||
Self::Methods => &GROUP_TRAITS,
|
||||
Self::Module => &CONTAINER_TRAITS,
|
||||
Self::Mustache => &DEFAULT_TRAITS,
|
||||
Self::Object => &DEFAULT_TRAITS,
|
||||
Self::Operation => &DEFAULT_TRAITS,
|
||||
Self::Operator => &DEFAULT_TRAITS,
|
||||
Self::Option => &DEFAULT_TRAITS,
|
||||
Self::Options => &GROUP_TRAITS,
|
||||
Self::OrderBy => &DEFAULT_TRAITS,
|
||||
Self::Parameters => &GROUP_TRAITS,
|
||||
Self::Preamble => &DEFAULT_TRAITS,
|
||||
Self::Proc => &CONTAINER_TRAITS,
|
||||
Self::Process => &CONTAINER_TRAITS,
|
||||
Self::Project => &DEFAULT_TRAITS,
|
||||
Self::Proto => &DEFAULT_TRAITS,
|
||||
Self::Py => &DEFAULT_TRAITS,
|
||||
Self::Python => &DEFAULT_TRAITS,
|
||||
Self::Query => &DEFAULT_TRAITS,
|
||||
Self::Receive => &DEFAULT_TRAITS,
|
||||
Self::Recipe => &DEFAULT_TRAITS,
|
||||
Self::Relations => &DEFAULT_TRAITS,
|
||||
Self::Render => &DEFAULT_TRAITS,
|
||||
Self::Return => &DEFAULT_TRAITS,
|
||||
Self::Row => &PACKED_LEAF_TRAITS,
|
||||
Self::Root => &DEFAULT_TRAITS,
|
||||
Self::Rule => &DEFAULT_TRAITS,
|
||||
Self::Schema => &DEFAULT_TRAITS,
|
||||
Self::Script => &DEFAULT_TRAITS,
|
||||
Self::ScriptModule => &DEFAULT_TRAITS,
|
||||
Self::ScriptSetup => &DEFAULT_TRAITS,
|
||||
Self::Section => &DEFAULT_TRAITS,
|
||||
Self::Select => &DEFAULT_TRAITS,
|
||||
Self::Setting => &DEFAULT_TRAITS,
|
||||
Self::Shebang => &DEFAULT_TRAITS,
|
||||
Self::Shell => &DEFAULT_TRAITS,
|
||||
Self::Slot => &DEFAULT_TRAITS,
|
||||
Self::Snippet => &DEFAULT_TRAITS,
|
||||
Self::Source => &DEFAULT_TRAITS,
|
||||
Self::Stage => &DEFAULT_TRAITS,
|
||||
Self::StaticInit => &DEFAULT_TRAITS,
|
||||
Self::Statements => &GROUP_TRAITS,
|
||||
Self::Struct => &ADDRESSABLE_CONTAINER_TRAITS,
|
||||
Self::Style => &DEFAULT_TRAITS,
|
||||
Self::StyleScoped => &DEFAULT_TRAITS,
|
||||
Self::Switch => &DEFAULT_TRAITS,
|
||||
Self::Table => &DEFAULT_TRAITS,
|
||||
Self::Tag => &DEFAULT_TRAITS,
|
||||
Self::Target => &DEFAULT_TRAITS,
|
||||
Self::Template => &DEFAULT_TRAITS,
|
||||
Self::Text => &GROUP_TRAITS,
|
||||
Self::Trait => &PRESERVE_CHILDREN_TRAITS,
|
||||
Self::Translation => &DEFAULT_TRAITS,
|
||||
Self::Try => &DEFAULT_TRAITS,
|
||||
Self::Ts => &DEFAULT_TRAITS,
|
||||
Self::Type => &ADDRESSABLE_CONTAINER_TRAITS,
|
||||
Self::Typescript => &DEFAULT_TRAITS,
|
||||
Self::Union => &DEFAULT_TRAITS,
|
||||
Self::User => &DEFAULT_TRAITS,
|
||||
Self::Val => &DEFAULT_TRAITS,
|
||||
Self::Variable => &VARIABLE_TRAITS,
|
||||
Self::Variant => &PACKED_LEAF_TRAITS,
|
||||
Self::Variants => &DEFAULT_TRAITS,
|
||||
Self::VersionGate => &DEFAULT_TRAITS,
|
||||
Self::When => &DEFAULT_TRAITS,
|
||||
Self::Where => &DEFAULT_TRAITS,
|
||||
Self::While => &DEFAULT_TRAITS,
|
||||
Self::With => &GROUP_TRAITS,
|
||||
Self::Workdir => &DEFAULT_TRAITS,
|
||||
Self::Conflict => &DEFAULT_TRAITS,
|
||||
Self::Ours => &PACKED_LEAF_TRAITS,
|
||||
Self::Theirs => &PACKED_LEAF_TRAITS,
|
||||
Self::Chunk => &DEFAULT_TRAITS,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn path_segment(self, identifier: Option<&str>) -> String {
|
||||
match identifier {
|
||||
Some(identifier) => format!("{}_{identifier}", self.prefix()),
|
||||
None => self.prefix().to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn from_sanitized_kind(kind: &str) -> Self {
|
||||
match kind {
|
||||
"add" => Self::Add,
|
||||
"after" => Self::After,
|
||||
"alias" => Self::Alias,
|
||||
"algo" => Self::Algo,
|
||||
"arg" => Self::Arg,
|
||||
"argv" => Self::Argv,
|
||||
"array" => Self::Array,
|
||||
"at" => Self::At,
|
||||
"attr" => Self::Attr,
|
||||
"attr_expr" => Self::AttrExpr,
|
||||
"attrs" => Self::Attrs,
|
||||
"block" => Self::Block,
|
||||
"block_if" => Self::BlockIf,
|
||||
"block_locals" => Self::BlockLocals,
|
||||
"body" => Self::Body,
|
||||
"case" => Self::Case,
|
||||
"cases" => Self::Cases,
|
||||
"catch" => Self::Catch,
|
||||
"cell" => Self::Cell,
|
||||
"class" => Self::Class,
|
||||
"clause" => Self::Clause,
|
||||
"cmd" => Self::Cmd,
|
||||
"code" => Self::Code,
|
||||
"cond" => Self::Cond,
|
||||
"constructor" => Self::Constructor,
|
||||
"contract" => Self::Contract,
|
||||
"copy" => Self::Copy,
|
||||
"custom" => Self::Custom,
|
||||
"decls" | "declarations" => Self::Declarations,
|
||||
"decl" => Self::Decl,
|
||||
"default_export" => Self::DefaultExport,
|
||||
"define" => Self::Define,
|
||||
"directive" => Self::Directive,
|
||||
"either" => Self::Either,
|
||||
"elif" => Self::Elif,
|
||||
"else" => Self::Else,
|
||||
"enum" => Self::Enum,
|
||||
"env" => Self::Env,
|
||||
"error" => Self::Error,
|
||||
"except" => Self::Except,
|
||||
"exports" => Self::Exports,
|
||||
"expose" => Self::Expose,
|
||||
"expr" | "expression" => Self::Expression,
|
||||
"field" => Self::Field,
|
||||
"fields" => Self::Fields,
|
||||
"file" => Self::File,
|
||||
"frame" => Self::Frame,
|
||||
"fn" | "function" => Self::Function,
|
||||
"for" => Self::For,
|
||||
"for_in" => Self::ForIn,
|
||||
"for_of" => Self::ForOf,
|
||||
"form" => Self::Form,
|
||||
"frontmatter" => Self::Frontmatter,
|
||||
"group" => Self::Group,
|
||||
"group_by" => Self::GroupBy,
|
||||
"headers" => Self::Headers,
|
||||
"healthcheck" => Self::Healthcheck,
|
||||
"html" => Self::Html,
|
||||
"hunks" => Self::Hunks,
|
||||
"hunk" => Self::Hunk,
|
||||
"if" => Self::If,
|
||||
"iface" => Self::Iface,
|
||||
"impl" => Self::Impl,
|
||||
"imports" => Self::Imports,
|
||||
"includes" => Self::Includes,
|
||||
"inline_fragment" => Self::InlineFragment,
|
||||
"install" => Self::Install,
|
||||
"interface" => Self::Interface,
|
||||
"interpolation" => Self::Interpolation,
|
||||
"item" => Self::Item,
|
||||
"join" => Self::Join,
|
||||
"key" => Self::Key,
|
||||
"key_scripts" => Self::KeyScripts,
|
||||
"label" => Self::Label,
|
||||
"let" => Self::Let,
|
||||
"list" => Self::List,
|
||||
"loop" => Self::Loop,
|
||||
"macro" => Self::Macro,
|
||||
"map" => Self::Map,
|
||||
"markdown" => Self::Markdown,
|
||||
"match" => Self::Match,
|
||||
"meth" | "method" => Self::Method,
|
||||
"methods" => Self::Methods,
|
||||
"mod" | "module" => Self::Module,
|
||||
"mustache" => Self::Mustache,
|
||||
"object" => Self::Object,
|
||||
"operation" => Self::Operation,
|
||||
"operator" => Self::Operator,
|
||||
"option" => Self::Option,
|
||||
"options" => Self::Options,
|
||||
"order_by" => Self::OrderBy,
|
||||
"params" | "parameters" => Self::Parameters,
|
||||
"preamble" => Self::Preamble,
|
||||
"proc" => Self::Proc,
|
||||
"process" => Self::Process,
|
||||
"project" => Self::Project,
|
||||
"proto" => Self::Proto,
|
||||
"py" => Self::Py,
|
||||
"python" => Self::Python,
|
||||
"query" => Self::Query,
|
||||
"receive" => Self::Receive,
|
||||
"recipe" => Self::Recipe,
|
||||
"relations" => Self::Relations,
|
||||
"render" => Self::Render,
|
||||
"ret" | "return" => Self::Return,
|
||||
"row" => Self::Row,
|
||||
"root" => Self::Root,
|
||||
"rule" => Self::Rule,
|
||||
"schema" => Self::Schema,
|
||||
"script" => Self::Script,
|
||||
"script_module" => Self::ScriptModule,
|
||||
"script_setup" => Self::ScriptSetup,
|
||||
"section" => Self::Section,
|
||||
"select" => Self::Select,
|
||||
"setting" => Self::Setting,
|
||||
"shebang" => Self::Shebang,
|
||||
"shell" => Self::Shell,
|
||||
"slot" => Self::Slot,
|
||||
"snippet" => Self::Snippet,
|
||||
"source" => Self::Source,
|
||||
"stage" => Self::Stage,
|
||||
"static_init" => Self::StaticInit,
|
||||
"stmts" | "statements" => Self::Statements,
|
||||
"struct" => Self::Struct,
|
||||
"style" => Self::Style,
|
||||
"style_scoped" => Self::StyleScoped,
|
||||
"switch" => Self::Switch,
|
||||
"table" => Self::Table,
|
||||
"tag" => Self::Tag,
|
||||
"target" => Self::Target,
|
||||
"template" => Self::Template,
|
||||
"text" => Self::Text,
|
||||
"trait" => Self::Trait,
|
||||
"translation" => Self::Translation,
|
||||
"try" => Self::Try,
|
||||
"ts" => Self::Ts,
|
||||
"type" => Self::Type,
|
||||
"typescript" => Self::Typescript,
|
||||
"union" => Self::Union,
|
||||
"user" => Self::User,
|
||||
"val" => Self::Val,
|
||||
"var" | "variable" => Self::Variable,
|
||||
"variant" => Self::Variant,
|
||||
"variants" => Self::Variants,
|
||||
"version_gate" => Self::VersionGate,
|
||||
"when" => Self::When,
|
||||
"where" => Self::Where,
|
||||
"while" => Self::While,
|
||||
"with" => Self::With,
|
||||
"workdir" => Self::Workdir,
|
||||
"conflict" => Self::Conflict,
|
||||
"ours" => Self::Ours,
|
||||
"theirs" => Self::Theirs,
|
||||
"chunk" => Self::Chunk,
|
||||
_ => Self::Chunk,
|
||||
}
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,130 +0,0 @@
|
||||
use std::{cell::RefCell, collections::HashMap};
|
||||
|
||||
use serde::Deserialize;
|
||||
|
||||
#[derive(Debug, Deserialize)]
|
||||
struct GeneratedSchema {
|
||||
languages: HashMap<String, HashMap<String, NodeTypeSchema>>,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, Deserialize)]
|
||||
pub struct NodeTypeSchema {
|
||||
pub identifier_fields: Vec<String>,
|
||||
pub body_fields: Vec<String>,
|
||||
pub promotion_fields: Vec<String>,
|
||||
pub container_child_kinds: Vec<String>,
|
||||
pub is_supertype: bool,
|
||||
pub has_structural_children: bool,
|
||||
}
|
||||
|
||||
impl NodeTypeSchema {
|
||||
pub const fn is_structural(&self) -> bool {
|
||||
self.is_supertype
|
||||
|| !self.identifier_fields.is_empty()
|
||||
|| !self.body_fields.is_empty()
|
||||
|| !self.container_child_kinds.is_empty()
|
||||
|| self.has_structural_children
|
||||
}
|
||||
}
|
||||
|
||||
thread_local! {
|
||||
static CURRENT_LANGUAGE: RefCell<Option<&'static str>> = const { RefCell::new(None) };
|
||||
}
|
||||
|
||||
static GENERATED_SCHEMA: std::sync::LazyLock<HashMap<String, HashMap<String, NodeTypeSchema>>> =
|
||||
std::sync::LazyLock::new(|| {
|
||||
let raw = include_str!(concat!(env!("OUT_DIR"), "/chunk_schema.json"));
|
||||
let generated: GeneratedSchema =
|
||||
serde_json::from_str(raw).expect("generated chunk schema should parse");
|
||||
generated.languages
|
||||
});
|
||||
|
||||
pub struct SchemaLanguageGuard {
|
||||
previous: Option<&'static str>,
|
||||
}
|
||||
|
||||
impl Drop for SchemaLanguageGuard {
|
||||
fn drop(&mut self) {
|
||||
CURRENT_LANGUAGE.with(|current| {
|
||||
*current.borrow_mut() = self.previous;
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
pub fn enter_language(language: &'static str) -> SchemaLanguageGuard {
|
||||
let previous = CURRENT_LANGUAGE.with(|current| current.replace(Some(language)));
|
||||
SchemaLanguageGuard { previous }
|
||||
}
|
||||
|
||||
pub fn current_language() -> Option<&'static str> {
|
||||
CURRENT_LANGUAGE.with(|current| *current.borrow())
|
||||
}
|
||||
|
||||
pub fn schema_for(language: &str, kind: &str) -> Option<&'static NodeTypeSchema> {
|
||||
GENERATED_SCHEMA
|
||||
.get(language)
|
||||
.and_then(|schemas| schemas.get(kind))
|
||||
}
|
||||
|
||||
pub fn schema_for_current(kind: &str) -> Option<&'static NodeTypeSchema> {
|
||||
current_language().and_then(|language| schema_for(language, kind))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub fn has_schema(language: &str) -> bool {
|
||||
GENERATED_SCHEMA.contains_key(language)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{has_schema, schema_for};
|
||||
|
||||
#[test]
|
||||
fn python_function_definition_schema_has_name_and_body() {
|
||||
let schema = schema_for("python", "function_definition")
|
||||
.expect("python function_definition schema should exist");
|
||||
assert_eq!(schema.identifier_fields, vec!["name".to_string()]);
|
||||
assert_eq!(schema.body_fields, vec!["body".to_string()]);
|
||||
assert!(schema.promotion_fields.is_empty());
|
||||
assert!(schema.is_structural());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nix_let_expression_schema_exposes_binding_set_child() {
|
||||
let schema =
|
||||
schema_for("nix", "let_expression").expect("nix let_expression schema should exist");
|
||||
assert_eq!(schema.body_fields, vec!["body".to_string()]);
|
||||
assert!(
|
||||
schema
|
||||
.container_child_kinds
|
||||
.iter()
|
||||
.any(|kind| kind == "binding_set"),
|
||||
"let_expression should surface binding_set children"
|
||||
);
|
||||
assert!(schema.has_structural_children);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wrapper_schemas_preserve_promotable_definition_fields() {
|
||||
let python = schema_for("python", "decorated_definition")
|
||||
.expect("python decorated_definition schema should exist");
|
||||
assert_eq!(python.promotion_fields, vec!["definition".to_string()]);
|
||||
|
||||
let typescript = schema_for("typescript", "export_statement")
|
||||
.expect("typescript export_statement schema should exist");
|
||||
assert!(
|
||||
typescript
|
||||
.promotion_fields
|
||||
.iter()
|
||||
.any(|field| field == "declaration"),
|
||||
"export_statement should preserve declaration field for promotion"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn generated_schema_covers_expected_languages() {
|
||||
for language in ["python", "nix", "toml", "typescript", "rust", "yaml", "handlebars"] {
|
||||
assert!(has_schema(language), "{language} should have generated schema data");
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,354 +0,0 @@
|
||||
use tree_sitter::Node;
|
||||
|
||||
use super::{atom_list, schema};
|
||||
|
||||
const IDENTIFIER_FIELD_PRIORITY: &[&str] = &[
|
||||
"name",
|
||||
"identifier",
|
||||
"attrpath",
|
||||
"key",
|
||||
"label",
|
||||
"alias",
|
||||
"field",
|
||||
"member",
|
||||
"property",
|
||||
"tag",
|
||||
"target",
|
||||
"variable",
|
||||
];
|
||||
|
||||
const BODY_FIELD_PRIORITY: &[&str] = &["body", "value", "declaration_list", "block", "members"];
|
||||
|
||||
pub fn signature_end_byte(node: Node<'_>) -> Option<usize> {
|
||||
recurse_target(node)
|
||||
.filter(|child| child.start_byte() > node.start_byte())
|
||||
.map(|child| child.start_byte())
|
||||
}
|
||||
|
||||
pub fn identifier_node(node: Node<'_>) -> Option<Node<'_>> {
|
||||
schema_identifier_node(node)
|
||||
.or_else(|| probe_field_child(node, IDENTIFIER_FIELD_PRIORITY))
|
||||
.or_else(|| {
|
||||
local_named_children(node)
|
||||
.into_iter()
|
||||
.find(|child| looks_like_identifier_kind(child.kind()))
|
||||
})
|
||||
}
|
||||
|
||||
pub fn recurse_target(node: Node<'_>) -> Option<Node<'_>> {
|
||||
schema_body_child(node)
|
||||
.or_else(|| schema_container_child(node))
|
||||
.or_else(|| probe_field_child(node, BODY_FIELD_PRIORITY))
|
||||
.or_else(|| probe_structural_child(node))
|
||||
.and_then(descend_to_structural_target)
|
||||
}
|
||||
|
||||
pub fn value_container_target(node: Node<'_>) -> Option<Node<'_>> {
|
||||
for field in ["value", "body"] {
|
||||
if let Some(child) = node.child_by_field_name(field)
|
||||
&& let Some(target) = descend_to_structural_target(child)
|
||||
{
|
||||
return Some(target);
|
||||
}
|
||||
}
|
||||
|
||||
recurse_target(node)
|
||||
}
|
||||
|
||||
pub fn is_root_wrapper_node(node: Node<'_>) -> bool {
|
||||
if is_atom_node(node) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if let Some(schema) = schema::schema_for_current(node.kind()) {
|
||||
if schema.is_supertype {
|
||||
return true;
|
||||
}
|
||||
if schema.is_structural() && !is_transparent_recurse_wrapper(node) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
if identifier_node(node).is_some() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let non_trivia = local_named_children(node)
|
||||
.into_iter()
|
||||
.filter(|child| !is_generic_trivia(*child))
|
||||
.collect::<Vec<_>>();
|
||||
non_trivia.len() == 1 && node_looks_structural(non_trivia[0])
|
||||
}
|
||||
|
||||
pub fn is_generic_trivia(node: Node<'_>) -> bool {
|
||||
node.is_extra() || kind_looks_like_comment(node.kind())
|
||||
}
|
||||
|
||||
pub fn is_generic_absorbable_attr(kind: &str) -> bool {
|
||||
matches!(kind, "attribute_item" | "inner_attribute_item")
|
||||
}
|
||||
|
||||
pub fn is_atom_node(node: Node<'_>) -> bool {
|
||||
atom_list::is_atom_node_current(node.kind())
|
||||
}
|
||||
|
||||
fn schema_identifier_node(node: Node<'_>) -> Option<Node<'_>> {
|
||||
let schema = schema::schema_for_current(node.kind())?;
|
||||
for field in &schema.identifier_fields {
|
||||
if let Some(child) = node.child_by_field_name(field) {
|
||||
return Some(child);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn schema_body_child(node: Node<'_>) -> Option<Node<'_>> {
|
||||
let schema = schema::schema_for_current(node.kind())?;
|
||||
for field in &schema.body_fields {
|
||||
if let Some(child) = node.child_by_field_name(field) {
|
||||
return Some(child);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn schema_container_child(node: Node<'_>) -> Option<Node<'_>> {
|
||||
let schema = schema::schema_for_current(node.kind())?;
|
||||
local_named_children(node).into_iter().find(|child| {
|
||||
schema
|
||||
.container_child_kinds
|
||||
.iter()
|
||||
.any(|kind| child.kind() == kind)
|
||||
})
|
||||
}
|
||||
|
||||
fn probe_field_child<'tree>(node: Node<'tree>, fields: &[&str]) -> Option<Node<'tree>> {
|
||||
fields
|
||||
.iter()
|
||||
.find_map(|field| node.child_by_field_name(field))
|
||||
}
|
||||
|
||||
fn probe_structural_child(node: Node<'_>) -> Option<Node<'_>> {
|
||||
local_named_children(node).into_iter().find(|child| {
|
||||
!is_generic_trivia(*child)
|
||||
&& !looks_like_identifier_kind(child.kind())
|
||||
&& !is_atom_node(*child)
|
||||
&& node_looks_structural(*child)
|
||||
})
|
||||
}
|
||||
|
||||
fn descend_to_structural_target(node: Node<'_>) -> Option<Node<'_>> {
|
||||
if is_atom_node(node) {
|
||||
return None;
|
||||
}
|
||||
|
||||
if is_transparent_recurse_wrapper(node) {
|
||||
let child = local_named_children(node)
|
||||
.into_iter()
|
||||
.find(|child| !is_generic_trivia(*child) && node_looks_structural(*child))?;
|
||||
return descend_to_structural_target(child).or(Some(child));
|
||||
}
|
||||
|
||||
if node_looks_structural(node) {
|
||||
return Some(node);
|
||||
}
|
||||
|
||||
probe_structural_child(node).and_then(descend_to_structural_target)
|
||||
}
|
||||
|
||||
fn is_transparent_recurse_wrapper(node: Node<'_>) -> bool {
|
||||
if is_atom_node(node) || identifier_node(node).is_some() {
|
||||
return false;
|
||||
}
|
||||
|
||||
schema::schema_for_current(node.kind()).is_some_and(|schema| schema.is_supertype)
|
||||
|| matches!(node.kind(), "document" | "stream")
|
||||
|| node.kind().ends_with("_node")
|
||||
}
|
||||
|
||||
fn node_looks_structural(node: Node<'_>) -> bool {
|
||||
if is_atom_node(node) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if schema::schema_for_current(node.kind()).is_some_and(|schema| schema.is_structural()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
let non_trivia_children = local_named_children(node)
|
||||
.into_iter()
|
||||
.filter(|child| !is_generic_trivia(*child))
|
||||
.collect::<Vec<_>>();
|
||||
!non_trivia_children.is_empty()
|
||||
&& non_trivia_children
|
||||
.iter()
|
||||
.any(|child| !looks_like_identifier_kind(child.kind()) || child.named_child_count() > 0)
|
||||
}
|
||||
|
||||
fn local_named_children(node: Node<'_>) -> Vec<Node<'_>> {
|
||||
let mut children = Vec::new();
|
||||
for index in 0..node.child_count() {
|
||||
if let Some(child) = node.child(index)
|
||||
&& (child.is_named() || child.is_error() || child.kind() == "ERROR")
|
||||
{
|
||||
children.push(child);
|
||||
}
|
||||
}
|
||||
children
|
||||
}
|
||||
|
||||
fn looks_like_identifier_kind(kind: &str) -> bool {
|
||||
kind == "identifier"
|
||||
|| kind == "name"
|
||||
|| kind.ends_with("_identifier")
|
||||
|| kind.ends_with("_name")
|
||||
|| kind.ends_with("_label")
|
||||
}
|
||||
|
||||
fn kind_looks_like_comment(kind: &str) -> bool {
|
||||
kind == "comment" || kind.contains("comment")
|
||||
}
|
||||
|
||||
/// Detect a call-with-trailing-callback pattern inside a node.
|
||||
///
|
||||
/// Returns `(call_text_node, body_node)` where `call_text_node` is the function
|
||||
/// being called (for name extraction) and `body_node` is the block body of the
|
||||
/// trailing callback argument.
|
||||
///
|
||||
/// This handles patterns like:
|
||||
/// - JS/TS: `describe('x', () => { ... })`, `app.use(handler)`
|
||||
/// - Go: `t.Run("x", func(t *testing.T) { ... })`
|
||||
/// - Rust: `tokio::spawn(async { ... })`
|
||||
/// - Ruby: `describe 'x' do ... end` (via `do_block`)
|
||||
///
|
||||
/// Only matches when the last argument has a structural block body, so
|
||||
/// expression-body arrows and simple value arguments are not promoted.
|
||||
pub fn trailing_callback_body(node: Node<'_>) -> Option<(Node<'_>, Node<'_>)> {
|
||||
// Look through the node's children for a call-like node.
|
||||
let call = local_named_children(node)
|
||||
.into_iter()
|
||||
.find(|c| is_call_like(*c))?;
|
||||
|
||||
// Find the arguments / parameter list.
|
||||
let args_node = call
|
||||
.child_by_field_name("arguments")
|
||||
.or_else(|| call.child_by_field_name("args"))
|
||||
.or_else(|| call.child_by_field_name("block"))?;
|
||||
|
||||
// Get the last named child of the arguments list.
|
||||
let last_arg = local_named_children(args_node).into_iter().last()?;
|
||||
|
||||
// Try to find a structural body inside the last argument.
|
||||
let body = recurse_target(last_arg)?;
|
||||
|
||||
// Extract the call target node for name purposes.
|
||||
let func_node = call
|
||||
.child_by_field_name("function")
|
||||
.or_else(|| call.child_by_field_name("method"))
|
||||
.or_else(|| call.child_by_field_name("name"))
|
||||
.unwrap_or(call);
|
||||
|
||||
Some((func_node, body))
|
||||
}
|
||||
|
||||
/// Returns `true` if the node looks like a function/method call.
|
||||
fn is_call_like(node: Node<'_>) -> bool {
|
||||
let kind = node.kind();
|
||||
// Explicit known call kinds.
|
||||
if matches!(
|
||||
kind,
|
||||
"call_expression"
|
||||
| "call"
|
||||
| "function_call"
|
||||
| "method_call"
|
||||
| "method_call_expression"
|
||||
| "invocation_expression"
|
||||
) {
|
||||
return true;
|
||||
}
|
||||
// Heuristic: has an `arguments` field.
|
||||
node.child_by_field_name("arguments").is_some()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use ast_grep_core::tree_sitter::LanguageExt;
|
||||
use tree_sitter::{Node, Parser};
|
||||
|
||||
use super::{
|
||||
identifier_node, is_root_wrapper_node, recurse_target, signature_end_byte,
|
||||
trailing_callback_body,
|
||||
};
|
||||
use crate::language::SupportLang;
|
||||
|
||||
fn parse_tree_with_language(
|
||||
source: &str,
|
||||
language: SupportLang,
|
||||
) -> (crate::chunk::schema::SchemaLanguageGuard, tree_sitter::Tree) {
|
||||
let schema_language = crate::chunk::schema::enter_language(language.canonical_name());
|
||||
let mut parser = Parser::new();
|
||||
parser
|
||||
.set_language(&language.get_ts_language())
|
||||
.expect("language should parse");
|
||||
let tree = parser.parse(source, None).expect("tree should parse");
|
||||
(schema_language, tree)
|
||||
}
|
||||
|
||||
fn find_named_node<'tree>(node: Node<'tree>, kind: &str) -> Option<Node<'tree>> {
|
||||
if node.kind() == kind {
|
||||
return Some(node);
|
||||
}
|
||||
for index in 0..node.child_count() {
|
||||
let Some(child) = node.child(index) else {
|
||||
continue;
|
||||
};
|
||||
if let Some(found) = find_named_node(child, kind) {
|
||||
return Some(found);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn python_function_definition_resolves_identifier_and_body() {
|
||||
let (_schema_language, tree) =
|
||||
parse_tree_with_language("def greet(name):\n return name\n", SupportLang::Python);
|
||||
let node =
|
||||
find_named_node(tree.root_node(), "function_definition").expect("function_definition");
|
||||
assert_eq!(identifier_node(node).expect("name").kind(), "identifier");
|
||||
assert_eq!(recurse_target(node).expect("body").kind(), "block");
|
||||
assert!(signature_end_byte(node).is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn typescript_class_declaration_resolves_body() {
|
||||
let (_schema_language, tree) =
|
||||
parse_tree_with_language("class Greeter {\n hello() {}\n}\n", SupportLang::TypeScript);
|
||||
let node = find_named_node(tree.root_node(), "class_declaration").expect("class_declaration");
|
||||
assert_eq!(recurse_target(node).expect("body").kind(), "class_body");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn yaml_wrapper_nodes_are_structural_without_shared_kind_lists() {
|
||||
let (_schema_language, tree) =
|
||||
parse_tree_with_language("root:\n child: 1\n", SupportLang::Yaml);
|
||||
let node = find_named_node(tree.root_node(), "block_node").expect("block_node");
|
||||
assert!(is_root_wrapper_node(node), "block_node should be treated as a structural wrapper");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn js_expression_statement_with_callback_detects_body() {
|
||||
let source = "describe(\"suite\", () => {\n\tit(\"a\", () => {});\n});\n";
|
||||
let (_schema_language, tree) = parse_tree_with_language(source, SupportLang::TypeScript);
|
||||
let expr_stmt =
|
||||
find_named_node(tree.root_node(), "expression_statement").expect("expression_statement");
|
||||
let (func_node, body) =
|
||||
trailing_callback_body(expr_stmt).expect("should detect trailing callback body");
|
||||
assert_eq!(body.kind(), "statement_block");
|
||||
assert!(
|
||||
matches!(func_node.kind(), "identifier" | "member_expression"),
|
||||
"func_node should be an identifier or member_expression, got {}",
|
||||
func_node.kind()
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1,689 +0,0 @@
|
||||
use std::{
|
||||
collections::HashMap,
|
||||
sync::{Arc, LazyLock},
|
||||
};
|
||||
|
||||
use napi::{Error, Result};
|
||||
use napi_derive::napi;
|
||||
use regex::Regex;
|
||||
|
||||
use super::{
|
||||
build_chunk_tree,
|
||||
indent::{detect_file_indent_char, detect_file_indent_step, normalize_to_tabs},
|
||||
resolve::{
|
||||
ParsedSelector, chunk_region_range, format_region_ref, format_selector_tree,
|
||||
resolve_chunk_selector, resolve_chunk_with_crc, split_selector_crc_and_region,
|
||||
},
|
||||
};
|
||||
use crate::chunk::types::{
|
||||
ChunkInfo, ChunkNode, ChunkReadStatus, ChunkReadTarget, ChunkRegion, ChunkTree, EditParams,
|
||||
EditResult, ReadRenderParams, ReadResult, RenderParams, VisibleLineRange,
|
||||
};
|
||||
|
||||
const LINE_RANGE_SELECTOR_RE: &str = r"^L(\d+)(?:-L?(\d+))?$";
|
||||
const TLAPLUS_BEGIN_TRANSLATION_RE: &str = r"^\s*\\\*\s*BEGIN TRANSLATION\s*$";
|
||||
const TLAPLUS_END_TRANSLATION_RE: &str = r"^\s*\\\*\s*END TRANSLATION\s*$";
|
||||
|
||||
static LINE_RANGE_SELECTOR_REGEX: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(LINE_RANGE_SELECTOR_RE).expect("line range regex must compile"));
|
||||
static TLAPLUS_BEGIN_TRANSLATION_REGEX: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(TLAPLUS_BEGIN_TRANSLATION_RE).expect("tlaplus begin regex must compile")
|
||||
});
|
||||
static TLAPLUS_END_TRANSLATION_REGEX: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(TLAPLUS_END_TRANSLATION_RE).expect("tlaplus end regex must compile")
|
||||
});
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct ConflictMeta {
|
||||
pub theirs_content: String,
|
||||
pub ours_label: String,
|
||||
pub theirs_label: String,
|
||||
pub base_content: Option<String>,
|
||||
pub base_label: Option<String>,
|
||||
pub ours_start_byte: usize,
|
||||
pub ours_end_byte: usize,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct ChunkStateInner {
|
||||
pub(crate) source: String,
|
||||
pub(crate) language: String,
|
||||
pub(crate) tree: ChunkTree,
|
||||
pub(crate) notebook: Option<crate::chunk::ast_ipynb::SharedNotebookContext>,
|
||||
pub(crate) conflict_meta: HashMap<String, ConflictMeta>,
|
||||
lookup: HashMap<String, usize>,
|
||||
checksum_lookup: HashMap<String, Vec<usize>>,
|
||||
leaf_lookup: HashMap<String, Vec<usize>>,
|
||||
suffix_lookup: HashMap<String, Vec<usize>>,
|
||||
}
|
||||
|
||||
impl ChunkStateInner {
|
||||
pub(crate) fn parse(source: String, language: String) -> Result<Self> {
|
||||
let normalized_language = normalize_language(language.as_str());
|
||||
if normalized_language == "ipynb" {
|
||||
let parsed =
|
||||
crate::chunk::ast_ipynb::parse_notebook(&source).map_err(napi::Error::from_reason)?;
|
||||
let kernel_lang = parsed.context.kernel_language.clone();
|
||||
let tree = crate::chunk::ast_ipynb::build_notebook_tree_from_virtual(
|
||||
parsed.virtual_source.as_str(),
|
||||
kernel_lang.as_str(),
|
||||
)
|
||||
.map_err(napi::Error::from_reason)?;
|
||||
let ctx = std::sync::Arc::new(parsed.context);
|
||||
let mut inner = Self::new(parsed.virtual_source, normalized_language, tree);
|
||||
inner.notebook = Some(ctx);
|
||||
return Ok(inner);
|
||||
}
|
||||
if crate::chunk::conflict::has_conflict_markers(source.as_str()) {
|
||||
let conflicts = crate::chunk::conflict::detect_conflicts(source.as_str());
|
||||
if !conflicts.is_empty() {
|
||||
let clean_result = crate::chunk::conflict::accept_ours(source.as_str(), &conflicts);
|
||||
let mut tree =
|
||||
build_chunk_tree(clean_result.source.as_str(), normalized_language.as_str())?;
|
||||
let conflict_meta = crate::chunk::conflict::inject_conflict_chunks(
|
||||
&mut tree,
|
||||
clean_result.source.as_str(),
|
||||
&clean_result,
|
||||
);
|
||||
let mut inner = Self::new(clean_result.source, normalized_language, tree);
|
||||
inner.conflict_meta = conflict_meta;
|
||||
return Ok(inner);
|
||||
}
|
||||
}
|
||||
let tree = build_chunk_tree(source.as_str(), normalized_language.as_str())?;
|
||||
Ok(Self::new(source, normalized_language, tree))
|
||||
}
|
||||
|
||||
pub(crate) fn new(source: String, language: String, tree: ChunkTree) -> Self {
|
||||
let mut lookup = HashMap::new();
|
||||
let mut checksum_lookup = HashMap::new();
|
||||
let mut leaf_lookup = HashMap::new();
|
||||
let mut suffix_lookup = HashMap::new();
|
||||
for (index, chunk) in tree.chunks.iter().enumerate() {
|
||||
lookup.insert(chunk.path.clone(), index);
|
||||
checksum_lookup
|
||||
.entry(chunk.checksum.clone())
|
||||
.or_insert_with(Vec::new)
|
||||
.push(index);
|
||||
if chunk.path.is_empty() {
|
||||
continue;
|
||||
}
|
||||
if let Some(leaf) = chunk.path.rsplit('.').next() {
|
||||
leaf_lookup
|
||||
.entry(leaf.to_string())
|
||||
.or_insert_with(Vec::new)
|
||||
.push(index);
|
||||
}
|
||||
let segments = chunk.path.split('.').collect::<Vec<_>>();
|
||||
for start in 1..segments.len() {
|
||||
suffix_lookup
|
||||
.entry(segments[start..].join("."))
|
||||
.or_insert_with(Vec::new)
|
||||
.push(index);
|
||||
}
|
||||
}
|
||||
Self {
|
||||
source,
|
||||
language,
|
||||
tree,
|
||||
notebook: None,
|
||||
conflict_meta: HashMap::new(),
|
||||
lookup,
|
||||
checksum_lookup,
|
||||
leaf_lookup,
|
||||
suffix_lookup,
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) const fn source(&self) -> &str {
|
||||
self.source.as_str()
|
||||
}
|
||||
|
||||
pub(crate) const fn language(&self) -> &str {
|
||||
self.language.as_str()
|
||||
}
|
||||
|
||||
pub(crate) const fn tree(&self) -> &ChunkTree {
|
||||
&self.tree
|
||||
}
|
||||
|
||||
pub(crate) fn root(&self) -> Option<&ChunkNode> {
|
||||
self.chunk("")
|
||||
}
|
||||
|
||||
pub(crate) fn chunk(&self, path: &str) -> Option<&ChunkNode> {
|
||||
self
|
||||
.lookup
|
||||
.get(path)
|
||||
.and_then(|index| self.tree.chunks.get(*index))
|
||||
}
|
||||
|
||||
pub(crate) fn chunk_by_index(&self, index: usize) -> Option<&ChunkNode> {
|
||||
self.tree.chunks.get(index)
|
||||
}
|
||||
|
||||
pub(crate) fn chunks_by_checksum(&self, checksum: &str) -> Vec<&ChunkNode> {
|
||||
self
|
||||
.checksum_lookup
|
||||
.get(checksum)
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|index| self.chunk_by_index(*index))
|
||||
.collect()
|
||||
}
|
||||
|
||||
pub(crate) fn chunks_by_leaf(&self, leaf: &str) -> Vec<&ChunkNode> {
|
||||
self
|
||||
.leaf_lookup
|
||||
.get(leaf)
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|index| self.chunk_by_index(*index))
|
||||
.collect()
|
||||
}
|
||||
|
||||
pub(crate) fn chunks_by_suffix(&self, suffix: &str) -> Vec<&ChunkNode> {
|
||||
self
|
||||
.suffix_lookup
|
||||
.get(suffix)
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|index| self.chunk_by_index(*index))
|
||||
.collect()
|
||||
}
|
||||
|
||||
pub(crate) fn chunks(&self) -> impl Iterator<Item = &ChunkNode> {
|
||||
self.tree.chunks.iter()
|
||||
}
|
||||
|
||||
pub(crate) fn child_chunks(&self, parent_path: &str) -> Vec<&ChunkNode> {
|
||||
self
|
||||
.chunk(parent_path)
|
||||
.map(|parent| {
|
||||
parent
|
||||
.children
|
||||
.iter()
|
||||
.filter_map(|path| self.chunk(path.as_str()))
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
pub(crate) fn line_to_containing_chunk_path(&self, line: u32) -> Option<String> {
|
||||
crate::chunk::line_to_chunk_path(&self.tree, line)
|
||||
}
|
||||
}
|
||||
|
||||
/// Parsed file as a chunk tree: query nodes, render views, format grep hits,
|
||||
/// and apply edits.
|
||||
#[napi]
|
||||
#[derive(Clone)]
|
||||
pub struct ChunkState {
|
||||
inner: Arc<ChunkStateInner>,
|
||||
}
|
||||
|
||||
impl ChunkState {
|
||||
pub(crate) fn from_inner(inner: ChunkStateInner) -> Self {
|
||||
Self { inner: Arc::new(inner) }
|
||||
}
|
||||
|
||||
pub(crate) fn inner(&self) -> &ChunkStateInner {
|
||||
self.inner.as_ref()
|
||||
}
|
||||
}
|
||||
|
||||
#[napi]
|
||||
impl ChunkState {
|
||||
/// Build chunk state by parsing `source` with the given `language` id (e.g.
|
||||
/// `typescript`).
|
||||
#[napi(factory)]
|
||||
pub fn parse(source: String, language: String) -> Result<Self> {
|
||||
ChunkStateInner::parse(source, language).map(Self::from_inner)
|
||||
}
|
||||
|
||||
/// Normalized language identifier used for the tree-sitter parse.
|
||||
#[napi(getter)]
|
||||
pub fn language(&self) -> String {
|
||||
self.inner.language().to_string()
|
||||
}
|
||||
|
||||
/// Full source text for this file.
|
||||
#[napi(getter)]
|
||||
pub fn source(&self) -> String {
|
||||
self.inner.source().to_string()
|
||||
}
|
||||
|
||||
/// Stable checksum for the entire file contents.
|
||||
#[napi(getter)]
|
||||
pub fn checksum(&self) -> String {
|
||||
self.inner.tree().checksum.clone()
|
||||
}
|
||||
|
||||
/// Line count of the source buffer.
|
||||
#[napi(getter)]
|
||||
pub fn line_count(&self) -> u32 {
|
||||
self.inner.tree().line_count
|
||||
}
|
||||
|
||||
/// Count of tree-sitter error nodes seen while building the tree.
|
||||
#[napi(getter)]
|
||||
pub fn parse_errors(&self) -> u32 {
|
||||
self.inner.tree().parse_errors
|
||||
}
|
||||
|
||||
/// True when a fallback classifier produced the tree.
|
||||
#[napi(getter)]
|
||||
pub fn fallback(&self) -> bool {
|
||||
self.inner.tree().fallback
|
||||
}
|
||||
|
||||
/// Selector path string for the synthetic root (often empty).
|
||||
#[napi(getter)]
|
||||
pub fn root_path(&self) -> String {
|
||||
self.inner.tree().root_path.clone()
|
||||
}
|
||||
|
||||
/// Top-level child chunk paths under the root.
|
||||
#[napi(getter)]
|
||||
pub fn root_children(&self) -> Vec<String> {
|
||||
self.inner.tree().root_children.clone()
|
||||
}
|
||||
|
||||
/// Total number of chunk nodes.
|
||||
#[napi(getter)]
|
||||
pub fn chunk_count(&self) -> u32 {
|
||||
self.inner.tree().chunks.len() as u32
|
||||
}
|
||||
|
||||
/// True when the parsed file contains unresolved merge conflicts.
|
||||
#[napi]
|
||||
pub fn has_conflicts(&self) -> bool {
|
||||
!self.inner.conflict_meta.is_empty()
|
||||
}
|
||||
|
||||
/// Count of unresolved merge conflicts represented in the chunk tree.
|
||||
#[napi]
|
||||
pub fn conflict_count(&self) -> u32 {
|
||||
self.inner.conflict_meta.len() as u32
|
||||
}
|
||||
|
||||
/// Summary for the root chunk, if it exists.
|
||||
#[napi]
|
||||
pub fn root(&self) -> Option<ChunkInfo> {
|
||||
self.inner.root().map(chunk_info)
|
||||
}
|
||||
|
||||
/// Look up [`ChunkInfo`] for a chunk selector path.
|
||||
#[napi]
|
||||
pub fn chunk(&self, chunk_path: String) -> Option<ChunkInfo> {
|
||||
let mut warnings = Vec::new();
|
||||
resolve_chunk_selector(self.inner(), Some(chunk_path.as_str()), &mut warnings)
|
||||
.ok()
|
||||
.map(chunk_info)
|
||||
}
|
||||
|
||||
/// Every chunk node as a [`ChunkInfo`] list.
|
||||
#[napi]
|
||||
pub fn chunks(&self) -> Vec<ChunkInfo> {
|
||||
self.inner.chunks().map(chunk_info).collect()
|
||||
}
|
||||
|
||||
/// Direct children of `chunkPath` (use empty or omit for root); errors if
|
||||
/// the path is missing.
|
||||
#[napi]
|
||||
pub fn children(&self, chunk_path: Option<String>) -> Result<Vec<ChunkInfo>> {
|
||||
let parent = if let Some(chunk_path) = chunk_path {
|
||||
let mut warnings = Vec::new();
|
||||
resolve_chunk_selector(self.inner(), Some(chunk_path.as_str()), &mut warnings)
|
||||
.map_err(Error::from_reason)?
|
||||
} else {
|
||||
self
|
||||
.inner
|
||||
.root()
|
||||
.ok_or_else(|| Error::from_reason("Chunk tree is missing the root chunk".to_string()))?
|
||||
};
|
||||
Ok(self
|
||||
.inner
|
||||
.child_chunks(parent.path.as_str())
|
||||
.into_iter()
|
||||
.map(chunk_info)
|
||||
.collect())
|
||||
}
|
||||
|
||||
/// Chunk selector path that contains 1-based source line `line`, if any.
|
||||
#[napi]
|
||||
pub fn line_to_containing_chunk_path(&self, line: u32) -> Option<String> {
|
||||
self.inner.line_to_containing_chunk_path(line)
|
||||
}
|
||||
|
||||
/// Render a chunk subtree or listing as UTF-8 text for tools.
|
||||
#[napi]
|
||||
pub fn render(&self, params: RenderParams) -> String {
|
||||
crate::chunk::render::render_state(self.inner(), ¶ms)
|
||||
}
|
||||
|
||||
/// Parse `readPath` (selector, line scope, etc.) and return rendered text or
|
||||
/// errors.
|
||||
#[napi]
|
||||
pub fn render_read(&self, params: ReadRenderParams) -> Result<ReadResult> {
|
||||
let ParsedChunkReadPath { selector, crc, region } =
|
||||
match parse_chunk_read_path(params.read_path.as_str()) {
|
||||
Ok(parsed) => parsed,
|
||||
Err(err) => {
|
||||
return Ok(ReadResult {
|
||||
text: format!("{}\n\n{}", params.display_path, err),
|
||||
chunk: Some(ChunkReadTarget {
|
||||
status: ChunkReadStatus::UnsupportedRegion,
|
||||
selector: params.read_path.clone(),
|
||||
}),
|
||||
});
|
||||
},
|
||||
};
|
||||
let visible_range = selector.as_deref().and_then(parse_visible_line_range);
|
||||
let Some(root) = self.inner.root() else {
|
||||
return Ok(ReadResult {
|
||||
text: format!("{}\n\n[Chunk tree root missing]", params.display_path),
|
||||
chunk: None,
|
||||
});
|
||||
};
|
||||
|
||||
if let Some(visible_range) = visible_range {
|
||||
if visible_range.start_line > self.inner.tree().line_count {
|
||||
let suggestion = if self.inner.tree().line_count == 0 {
|
||||
"The file is empty.".to_string()
|
||||
} else {
|
||||
format!(
|
||||
"Use sel=L1 to read from the start, or sel=L{} to read the last line.",
|
||||
self.inner.tree().line_count
|
||||
)
|
||||
};
|
||||
return Ok(ReadResult {
|
||||
text: format!(
|
||||
"Line {} is beyond end of file ({} lines total). {suggestion}",
|
||||
visible_range.start_line,
|
||||
self.inner.tree().line_count,
|
||||
),
|
||||
chunk: None,
|
||||
});
|
||||
}
|
||||
|
||||
let clamped_range = VisibleLineRange {
|
||||
start_line: visible_range.start_line,
|
||||
end_line: visible_range.end_line.min(self.inner.tree().line_count),
|
||||
};
|
||||
let notice = format!(
|
||||
"[Notice: chunk view scoped to requested lines L{}-L{}; clipped chunks keep head/tail \
|
||||
context and collapse non-overlapping children.]",
|
||||
clamped_range.start_line, clamped_range.end_line
|
||||
);
|
||||
let text = self.render(RenderParams {
|
||||
chunk_path: Some(root.path.clone()),
|
||||
title: params.display_path.clone(),
|
||||
language_tag: params.language_tag.clone(),
|
||||
visible_range: Some(clamped_range),
|
||||
render_children_only: true,
|
||||
omit_checksum: params.omit_checksum,
|
||||
anchor_style: params.anchor_style,
|
||||
show_leaf_preview: true,
|
||||
tab_replacement: params.tab_replacement,
|
||||
normalize_indent: params.normalize_indent,
|
||||
focused_paths: None,
|
||||
});
|
||||
return Ok(ReadResult { text: format!("{notice}\n\n{text}"), chunk: None });
|
||||
}
|
||||
|
||||
if selector.as_deref().is_none_or(str::is_empty) && crc.is_none() && region.is_none() {
|
||||
return Ok(ReadResult {
|
||||
text: self.render(RenderParams {
|
||||
chunk_path: Some(root.path.clone()),
|
||||
title: params.display_path.clone(),
|
||||
language_tag: params.language_tag.clone(),
|
||||
visible_range: None,
|
||||
render_children_only: true,
|
||||
omit_checksum: params.omit_checksum,
|
||||
anchor_style: params.anchor_style,
|
||||
show_leaf_preview: true,
|
||||
tab_replacement: params.tab_replacement,
|
||||
normalize_indent: params.normalize_indent,
|
||||
focused_paths: None,
|
||||
}),
|
||||
chunk: None,
|
||||
});
|
||||
}
|
||||
|
||||
if selector.as_deref() == Some("?") {
|
||||
let mut lines = vec![format!("{} chunks (dot-joined paths):", params.display_path)];
|
||||
lines.extend(format_selector_tree(
|
||||
self.inner.tree(),
|
||||
&self.inner.tree().root_children,
|
||||
false,
|
||||
));
|
||||
return Ok(ReadResult { text: lines.join("\n"), chunk: None });
|
||||
}
|
||||
|
||||
let mut warnings = Vec::new();
|
||||
let resolved = match resolve_chunk_with_crc(
|
||||
self.inner(),
|
||||
selector.as_deref(),
|
||||
crc.as_deref(),
|
||||
&mut warnings,
|
||||
) {
|
||||
Ok(resolved) => resolved,
|
||||
Err(err) => {
|
||||
let sel = selector.unwrap_or_default();
|
||||
return Ok(ReadResult {
|
||||
text: format!("{}:{}\n\n{}", params.display_path, sel, err),
|
||||
chunk: Some(ChunkReadTarget { status: ChunkReadStatus::NotFound, selector: sel }),
|
||||
});
|
||||
},
|
||||
};
|
||||
let chunk = resolved.chunk;
|
||||
// Use the region from parse_chunk_read_path, NOT region,
|
||||
// because resolve_chunk_with_crc re-parses the already-cleaned
|
||||
// selector and loses the region suffix.
|
||||
let selector_ref = format_region_ref(chunk, region);
|
||||
|
||||
// Fall back to whole-chunk read when the chunk has no real region
|
||||
// boundaries (e.g. markdown fenced blocks, leaf chunks without
|
||||
// prologue/epilogue). Without this, `^` on such chunks returns
|
||||
// "[Empty @^ region]" instead of the whole chunk.
|
||||
let region = if region.is_some()
|
||||
&& (chunk.prologue_end_byte.is_none() || chunk.epilogue_start_byte.is_none())
|
||||
{
|
||||
None
|
||||
} else {
|
||||
region
|
||||
};
|
||||
|
||||
if let Some(absolute_line_range) = params.absolute_line_range {
|
||||
let req_start = absolute_line_range.start_line;
|
||||
let req_end = absolute_line_range.end_line;
|
||||
let low = chunk.start_line.max(req_start.min(req_end));
|
||||
let high = chunk.end_line.min(req_start.max(req_end));
|
||||
if low > high {
|
||||
let requested = if req_start == req_end {
|
||||
format!("L{req_start}")
|
||||
} else {
|
||||
format!("L{req_start}-L{req_end}")
|
||||
};
|
||||
return Ok(ReadResult {
|
||||
text: format!(
|
||||
"Requested range {requested} does not overlap {}:{} (lines {}-{}).",
|
||||
params.display_path, chunk.path, chunk.start_line, chunk.end_line
|
||||
),
|
||||
chunk: Some(ChunkReadTarget {
|
||||
status: ChunkReadStatus::Ok,
|
||||
selector: selector_ref,
|
||||
}),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(target_region) = region {
|
||||
let masked_source = mask_chunk_display_source(self.inner.source(), self.inner.language());
|
||||
let (start, end) = chunk_region_range(chunk, target_region);
|
||||
let tab_replacement = params.tab_replacement.as_deref().unwrap_or(" ");
|
||||
let normalize_indent = params.normalize_indent.unwrap_or(false).then(|| {
|
||||
(
|
||||
detect_file_indent_char(self.inner.source(), self.inner.tree()),
|
||||
detect_file_indent_step(self.inner.source(), self.inner.tree()) as usize,
|
||||
)
|
||||
});
|
||||
// Extend the region start to the beginning of the line so that the
|
||||
// leading indentation of the first line is included. Without this,
|
||||
// regions whose start_byte is mid-line (e.g. a decorator `@property`
|
||||
// inside a class) would show the first line without indentation,
|
||||
// making the normalization inconsistent with subsequent lines.
|
||||
let display_start = masked_source[..start].rfind('\n').map_or(0, |nl| nl + 1);
|
||||
let region_text = masked_source
|
||||
.get(display_start..end)
|
||||
.unwrap_or_default()
|
||||
.split('\n')
|
||||
.map(|line| match normalize_indent {
|
||||
Some((indent_char, indent_step)) => {
|
||||
normalize_to_tabs(line, indent_char, indent_step)
|
||||
},
|
||||
None => line.replace('\t', tab_replacement),
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n");
|
||||
let text = if region_text.is_empty() {
|
||||
format!("{selector_ref}\n\n[Empty @{} region]", target_region.as_str())
|
||||
} else {
|
||||
format!("{selector_ref}\n\n{region_text}")
|
||||
};
|
||||
return Ok(ReadResult {
|
||||
text,
|
||||
chunk: Some(ChunkReadTarget { status: ChunkReadStatus::Ok, selector: selector_ref }),
|
||||
});
|
||||
}
|
||||
|
||||
Ok(ReadResult {
|
||||
text: self.render(RenderParams {
|
||||
chunk_path: Some(chunk.path.clone()),
|
||||
title: format!("{}:{}", params.display_path, chunk.path),
|
||||
language_tag: params.language_tag.clone(),
|
||||
visible_range: None,
|
||||
render_children_only: false,
|
||||
omit_checksum: params.omit_checksum,
|
||||
anchor_style: params.anchor_style,
|
||||
show_leaf_preview: true,
|
||||
tab_replacement: params.tab_replacement,
|
||||
normalize_indent: params.normalize_indent,
|
||||
focused_paths: None,
|
||||
}),
|
||||
chunk: Some(ChunkReadTarget { status: ChunkReadStatus::Ok, selector: selector_ref }),
|
||||
})
|
||||
}
|
||||
|
||||
/// Prefix a grep line with `display_path` and the chunk path for
|
||||
/// `line_number`, when known.
|
||||
#[napi]
|
||||
pub fn format_grep_line(&self, display_path: String, line_number: u32, line: String) -> String {
|
||||
let chunk_path = self.inner.line_to_containing_chunk_path(line_number);
|
||||
let location = chunk_path.map_or_else(
|
||||
|| display_path.clone(),
|
||||
|chunk_path| {
|
||||
if chunk_path.is_empty() {
|
||||
display_path.clone()
|
||||
} else {
|
||||
format!("{display_path}:{chunk_path}")
|
||||
}
|
||||
},
|
||||
);
|
||||
format!("{location}>{line_number}|{line}")
|
||||
}
|
||||
|
||||
/// Apply batch edits, re-parse, write files, and return updated state and
|
||||
/// messaging.
|
||||
#[napi]
|
||||
pub fn apply_edits(&self, params: EditParams) -> Result<EditResult> {
|
||||
crate::chunk::edit::apply_edits(self, ¶ms).map_err(Error::from_reason)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
struct ParsedChunkReadPath {
|
||||
selector: Option<String>,
|
||||
crc: Option<String>,
|
||||
region: Option<ChunkRegion>,
|
||||
}
|
||||
|
||||
fn normalize_language(language: &str) -> String {
|
||||
language.trim().to_ascii_lowercase()
|
||||
}
|
||||
|
||||
fn chunk_info(chunk: &ChunkNode) -> ChunkInfo {
|
||||
ChunkInfo {
|
||||
path: chunk.path.clone(),
|
||||
identifier: chunk.identifier.clone(),
|
||||
checksum: chunk.checksum.clone(),
|
||||
start_line: chunk.start_line,
|
||||
end_line: chunk.end_line,
|
||||
leaf: chunk.leaf,
|
||||
}
|
||||
}
|
||||
|
||||
fn chunk_read_path_separator_index(read_path: &str) -> Option<usize> {
|
||||
if read_path.len() >= 3 {
|
||||
let bytes = read_path.as_bytes();
|
||||
if bytes[0].is_ascii_alphabetic() && bytes[1] == b':' && matches!(bytes[2], b'/' | b'\\') {
|
||||
return read_path[2..].find(':').map(|index| index + 2);
|
||||
}
|
||||
}
|
||||
read_path.find(':')
|
||||
}
|
||||
|
||||
fn parse_chunk_read_path(read_path: &str) -> std::result::Result<ParsedChunkReadPath, String> {
|
||||
let raw_selector =
|
||||
chunk_read_path_separator_index(read_path).map(|index| &read_path[(index + 1)..]);
|
||||
let ParsedSelector { selector, crc, region, .. } =
|
||||
split_selector_crc_and_region(raw_selector, None, None)?;
|
||||
Ok(ParsedChunkReadPath { selector, crc, region })
|
||||
}
|
||||
|
||||
fn parse_visible_line_range(selector: &str) -> Option<VisibleLineRange> {
|
||||
let captures = LINE_RANGE_SELECTOR_REGEX.captures(selector)?;
|
||||
let start_line = captures.get(1)?.as_str().parse::<u32>().ok()?.max(1);
|
||||
let end_line = captures
|
||||
.get(2)
|
||||
.and_then(|m| m.as_str().parse::<u32>().ok())
|
||||
.unwrap_or(start_line)
|
||||
.max(start_line);
|
||||
Some(VisibleLineRange { start_line, end_line })
|
||||
}
|
||||
|
||||
pub fn mask_chunk_display_source(source: &str, language: &str) -> String {
|
||||
if language != "tlaplus" {
|
||||
return source.to_string();
|
||||
}
|
||||
let lines = source.split('\n').collect::<Vec<_>>();
|
||||
let mut masked = lines
|
||||
.iter()
|
||||
.map(|line| (*line).to_string())
|
||||
.collect::<Vec<_>>();
|
||||
let mut index = 0usize;
|
||||
while index < lines.len() {
|
||||
if !TLAPLUS_BEGIN_TRANSLATION_REGEX.is_match(lines[index]) {
|
||||
index += 1;
|
||||
continue;
|
||||
}
|
||||
let begin_index = index;
|
||||
let mut end_index = begin_index + 1;
|
||||
while end_index < lines.len() && !TLAPLUS_END_TRANSLATION_REGEX.is_match(lines[end_index]) {
|
||||
end_index += 1;
|
||||
}
|
||||
if begin_index + 1 < lines.len() {
|
||||
masked[begin_index + 1] = "\\* [translation hidden]".to_string();
|
||||
for line in masked
|
||||
.iter_mut()
|
||||
.take(end_index.min(lines.len()))
|
||||
.skip(begin_index + 2)
|
||||
{
|
||||
line.clear();
|
||||
}
|
||||
}
|
||||
index = end_index + 1;
|
||||
}
|
||||
masked.join("\n")
|
||||
}
|
||||
@@ -1,403 +0,0 @@
|
||||
//! Shared types for the chunk-tree system.
|
||||
|
||||
use napi_derive::napi;
|
||||
|
||||
use crate::chunk::{kind::ChunkKind, state::ChunkState};
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct ChunkNode {
|
||||
pub path: String,
|
||||
pub identifier: Option<String>,
|
||||
pub kind: ChunkKind,
|
||||
pub leaf: bool,
|
||||
/// For virtual chunks (for example `theirs` branches in conflicts), content
|
||||
/// that is rendered instead of slicing `source`.
|
||||
pub virtual_content: Option<String>,
|
||||
pub parent_path: Option<String>,
|
||||
pub children: Vec<String>,
|
||||
pub signature: Option<String>,
|
||||
pub start_line: u32,
|
||||
pub end_line: u32,
|
||||
pub line_count: u32,
|
||||
pub start_byte: u32,
|
||||
pub end_byte: u32,
|
||||
/// Start byte of the semantic declaration used for checksums. This can be
|
||||
/// later than `start_byte` when the chunk absorbs attached leading trivia
|
||||
/// such as doc comments or attributes.
|
||||
pub checksum_start_byte: u32,
|
||||
|
||||
/// End byte of the prologue region. `None` means the chunk does not expose
|
||||
/// regions.
|
||||
pub prologue_end_byte: Option<u32>,
|
||||
/// Start byte of the epilogue region. `None` means the chunk does not expose
|
||||
/// regions.
|
||||
pub epilogue_start_byte: Option<u32>,
|
||||
|
||||
pub checksum: String,
|
||||
pub error: bool,
|
||||
pub indent: u32,
|
||||
pub indent_char: String,
|
||||
/// True for group-candidate chunks (e.g. `stmts`, `imports`, `decls`) that
|
||||
/// represent an ordered list of similar items. Append/prepend is valid on
|
||||
/// these even when they are leaf nodes.
|
||||
pub group: bool,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct ChunkTree {
|
||||
pub language: String,
|
||||
pub checksum: String,
|
||||
pub line_count: u32,
|
||||
pub parse_errors: u32,
|
||||
/// 1-indexed line numbers of tree-sitter ERROR / MISSING nodes surfaced
|
||||
/// during parsing. Used to focus error messages around failure locations
|
||||
/// when an edit introduces a parse error far from the edit target.
|
||||
pub parse_error_lines: Vec<u32>,
|
||||
pub fallback: bool,
|
||||
pub root_path: String,
|
||||
pub root_children: Vec<String>,
|
||||
pub chunks: Vec<ChunkNode>,
|
||||
}
|
||||
|
||||
/// Summary of a single chunk node for tool output and navigation.
|
||||
#[derive(Clone)]
|
||||
#[napi(object)]
|
||||
pub struct ChunkInfo {
|
||||
/// Chunk selector path within the tree.
|
||||
pub path: String,
|
||||
/// Bare chunk identifier (without kind prefix), if available.
|
||||
pub identifier: Option<String>,
|
||||
/// Stable checksum anchor for this chunk.
|
||||
pub checksum: String,
|
||||
/// 1-based start line in the source file (inclusive).
|
||||
pub start_line: u32,
|
||||
/// 1-based end line in the source file (inclusive).
|
||||
pub end_line: u32,
|
||||
/// Whether this node is a leaf (no child chunks).
|
||||
pub leaf: bool,
|
||||
}
|
||||
|
||||
/// Result of resolving a chunk read request against the tree.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
#[napi(string_enum)]
|
||||
pub enum ChunkReadStatus {
|
||||
/// Selector matched a chunk and content was produced.
|
||||
#[napi(value = "ok")]
|
||||
Ok,
|
||||
/// No chunk matched the requested selector.
|
||||
#[napi(value = "not_found")]
|
||||
NotFound,
|
||||
/// Chunk matched but does not support the requested region.
|
||||
#[napi(value = "unsupported_region")]
|
||||
UnsupportedRegion,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
#[napi(string_enum)]
|
||||
pub enum ChunkRegion {
|
||||
#[napi(value = "^")]
|
||||
Head,
|
||||
#[napi(value = "~")]
|
||||
Body,
|
||||
}
|
||||
|
||||
impl ChunkRegion {
|
||||
pub const fn as_str(self) -> &'static str {
|
||||
match self {
|
||||
Self::Head => "^",
|
||||
Self::Body => "~",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Structural edit to apply relative to a chunk anchor.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
#[napi(string_enum)]
|
||||
pub enum ChunkEditOp {
|
||||
/// Put new content into the targeted region.
|
||||
#[napi(value = "put")]
|
||||
Put,
|
||||
/// Find and replace a literal substring within the targeted region.
|
||||
#[napi(value = "replace")]
|
||||
Replace,
|
||||
/// Remove the targeted region.
|
||||
#[napi(value = "delete")]
|
||||
Delete,
|
||||
/// Insert `content` before the targeted region span.
|
||||
#[napi(value = "before")]
|
||||
Before,
|
||||
/// Insert `content` after the targeted region span.
|
||||
#[napi(value = "after")]
|
||||
After,
|
||||
/// Insert `content` at the start inside the targeted region.
|
||||
#[napi(value = "prepend")]
|
||||
Prepend,
|
||||
/// Insert `content` at the end inside the targeted region.
|
||||
#[napi(value = "append")]
|
||||
Append,
|
||||
}
|
||||
|
||||
impl ChunkEditOp {
|
||||
pub const fn as_str(self) -> &'static str {
|
||||
match self {
|
||||
Self::Put => "put",
|
||||
Self::Replace => "replace",
|
||||
Self::Delete => "delete",
|
||||
Self::Before => "before",
|
||||
Self::After => "after",
|
||||
Self::Prepend => "prepend",
|
||||
Self::Append => "append",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Outcome of resolving which chunk was read for a `renderRead`-style request.
|
||||
#[derive(Clone)]
|
||||
#[napi(object)]
|
||||
pub struct ChunkReadTarget {
|
||||
/// Whether the selector matched.
|
||||
pub status: ChunkReadStatus,
|
||||
/// Sanitized selector string that was applied.
|
||||
pub selector: String,
|
||||
}
|
||||
|
||||
/// Inclusive 1-based line range within a source file (used for scoped chunk
|
||||
/// rendering).
|
||||
#[derive(Clone)]
|
||||
#[napi(object)]
|
||||
pub struct VisibleLineRange {
|
||||
/// First line to include.
|
||||
pub start_line: u32,
|
||||
/// Last line to include.
|
||||
pub end_line: u32,
|
||||
}
|
||||
|
||||
/// How chunk anchors are formatted in rendered output (name and checksum
|
||||
/// visibility).
|
||||
#[derive(Clone, Copy, Default)]
|
||||
#[napi(string_enum)]
|
||||
pub enum ChunkAnchorStyle {
|
||||
/// `[.name#crc]` style anchor.
|
||||
#[default]
|
||||
#[napi(value = "full")]
|
||||
Full,
|
||||
/// `[.kind#crc]` style anchor (kind is the name prefix before `_`).
|
||||
#[napi(value = "kind")]
|
||||
Kind,
|
||||
/// `[#crc]` style anchor.
|
||||
#[napi(value = "bare")]
|
||||
Bare,
|
||||
/// `[.name]` without checksum.
|
||||
#[napi(value = "full-omit")]
|
||||
FullOmit,
|
||||
/// `[.kind]` without checksum.
|
||||
#[napi(value = "kind-omit")]
|
||||
KindOmit,
|
||||
/// Minimal anchor without name or checksum.
|
||||
#[napi(value = "none")]
|
||||
None,
|
||||
}
|
||||
|
||||
impl ChunkAnchorStyle {
|
||||
pub const fn with_omit_checksum(self, omit: bool) -> Self {
|
||||
if !omit {
|
||||
return self;
|
||||
}
|
||||
match self {
|
||||
Self::Full => Self::FullOmit,
|
||||
Self::Kind => Self::KindOmit,
|
||||
Self::Bare => Self::None,
|
||||
Self::FullOmit => Self::FullOmit,
|
||||
Self::KindOmit => Self::KindOmit,
|
||||
Self::None => Self::None,
|
||||
}
|
||||
}
|
||||
|
||||
fn render_i(
|
||||
&self,
|
||||
marker: (&str, &str),
|
||||
indent: &str,
|
||||
name: &str,
|
||||
crc: &str,
|
||||
line_count_suffix: &str,
|
||||
) -> String {
|
||||
fn extract_kind(name: &str) -> &str {
|
||||
name.find('_').map_or_else(|| name, |index| &name[..index])
|
||||
}
|
||||
let (open, close) = marker;
|
||||
match self {
|
||||
Self::Full => format!("{indent}{open}{name}#{crc}{close}{line_count_suffix}"),
|
||||
Self::Kind => {
|
||||
format!(
|
||||
"{indent}{open}{kind}#{crc}{close}{line_count_suffix}",
|
||||
kind = extract_kind(name)
|
||||
)
|
||||
},
|
||||
Self::Bare => format!("{indent}{open}#{crc}{close}{line_count_suffix}"),
|
||||
Self::FullOmit => format!("{indent}{open}{name}{close}{line_count_suffix}"),
|
||||
Self::KindOmit => {
|
||||
format!("{indent}{open}{kind}{close}{line_count_suffix}", kind = extract_kind(name))
|
||||
},
|
||||
Self::None => String::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Render an opening anchor: `{prefix}@{name}#{crc}`.
|
||||
pub fn render(&self, prefix: &str, name: &str, crc: &str) -> String {
|
||||
self.render_i(("@", ""), prefix, name, crc, "")
|
||||
}
|
||||
}
|
||||
|
||||
/// How a chunk participates in a focus-scoped render pass.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
#[napi(string_enum)]
|
||||
pub enum ChunkFocusMode {
|
||||
/// Emit full content and recurse normally.
|
||||
#[napi(value = "expanded")]
|
||||
Expanded,
|
||||
/// Emit just the opening anchor; do not recurse or emit body.
|
||||
#[napi(value = "collapsed")]
|
||||
Collapsed,
|
||||
/// Emit opening + closing anchors; recurse into focused children only.
|
||||
/// Interior gap lines between children are suppressed.
|
||||
#[napi(value = "container")]
|
||||
Container,
|
||||
}
|
||||
|
||||
/// Path + focus mode pair for the N-API boundary (`HashMap` doesn't cross FFI).
|
||||
#[derive(Clone, Debug)]
|
||||
#[napi(object)]
|
||||
pub struct FocusedPath {
|
||||
pub path: String,
|
||||
pub mode: ChunkFocusMode,
|
||||
}
|
||||
|
||||
/// Options for `ChunkState.render`: which subtree to show and how anchors
|
||||
/// appear.
|
||||
#[derive(Clone)]
|
||||
#[napi(object)]
|
||||
pub struct RenderParams {
|
||||
/// Path of the chunk to render; `None` uses the tree root.
|
||||
pub chunk_path: Option<String>,
|
||||
/// Title line shown above the tree (often the file path).
|
||||
pub title: String,
|
||||
/// Optional language label for the header block.
|
||||
pub language_tag: Option<String>,
|
||||
/// Restrict output to an inclusive line range of the file.
|
||||
pub visible_range: Option<VisibleLineRange>,
|
||||
/// When true, list only direct children instead of a full subtree.
|
||||
pub render_children_only: bool,
|
||||
/// Hide checksums in anchors when true.
|
||||
pub omit_checksum: bool,
|
||||
/// Anchor formatting style for chunk headers.
|
||||
pub anchor_style: Option<ChunkAnchorStyle>,
|
||||
/// Include a one-line preview for leaf chunks.
|
||||
pub show_leaf_preview: bool,
|
||||
/// Replace tab characters in displayed previews (e.g. two spaces).
|
||||
pub tab_replacement: Option<String>,
|
||||
/// When true, normalize displayed indentation to canonical tabs.
|
||||
pub normalize_indent: Option<bool>,
|
||||
|
||||
/// When set, restrict rendering to these chunks with their specified focus
|
||||
/// modes. Everything not in this list is skipped.
|
||||
pub focused_paths: Option<Vec<FocusedPath>>,
|
||||
}
|
||||
|
||||
/// Options for `ChunkState.renderRead`: selector path, display path, and
|
||||
/// optional line scoping.
|
||||
#[derive(Clone)]
|
||||
#[napi(object)]
|
||||
pub struct ReadRenderParams {
|
||||
/// Read selector (`sel=...` path, line range, or empty for whole tree).
|
||||
pub read_path: String,
|
||||
/// Path shown in titles and error messages (often the file path).
|
||||
pub display_path: String,
|
||||
/// Optional language label for the rendered block.
|
||||
pub language_tag: Option<String>,
|
||||
/// Hide checksums in rendered anchors.
|
||||
pub omit_checksum: bool,
|
||||
/// Anchor formatting style.
|
||||
pub anchor_style: Option<ChunkAnchorStyle>,
|
||||
/// Optional absolute file line range to intersect with the resolved chunk.
|
||||
pub absolute_line_range: Option<VisibleLineRange>,
|
||||
/// Replace tabs in embedded previews.
|
||||
pub tab_replacement: Option<String>,
|
||||
/// When true, normalize displayed indentation to canonical tabs.
|
||||
pub normalize_indent: Option<bool>,
|
||||
}
|
||||
|
||||
/// Rendered chunk text plus optional resolution metadata for the read request.
|
||||
#[derive(Clone)]
|
||||
#[napi(object)]
|
||||
pub struct ReadResult {
|
||||
/// Rendered UTF-8 text (chunk tree, notice, or error message).
|
||||
pub text: String,
|
||||
/// When a selector was used, whether it matched and which selector applied.
|
||||
pub chunk: Option<ChunkReadTarget>,
|
||||
}
|
||||
|
||||
/// One edit in a batch; targets a chunk via `sel`/`crc` (with params-level
|
||||
/// defaults).
|
||||
#[derive(Clone)]
|
||||
#[napi(object)]
|
||||
pub struct EditOperation {
|
||||
/// Edit kind (replace, delete, insert relative to anchor).
|
||||
pub op: ChunkEditOp,
|
||||
/// Chunk selector path; falls back to `EditParams.defaultSelector` when
|
||||
/// omitted.
|
||||
pub sel: Option<String>,
|
||||
/// Optional checksum anchor; falls back to `EditParams.defaultCrc` when
|
||||
/// omitted.
|
||||
pub crc: Option<String>,
|
||||
/// Region to target. When omitted, targets the full chunk.
|
||||
pub region: Option<ChunkRegion>,
|
||||
/// Replacement or inserted text (meaning depends on `op`).
|
||||
pub content: Option<String>,
|
||||
/// For `replace` op: literal substring to find inside the target chunk.
|
||||
/// Must match exactly once.
|
||||
pub find: Option<String>,
|
||||
}
|
||||
|
||||
/// Arguments for applying a batch of chunk edits to a file.
|
||||
#[derive(Clone)]
|
||||
#[napi(object)]
|
||||
pub struct EditParams {
|
||||
/// Edits to apply in order.
|
||||
pub operations: Vec<EditOperation>,
|
||||
/// When true, normalize indentation for response rendering and inserted
|
||||
/// content. When false, preserve literal tabs/spaces.
|
||||
pub normalize_indent: Option<bool>,
|
||||
/// Default chunk selector when an `EditOperation` omits `sel`.
|
||||
pub default_selector: Option<String>,
|
||||
/// Default checksum when an `EditOperation` omits `crc`.
|
||||
pub default_crc: Option<String>,
|
||||
/// Anchor formatting for rendered response text.
|
||||
pub anchor_style: Option<ChunkAnchorStyle>,
|
||||
/// Working directory used to resolve `filePath` and display paths.
|
||||
pub cwd: String,
|
||||
/// Path to the source file to edit (often relative to `cwd`).
|
||||
pub file_path: String,
|
||||
}
|
||||
|
||||
/// Result of applying edits: new parse state plus before/after source and
|
||||
/// messaging.
|
||||
#[derive(Clone)]
|
||||
#[napi(object, object_from_js = false)]
|
||||
pub struct EditResult {
|
||||
/// Chunk tree state after applying edits and re-parsing.
|
||||
pub state: ChunkState,
|
||||
/// Full file text before edits.
|
||||
pub diff_before: String,
|
||||
/// Full file text after edits.
|
||||
pub diff_after: String,
|
||||
/// Rendered summary for tooling (hunks, anchors), driven by `anchorStyle`.
|
||||
pub response_text: String,
|
||||
/// Whether the on-disk source changed.
|
||||
pub changed: bool,
|
||||
/// Whether the updated source re-parsed without fatal issues.
|
||||
pub parse_valid: bool,
|
||||
/// Absolute or normalized paths that were written or touched.
|
||||
pub touched_paths: Vec<String>,
|
||||
/// Non-fatal issues (e.g. selector warnings) collected during apply.
|
||||
pub warnings: Vec<String>,
|
||||
}
|
||||
@@ -27,7 +27,6 @@ static GLOBAL: MiMalloc = MiMalloc;
|
||||
|
||||
pub mod appearance;
|
||||
pub mod ast;
|
||||
pub mod chunk;
|
||||
pub mod clipboard;
|
||||
pub mod fd;
|
||||
pub mod fs_cache;
|
||||
|
||||
@@ -1,6 +1,15 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Removed
|
||||
|
||||
- Removed the `chunk` edit mode, chunk-aware `read` selectors, chunk-aware `grep` rendering, and the `omp read` chunk CLI subcommand
|
||||
- Removed the `read.prosechunks`, `read.explorechunks`, and `read.anchorstyle` settings
|
||||
- Removed the underlying `chunk` native module and AST-based chunk schema generation from `pi-natives`
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed `poll` wait duration parsing to fall back to `30s` when the provided value is an empty string
|
||||
|
||||
## [14.4.1] - 2026-04-26
|
||||
### Breaking Changes
|
||||
|
||||
@@ -49,7 +49,6 @@ const commands: CommandEntry[] = [
|
||||
{ name: "config", load: () => import("./commands/config").then(m => m.default) },
|
||||
{ name: "grep", load: () => import("./commands/grep").then(m => m.default) },
|
||||
{ name: "grievances", load: () => import("./commands/grievances").then(m => m.default) },
|
||||
{ name: "read", load: () => import("./commands/read").then(m => m.default) },
|
||||
{ name: "jupyter", load: () => import("./commands/jupyter").then(m => m.default) },
|
||||
{ name: "plugin", load: () => import("./commands/plugin").then(m => m.default) },
|
||||
{ name: "setup", load: () => import("./commands/setup").then(m => m.default) },
|
||||
|
||||
@@ -1,67 +0,0 @@
|
||||
/**
|
||||
* Read CLI command handler.
|
||||
*
|
||||
* Handles `omp read` subcommand — emits chunk-mode read output for files,
|
||||
* and delegates URL reads through the read tool pipeline.
|
||||
*/
|
||||
import * as path from "node:path";
|
||||
import chalk from "chalk";
|
||||
import { Settings } from "../config/settings";
|
||||
import { formatChunkedRead, resolveAnchorStyle } from "../edit/modes/chunk";
|
||||
import { getLanguageFromPath } from "../modes/theme/theme";
|
||||
import type { ToolSession } from "../tools";
|
||||
import { parseReadUrlTarget } from "../tools/fetch";
|
||||
import { ReadTool } from "../tools/read";
|
||||
|
||||
export interface ReadCommandArgs {
|
||||
path: string;
|
||||
sel?: string;
|
||||
}
|
||||
|
||||
function createCliReadSession(cwd: string, settings: Settings): ToolSession {
|
||||
return {
|
||||
cwd,
|
||||
hasUI: false,
|
||||
hasEditTool: true,
|
||||
getSessionFile: () => null,
|
||||
getSessionSpawns: () => null,
|
||||
settings,
|
||||
};
|
||||
}
|
||||
|
||||
export async function runReadCommand(cmd: ReadCommandArgs): Promise<void> {
|
||||
const cwd = process.cwd();
|
||||
const parsedUrlTarget = parseReadUrlTarget(cmd.path, cmd.sel);
|
||||
if (parsedUrlTarget) {
|
||||
const settings = await Settings.init({ cwd });
|
||||
const tool = new ReadTool(createCliReadSession(cwd, settings));
|
||||
const result = await tool.execute("cli-read", { path: cmd.path, sel: cmd.sel });
|
||||
const text = result.content.find((content): content is { type: "text"; text: string } => content.type === "text");
|
||||
console.log(text?.text ?? "");
|
||||
return;
|
||||
}
|
||||
|
||||
const filePath = path.resolve(cmd.path);
|
||||
const file = Bun.file(filePath);
|
||||
if (!(await file.exists())) {
|
||||
console.error(chalk.red(`Error: File not found: ${cmd.path}`));
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const readPath = cmd.sel ? `${filePath}:${cmd.sel}` : filePath;
|
||||
const language = getLanguageFromPath(filePath);
|
||||
|
||||
try {
|
||||
const result = await formatChunkedRead({
|
||||
filePath,
|
||||
readPath,
|
||||
cwd,
|
||||
language,
|
||||
anchorStyle: resolveAnchorStyle(),
|
||||
});
|
||||
console.log(result.text);
|
||||
} catch (err) {
|
||||
console.error(chalk.red(`Error: ${err instanceof Error ? err.message : String(err)}`));
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
@@ -1,33 +0,0 @@
|
||||
/**
|
||||
* Chunk-mode read tool.
|
||||
*/
|
||||
import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli";
|
||||
import { type ReadCommandArgs, runReadCommand } from "../cli/read-cli";
|
||||
import { initTheme } from "../modes/theme/theme";
|
||||
|
||||
export default class Read extends Command {
|
||||
static description = "Read a file as a chunk tree";
|
||||
|
||||
static args = {
|
||||
path: Args.string({ description: "File path to read", required: true }),
|
||||
};
|
||||
|
||||
static flags = {
|
||||
sel: Flags.string({
|
||||
char: "s",
|
||||
description: "Chunk selector or line range (e.g. class_Foo.fn_bar, L10-L20)",
|
||||
}),
|
||||
};
|
||||
|
||||
async run(): Promise<void> {
|
||||
const { args, flags } = await this.parse(Read);
|
||||
|
||||
const cmd: ReadCommandArgs = {
|
||||
path: args.path ?? "",
|
||||
sel: flags.sel,
|
||||
};
|
||||
|
||||
await initTheme();
|
||||
await runReadCommand(cmd);
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,5 @@
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import { type ChunkAnchorStyle, formatAnchor } from "@oh-my-pi/pi-natives";
|
||||
import {
|
||||
getProjectDir,
|
||||
getProjectPromptsDir,
|
||||
@@ -62,35 +61,6 @@ prompt.registerHelper("hline", (lineNum: unknown, content: unknown): string => {
|
||||
return `${ref}${HASHLINE_CONTENT_SEPARATOR}${text}`;
|
||||
});
|
||||
|
||||
/**
|
||||
* {{anchor name checksum}} — render a branch anchor tag using the current anchor style.
|
||||
* Style is resolved from the template context (`anchorStyle`) or defaults to "full".
|
||||
*/
|
||||
prompt.registerHelper("anchor", function (this: prompt.TemplateContext, name: string, checksum: string): string {
|
||||
const style = (this.anchorStyle as ChunkAnchorStyle) ?? "full";
|
||||
return formatAnchor(name, checksum, style);
|
||||
});
|
||||
|
||||
/**
|
||||
* {{sel "parent_Name.child_Name"}} — render a chunk path for `sel` fields in examples.
|
||||
* In `full` style the path is returned as-is (`class_Server.fn_start`).
|
||||
* In `kind` style each segment is trimmed to its kind prefix (`class.fn`).
|
||||
* In `bare` style the path is omitted (the model uses only `crc` to identify chunks).
|
||||
*/
|
||||
prompt.registerHelper("sel", function (this: prompt.TemplateContext, chunkPath: string): string {
|
||||
const style = (this.anchorStyle as ChunkAnchorStyle) ?? "full";
|
||||
if (style === "full") return chunkPath;
|
||||
if (style === "bare") return "";
|
||||
// kind: trim each segment to its kind prefix (before the first `_`)
|
||||
return chunkPath
|
||||
.split(".")
|
||||
.map(seg => {
|
||||
const idx = seg.indexOf("_");
|
||||
return idx === -1 ? seg : seg.slice(0, idx);
|
||||
})
|
||||
.join(".");
|
||||
});
|
||||
|
||||
const INLINE_ARG_SHELL_PATTERN = /\$(?:ARGUMENTS|@(?:\[\d+(?::\d*)?\])?|\d+)/;
|
||||
const INLINE_ARG_TEMPLATE_PATTERN = /\{\{[\s\S]*?(?:\b(?:arguments|ARGUMENTS|args)\b|\barg\s+[^}]+)[\s\S]*?\}\}/;
|
||||
|
||||
|
||||
@@ -955,12 +955,12 @@ export const SETTINGS_SCHEMA = {
|
||||
// Edit tool
|
||||
"edit.mode": {
|
||||
type: "enum",
|
||||
values: ["replace", "patch", "hashline", "chunk", "vim", "apply_patch", "atom"] as const,
|
||||
values: ["replace", "patch", "hashline", "vim", "apply_patch", "atom"] as const,
|
||||
default: "hashline",
|
||||
ui: {
|
||||
tab: "editing",
|
||||
label: "Edit Mode",
|
||||
description: "Select the edit tool variant (replace, patch, hashline, chunk, vim, or apply_patch)",
|
||||
description: "Select the edit tool variant (replace, patch, hashline, vim, or apply_patch)",
|
||||
},
|
||||
},
|
||||
|
||||
@@ -1046,38 +1046,6 @@ export const SETTINGS_SCHEMA = {
|
||||
},
|
||||
},
|
||||
|
||||
"read.prosechunks": {
|
||||
type: "boolean",
|
||||
default: false,
|
||||
ui: {
|
||||
tab: "editing",
|
||||
label: "Prose Chunks",
|
||||
description: "Enable chunk rendering for prose files in chunk edit mode",
|
||||
},
|
||||
},
|
||||
|
||||
"read.explorechunks": {
|
||||
type: "boolean",
|
||||
default: false,
|
||||
ui: {
|
||||
tab: "editing",
|
||||
label: "Explore Chunks",
|
||||
description: "Show chunk tree without checksums for read-only agents like explore",
|
||||
},
|
||||
},
|
||||
|
||||
"read.anchorstyle": {
|
||||
type: "enum",
|
||||
values: ["full", "kind", "bare"],
|
||||
default: "full",
|
||||
ui: {
|
||||
tab: "editing",
|
||||
label: "Anchor Style",
|
||||
description: "Render chunk anchors with full names, kind prefixes, or checksum-only tags",
|
||||
submenu: true,
|
||||
},
|
||||
},
|
||||
|
||||
// LSP
|
||||
"lsp.enabled": {
|
||||
type: "boolean",
|
||||
|
||||
@@ -326,7 +326,7 @@ export class Settings {
|
||||
|
||||
/**
|
||||
* Get the edit variant for a specific model.
|
||||
* Returns "patch", "replace", "hashline", "chunk", "vim", "apply_patch", or null (use global default).
|
||||
* Returns "patch", "replace", "hashline", "vim", "apply_patch", or null (use global default).
|
||||
*/
|
||||
getEditVariantForModel(model: string | undefined): EditMode | null {
|
||||
if (!model) return null;
|
||||
|
||||
@@ -10,7 +10,6 @@ import {
|
||||
} from "../lsp";
|
||||
import applyPatchDescription from "../prompts/tools/apply-patch.md" with { type: "text" };
|
||||
import atomDescription from "../prompts/tools/atom.md" with { type: "text" };
|
||||
import chunkEditDescription from "../prompts/tools/chunk-edit.md" with { type: "text" };
|
||||
import hashlineDescription from "../prompts/tools/hashline.md" with { type: "text" };
|
||||
import patchDescription from "../prompts/tools/patch.md" with { type: "text" };
|
||||
import replaceDescription from "../prompts/tools/replace.md" with { type: "text" };
|
||||
@@ -27,15 +26,6 @@ import {
|
||||
executeAtomSingle,
|
||||
resolveAtomEntryPaths,
|
||||
} from "./modes/atom";
|
||||
import {
|
||||
type ChunkParams,
|
||||
type ChunkToolEdit,
|
||||
chunkEditParamsSchema,
|
||||
executeChunkSingle,
|
||||
parseChunkEditPath,
|
||||
resolveAnchorStyle,
|
||||
resolveChunkAutoIndent,
|
||||
} from "./modes/chunk";
|
||||
import {
|
||||
executeHashlineSingle,
|
||||
HashlineMismatchError,
|
||||
@@ -53,7 +43,6 @@ export * from "./diff";
|
||||
export * from "./line-hash";
|
||||
export * from "./modes/apply-patch";
|
||||
export * from "./modes/atom";
|
||||
export * from "./modes/chunk";
|
||||
export * from "./modes/hashline";
|
||||
export * from "./modes/patch";
|
||||
export * from "./modes/replace";
|
||||
@@ -66,7 +55,6 @@ type TInput =
|
||||
| typeof patchEditSchema
|
||||
| typeof hashlineEditParamsSchema
|
||||
| typeof atomEditParamsSchema
|
||||
| typeof chunkEditParamsSchema
|
||||
| typeof vimSchema
|
||||
| typeof applyPatchSchema;
|
||||
|
||||
@@ -76,7 +64,6 @@ type EditParams =
|
||||
| PatchParams
|
||||
| HashlineParams
|
||||
| AtomParams
|
||||
| ChunkParams
|
||||
| VimParams
|
||||
| ApplyPatchParams;
|
||||
type EditToolResultDetails = EditToolDetails | VimToolDetails;
|
||||
@@ -325,39 +312,6 @@ export class EditTool implements AgentTool<TInput> {
|
||||
|
||||
#getModeDefinition(): EditModeDefinition {
|
||||
return {
|
||||
chunk: {
|
||||
description: (session: ToolSession) =>
|
||||
prompt.render(chunkEditDescription, {
|
||||
anchorStyle: resolveAnchorStyle(session.settings),
|
||||
chunkAutoIndent: resolveChunkAutoIndent(),
|
||||
}),
|
||||
parameters: chunkEditParamsSchema,
|
||||
execute: (
|
||||
tool: EditTool,
|
||||
params: EditParams,
|
||||
signal: AbortSignal | undefined,
|
||||
batchRequest: LspBatchRequest | undefined,
|
||||
onUpdate?: (partialResult: AgentToolResult<EditToolDetails, TInput>) => void,
|
||||
) => {
|
||||
const { edits, path: topPath } = params as ChunkParams & { path?: string };
|
||||
const resolved = resolveEntryPaths(edits as ChunkToolEdit[], topPath);
|
||||
const byFile = groupBy(resolved, (e: ChunkToolEdit) => parseChunkEditPath(e.path).filePath);
|
||||
const entries = [...byFile.entries()].map(([filePath, fileEdits]) => ({
|
||||
path: filePath,
|
||||
run: (br: LspBatchRequest | undefined) =>
|
||||
executeChunkSingle({
|
||||
session: tool.session,
|
||||
path: filePath,
|
||||
edits: fileEdits,
|
||||
signal,
|
||||
batchRequest: br,
|
||||
writethrough: tool.#writethrough,
|
||||
beginDeferredDiagnosticsForPath: p => tool.#beginDeferredDiagnosticsForPath(p),
|
||||
}),
|
||||
}));
|
||||
return executePerFile(entries, batchRequest, onUpdate);
|
||||
},
|
||||
},
|
||||
patch: {
|
||||
description: () => prompt.render(patchDescription),
|
||||
parameters: patchEditSchema,
|
||||
|
||||
@@ -667,59 +667,6 @@ export const HASHLINE_BIGRAMS = [
|
||||
|
||||
export const HASHLINE_BIGRAMS_COUNT = HASHLINE_BIGRAMS.length;
|
||||
|
||||
/**
|
||||
* 40 common English BPE bigrams used by chunk checksums (`path#checksum`).
|
||||
* Kept separate from {@link HASHLINE_BIGRAMS} because the chunk checksum
|
||||
* format is `path#bigram1bigram2` (4 chars from a 1600-code namespace) and
|
||||
* is independent of the line-anchor format.
|
||||
*
|
||||
* Order is stable forever — changing it invalidates every saved chunk path.
|
||||
*/
|
||||
export const CHUNK_BIGRAMS = [
|
||||
"th",
|
||||
"he",
|
||||
"in",
|
||||
"er",
|
||||
"an",
|
||||
"re",
|
||||
"on",
|
||||
"at",
|
||||
"en",
|
||||
"nd",
|
||||
"ti",
|
||||
"es",
|
||||
"or",
|
||||
"te",
|
||||
"of",
|
||||
"ed",
|
||||
"is",
|
||||
"it",
|
||||
"al",
|
||||
"ar",
|
||||
"st",
|
||||
"to",
|
||||
"nt",
|
||||
"ng",
|
||||
"se",
|
||||
"ha",
|
||||
"as",
|
||||
"ou",
|
||||
"io",
|
||||
"le",
|
||||
"ve",
|
||||
"co",
|
||||
"me",
|
||||
"de",
|
||||
"hi",
|
||||
"ri",
|
||||
"ro",
|
||||
"ic",
|
||||
"ne",
|
||||
"ea",
|
||||
] as const;
|
||||
|
||||
export const CHUNK_BIGRAMS_COUNT = CHUNK_BIGRAMS.length;
|
||||
|
||||
/**
|
||||
* Regex source matching exactly one bigram from {@link HASHLINE_BIGRAMS}.
|
||||
* Used by hashline parsers — keep in sync with the alphabet array above.
|
||||
|
||||
@@ -1,832 +0,0 @@
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as nodePath from "node:path";
|
||||
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
|
||||
import { StringEnum } from "@oh-my-pi/pi-ai";
|
||||
import {
|
||||
ChunkAnchorStyle,
|
||||
ChunkEditOp,
|
||||
type ChunkInfo,
|
||||
ChunkReadStatus,
|
||||
type ChunkReadTarget,
|
||||
ChunkRegion,
|
||||
ChunkState,
|
||||
type EditOperation as NativeEditOperation,
|
||||
} from "@oh-my-pi/pi-natives";
|
||||
import { $envpos } from "@oh-my-pi/pi-utils";
|
||||
import { type Static, Type } from "@sinclair/typebox";
|
||||
import type { BunFile } from "bun";
|
||||
import { LRUCache } from "lru-cache";
|
||||
import type { Settings } from "../../config/settings";
|
||||
import type { WritethroughCallback, WritethroughDeferredHandle } from "../../lsp";
|
||||
import { getLanguageFromPath } from "../../modes/theme/theme";
|
||||
import type { ToolSession } from "../../tools";
|
||||
import { assertEditableFileContent } from "../../tools/auto-generated-guard";
|
||||
import { invalidateFsScanAfterWrite } from "../../tools/fs-cache-invalidation";
|
||||
import { outputMeta } from "../../tools/output-meta";
|
||||
import { enforcePlanModeWrite, resolvePlanPath } from "../../tools/plan-mode-guard";
|
||||
import { generateUnifiedDiffString } from "../diff";
|
||||
import { HASHLINE_BIGRAMS } from "../line-hash";
|
||||
import { detectLineEnding, normalizeToLF, restoreLineEndings, stripBom } from "../normalize";
|
||||
import type { EditToolDetails, LspBatchRequest } from "../renderer";
|
||||
|
||||
export type { ChunkReadTarget };
|
||||
|
||||
export type ChunkEditOperation =
|
||||
| { op: "put"; sel?: string; content: string }
|
||||
| { op: "delete"; sel?: string }
|
||||
| { op: "before"; sel?: string; content: string }
|
||||
| { op: "after"; sel?: string; content: string }
|
||||
| { op: "prepend"; sel?: string; content: string }
|
||||
| { op: "append"; sel?: string; content: string };
|
||||
|
||||
type ChunkEditResult = {
|
||||
diffSourceBefore: string;
|
||||
diffSourceAfter: string;
|
||||
responseText: string;
|
||||
changed: boolean;
|
||||
parseValid: boolean;
|
||||
touchedPaths: string[];
|
||||
warnings: string[];
|
||||
};
|
||||
|
||||
export type ParsedChunkReadPath = {
|
||||
filePath: string;
|
||||
selector?: string;
|
||||
};
|
||||
|
||||
type ChunkCacheEntry = {
|
||||
mtimeMs: number;
|
||||
size: number;
|
||||
source: string;
|
||||
state: ChunkState;
|
||||
};
|
||||
|
||||
const validAnchorStyles: Record<string, ChunkAnchorStyle> = {
|
||||
full: ChunkAnchorStyle.Full,
|
||||
kind: ChunkAnchorStyle.Kind,
|
||||
bare: ChunkAnchorStyle.Bare,
|
||||
};
|
||||
|
||||
export function resolveChunkAutoIndent(rawValue = Bun.env.PI_CHUNK_AUTOINDENT): boolean {
|
||||
if (!rawValue) return true;
|
||||
const normalized = rawValue.trim().toLowerCase();
|
||||
switch (normalized) {
|
||||
case "1":
|
||||
case "true":
|
||||
case "yes":
|
||||
case "on":
|
||||
return true;
|
||||
case "0":
|
||||
case "false":
|
||||
case "no":
|
||||
case "off":
|
||||
return false;
|
||||
default:
|
||||
throw new Error(`Invalid PI_CHUNK_AUTOINDENT: ${rawValue}`);
|
||||
}
|
||||
}
|
||||
|
||||
function getChunkRenderIndentOptions(): {
|
||||
normalizeIndent: boolean;
|
||||
tabReplacement: string;
|
||||
} {
|
||||
return resolveChunkAutoIndent()
|
||||
? { normalizeIndent: true, tabReplacement: " " }
|
||||
: { normalizeIndent: false, tabReplacement: "\t" };
|
||||
}
|
||||
|
||||
export function resolveAnchorStyle(settings?: Settings): ChunkAnchorStyle {
|
||||
const envStyle = Bun.env.PI_ANCHOR_STYLE;
|
||||
return (
|
||||
(envStyle && validAnchorStyles[envStyle]) ||
|
||||
(settings?.get("read.anchorstyle") as ChunkAnchorStyle | undefined) ||
|
||||
ChunkAnchorStyle.Full
|
||||
);
|
||||
}
|
||||
|
||||
const chunkStateCache = new LRUCache<string, ChunkCacheEntry>({
|
||||
max: $envpos("PI_CHUNK_CACHE_MAX_ENTRIES", 200),
|
||||
});
|
||||
|
||||
export function invalidateChunkCache(filePath: string): void {
|
||||
chunkStateCache.delete(filePath);
|
||||
}
|
||||
|
||||
type ChunkSourceContext = {
|
||||
resolvedPath: string;
|
||||
sourceFile: BunFile;
|
||||
sourceExists: boolean;
|
||||
rawContent: string;
|
||||
chunkLanguage: string | undefined;
|
||||
};
|
||||
|
||||
type ChunkSourceIntent = "read" | "write";
|
||||
|
||||
function normalizeLanguage(language: string | undefined): string {
|
||||
return language?.trim().toLowerCase() || "";
|
||||
}
|
||||
|
||||
function normalizeChunkSource(text: string): string {
|
||||
return normalizeToLF(stripBom(text).text);
|
||||
}
|
||||
|
||||
function displayPathForFile(filePath: string, cwd: string): string {
|
||||
const relative = nodePath.relative(cwd, filePath).replace(/\\/g, "/");
|
||||
return relative && !relative.startsWith("..") ? relative : filePath.replace(/\\/g, "/");
|
||||
}
|
||||
|
||||
function fileLanguageTag(filePath: string, language?: string): string | undefined {
|
||||
const normalizedLanguage = normalizeLanguage(language);
|
||||
if (normalizedLanguage.length > 0) return normalizedLanguage;
|
||||
const ext = nodePath.extname(filePath).replace(/^\./, "").toLowerCase();
|
||||
return ext.length > 0 ? ext : undefined;
|
||||
}
|
||||
|
||||
async function resolveChunkSourceContext(
|
||||
session: ToolSession,
|
||||
path: string,
|
||||
options?: { intent?: ChunkSourceIntent },
|
||||
): Promise<ChunkSourceContext> {
|
||||
const resolvedPath = resolvePlanPath(session, path);
|
||||
const sourceFile = Bun.file(resolvedPath);
|
||||
const sourceExists = await sourceFile.exists();
|
||||
if ((options?.intent ?? "write") === "write") {
|
||||
enforcePlanModeWrite(session, path, { op: sourceExists ? "update" : "create" });
|
||||
}
|
||||
|
||||
let rawContent = "";
|
||||
if (sourceExists) {
|
||||
rawContent = await sourceFile.text();
|
||||
assertEditableFileContent(rawContent, path);
|
||||
}
|
||||
|
||||
return {
|
||||
resolvedPath,
|
||||
sourceFile,
|
||||
sourceExists,
|
||||
rawContent,
|
||||
chunkLanguage: getLanguageFromPath(resolvedPath),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Preview-safe loader: read raw source without plan-mode enforcement or
|
||||
* editable-file guards. Used by streaming diff previews that must not throw
|
||||
* side-effecting errors while args are still being streamed.
|
||||
*/
|
||||
export async function loadChunkSource(params: {
|
||||
cwd: string;
|
||||
path: string;
|
||||
}): Promise<{ resolvedPath: string; rawContent: string; language: string | undefined; exists: boolean }> {
|
||||
const resolvedPath = nodePath.isAbsolute(params.path) ? params.path : nodePath.resolve(params.cwd, params.path);
|
||||
const sourceFile = Bun.file(resolvedPath);
|
||||
const exists = await sourceFile.exists();
|
||||
const rawContent = exists ? await sourceFile.text() : "";
|
||||
return { resolvedPath, rawContent, language: getLanguageFromPath(resolvedPath), exists };
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute a unified diff preview for a chunk edit without applying it.
|
||||
* Used for streaming previews while args are still arriving. Returns
|
||||
* `{ error }` on any failure so callers can decide whether to surface it.
|
||||
*/
|
||||
export async function computeChunkDiff(
|
||||
input: { path: string; edits: ChunkToolEdit[] },
|
||||
cwd: string,
|
||||
options?: { anchorStyle?: ChunkAnchorStyle; signal?: AbortSignal },
|
||||
): Promise<{ diff: string; firstChangedLine: number | undefined } | { error: string }> {
|
||||
try {
|
||||
options?.signal?.throwIfAborted?.();
|
||||
const { filePath } = parseChunkEditPath(input.path);
|
||||
if (!filePath) return { error: "chunk edit path is empty" };
|
||||
const { resolvedPath, rawContent, language } = await loadChunkSource({ cwd, path: filePath });
|
||||
options?.signal?.throwIfAborted?.();
|
||||
const { operations } = normalizeChunkEditOperations(input.edits);
|
||||
const result = applyChunkEdits({
|
||||
source: rawContent,
|
||||
language,
|
||||
cwd,
|
||||
filePath: resolvedPath,
|
||||
operations,
|
||||
anchorStyle: options?.anchorStyle,
|
||||
});
|
||||
options?.signal?.throwIfAborted?.();
|
||||
if (!result.changed) {
|
||||
return { diff: "", firstChangedLine: undefined };
|
||||
}
|
||||
return generateUnifiedDiffString(result.diffSourceBefore, result.diffSourceAfter);
|
||||
} catch (err) {
|
||||
return { error: err instanceof Error ? err.message : String(err) };
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeChunkRegionSyntax(text: string): string {
|
||||
return text.replaceAll("@body", "~").replaceAll("@head", "^");
|
||||
}
|
||||
|
||||
function buildChunkEditResult(result: {
|
||||
diffBefore: string;
|
||||
diffAfter: string;
|
||||
responseText: string;
|
||||
changed: boolean;
|
||||
parseValid: boolean;
|
||||
touchedPaths: string[];
|
||||
warnings: string[];
|
||||
}): ChunkEditResult {
|
||||
return {
|
||||
diffSourceBefore: result.diffBefore,
|
||||
diffSourceAfter: result.diffAfter,
|
||||
responseText: result.responseText,
|
||||
changed: result.changed,
|
||||
parseValid: result.parseValid,
|
||||
touchedPaths: result.touchedPaths,
|
||||
warnings: result.warnings.map(normalizeChunkRegionSyntax),
|
||||
};
|
||||
}
|
||||
|
||||
function chunkReadPathSeparatorIndex(readPath: string): number {
|
||||
if (/^[a-zA-Z]:[/\\]/.test(readPath)) {
|
||||
return readPath.indexOf(":", 2);
|
||||
}
|
||||
const urlMatch = readPath.match(/^([a-z][a-z0-9+.-]*):\/\//i);
|
||||
if (urlMatch) {
|
||||
const scheme = urlMatch[1].toLowerCase();
|
||||
const urlPrefixEnd = urlMatch[0].length;
|
||||
if (scheme === "local") {
|
||||
const index = readPath.lastIndexOf(":");
|
||||
return index >= urlPrefixEnd ? index : -1;
|
||||
}
|
||||
|
||||
const pathStart = readPath.indexOf("/", urlPrefixEnd);
|
||||
if (pathStart === -1) return -1;
|
||||
const index = readPath.lastIndexOf(":");
|
||||
return index >= pathStart ? index : -1;
|
||||
}
|
||||
return readPath.indexOf(":");
|
||||
}
|
||||
|
||||
export function parseChunkSelector(selector: string | undefined): { selector?: string } {
|
||||
if (!selector || selector.length === 0) {
|
||||
return {};
|
||||
}
|
||||
return { selector };
|
||||
}
|
||||
|
||||
/** Split a combined `file:selector` path into file path and chunk selector. */
|
||||
export function parseChunkEditPath(editPath: string | undefined): { filePath: string; selector?: string } {
|
||||
if (!editPath) return { filePath: "" };
|
||||
const colonIndex = chunkReadPathSeparatorIndex(editPath);
|
||||
if (colonIndex === -1) {
|
||||
return { filePath: editPath };
|
||||
}
|
||||
const sel = editPath.slice(colonIndex + 1) || undefined;
|
||||
return { filePath: editPath.slice(0, colonIndex), selector: sel };
|
||||
}
|
||||
|
||||
export function parseChunkReadPath(readPath: string): ParsedChunkReadPath {
|
||||
const colonIndex = chunkReadPathSeparatorIndex(readPath);
|
||||
if (colonIndex === -1) {
|
||||
return { filePath: readPath };
|
||||
}
|
||||
const parsedSelector = parseChunkSelector(readPath.slice(colonIndex + 1) || undefined);
|
||||
return {
|
||||
filePath: readPath.slice(0, colonIndex),
|
||||
selector: parsedSelector.selector,
|
||||
};
|
||||
}
|
||||
|
||||
export function isChunkReadablePath(readPath: string): boolean {
|
||||
return parseChunkReadPath(readPath).selector !== undefined;
|
||||
}
|
||||
|
||||
export async function loadChunkStateForFile(filePath: string, language: string | undefined): Promise<ChunkCacheEntry> {
|
||||
const file = Bun.file(filePath);
|
||||
const stat = await file.stat();
|
||||
const cached = chunkStateCache.get(filePath);
|
||||
if (cached && cached.mtimeMs === stat.mtimeMs && cached.size === stat.size) {
|
||||
return cached;
|
||||
}
|
||||
|
||||
const source = normalizeChunkSource(await file.text());
|
||||
const state = ChunkState.parse(source, normalizeLanguage(language));
|
||||
const entry = { mtimeMs: stat.mtimeMs, size: stat.size, source, state };
|
||||
chunkStateCache.set(filePath, entry);
|
||||
return entry;
|
||||
}
|
||||
|
||||
export async function formatChunkedRead(params: {
|
||||
filePath: string;
|
||||
readPath: string;
|
||||
cwd: string;
|
||||
language?: string;
|
||||
omitChecksum?: boolean;
|
||||
anchorStyle?: ChunkAnchorStyle;
|
||||
absoluteLineRange?: { startLine: number; endLine?: number };
|
||||
}): Promise<{ text: string; resolvedPath?: string; chunk?: ChunkReadTarget }> {
|
||||
const { filePath, readPath, cwd, language, omitChecksum = false, anchorStyle, absoluteLineRange } = params;
|
||||
const normalizedLanguage = normalizeLanguage(language);
|
||||
const { state } = await loadChunkStateForFile(filePath, normalizedLanguage);
|
||||
const displayPath = displayPathForFile(filePath, cwd);
|
||||
const renderIndentOptions = getChunkRenderIndentOptions();
|
||||
const result = state.renderRead({
|
||||
readPath,
|
||||
displayPath,
|
||||
languageTag: fileLanguageTag(filePath, normalizedLanguage),
|
||||
omitChecksum,
|
||||
anchorStyle,
|
||||
absoluteLineRange: absoluteLineRange
|
||||
? { startLine: absoluteLineRange.startLine, endLine: absoluteLineRange.endLine ?? absoluteLineRange.startLine }
|
||||
: undefined,
|
||||
tabReplacement: renderIndentOptions.tabReplacement,
|
||||
normalizeIndent: renderIndentOptions.normalizeIndent,
|
||||
});
|
||||
return { text: result.text, resolvedPath: filePath, chunk: result.chunk };
|
||||
}
|
||||
|
||||
export type ChunkedGrepMatch = {
|
||||
displayPath: string;
|
||||
fileLineCount: number;
|
||||
chunkPath?: string;
|
||||
chunkChecksum?: string;
|
||||
lineNumber: number;
|
||||
line: string;
|
||||
};
|
||||
|
||||
export async function describeChunkedGrepMatch(params: {
|
||||
filePath: string;
|
||||
lineNumber: number;
|
||||
line: string;
|
||||
cwd: string;
|
||||
language?: string;
|
||||
}): Promise<ChunkedGrepMatch> {
|
||||
const { filePath, lineNumber, line, cwd, language } = params;
|
||||
const { state } = await loadChunkStateForFile(filePath, language);
|
||||
const chunkPath = state.lineToContainingChunkPath(lineNumber) || undefined;
|
||||
const chunkInfo = chunkPath ? state.chunk(chunkPath) : null;
|
||||
return {
|
||||
displayPath: displayPathForFile(filePath, cwd),
|
||||
fileLineCount: state.lineCount,
|
||||
chunkPath,
|
||||
chunkChecksum: chunkInfo?.checksum,
|
||||
lineNumber,
|
||||
line,
|
||||
};
|
||||
}
|
||||
|
||||
const CHUNK_CHECKSUM_BIGRAMS = new Set<string>(HASHLINE_BIGRAMS);
|
||||
type NativeChunkRegion = "head" | "body";
|
||||
|
||||
function isChunkChecksumToken(value: string): boolean {
|
||||
if (value.length !== 4) return false;
|
||||
const lower = value.toLowerCase();
|
||||
return CHUNK_CHECKSUM_BIGRAMS.has(lower.slice(0, 2)) && CHUNK_CHECKSUM_BIGRAMS.has(lower.slice(2, 4));
|
||||
}
|
||||
|
||||
function parseChunkEditSelector(selector: string | undefined): {
|
||||
selector?: string;
|
||||
crc?: string;
|
||||
region?: NativeChunkRegion;
|
||||
} {
|
||||
if (!selector) {
|
||||
return {};
|
||||
}
|
||||
|
||||
let trimmed = selector.trim();
|
||||
if (trimmed.length === 0) {
|
||||
return {};
|
||||
}
|
||||
|
||||
let region: NativeChunkRegion | undefined;
|
||||
const suffix = trimmed.at(-1);
|
||||
if (suffix === "~" || suffix === "^") {
|
||||
region = suffix === "~" ? "body" : "head";
|
||||
trimmed = trimmed.slice(0, -1).trimEnd();
|
||||
}
|
||||
|
||||
let selectorPart = trimmed;
|
||||
let crc: string | undefined;
|
||||
const hashIndex = selectorPart.lastIndexOf("#");
|
||||
if (hashIndex >= 0) {
|
||||
const suffix = selectorPart.slice(hashIndex + 1).trim();
|
||||
if (isChunkChecksumToken(suffix)) {
|
||||
crc = suffix.toLowerCase();
|
||||
selectorPart = selectorPart.slice(0, hashIndex).trimEnd();
|
||||
}
|
||||
} else if (isChunkChecksumToken(selectorPart)) {
|
||||
crc = selectorPart.toLowerCase();
|
||||
selectorPart = "";
|
||||
}
|
||||
|
||||
return { selector: selectorPart || undefined, crc, region };
|
||||
}
|
||||
|
||||
type NativeChunkRegionEncoding = "named" | "symbolic";
|
||||
|
||||
function toNativeEditRegion(
|
||||
region: NativeChunkRegion | undefined,
|
||||
encoding: NativeChunkRegionEncoding,
|
||||
): NativeEditOperation["region"] | undefined {
|
||||
if (!region) {
|
||||
return undefined;
|
||||
}
|
||||
if (encoding === "symbolic") {
|
||||
return region === "body" ? ChunkRegion.Body : ChunkRegion.Head;
|
||||
}
|
||||
return region as unknown as NativeEditOperation["region"] | undefined;
|
||||
}
|
||||
|
||||
function toNativeEditOperation(
|
||||
operation: ChunkEditOperation,
|
||||
defaultRegion: NativeChunkRegion | undefined,
|
||||
encoding: NativeChunkRegionEncoding,
|
||||
): NativeEditOperation {
|
||||
const { selector, crc, region } = parseChunkEditSelector(operation.sel);
|
||||
const nativeRegion = toNativeEditRegion(operation.sel === undefined ? (region ?? defaultRegion) : region, encoding);
|
||||
switch (operation.op) {
|
||||
case "put":
|
||||
return {
|
||||
op: ChunkEditOp.Put,
|
||||
sel: selector,
|
||||
crc,
|
||||
region: nativeRegion,
|
||||
content: operation.content,
|
||||
};
|
||||
case "before":
|
||||
return { op: ChunkEditOp.Before, sel: selector, crc, region: nativeRegion, content: operation.content };
|
||||
case "after":
|
||||
return { op: ChunkEditOp.After, sel: selector, crc, region: nativeRegion, content: operation.content };
|
||||
case "prepend":
|
||||
return { op: ChunkEditOp.Prepend, sel: selector, crc, region: nativeRegion, content: operation.content };
|
||||
case "append":
|
||||
return { op: ChunkEditOp.Append, sel: selector, crc, region: nativeRegion, content: operation.content };
|
||||
case "delete":
|
||||
return { op: ChunkEditOp.Delete, sel: selector, crc, region: nativeRegion };
|
||||
default: {
|
||||
const exhaustive: never = operation;
|
||||
return exhaustive;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function buildNativeChunkEditRequest(
|
||||
params: { defaultSelector?: string; defaultCrc?: string; operations: ChunkEditOperation[] },
|
||||
encoding: NativeChunkRegionEncoding,
|
||||
): Pick<Parameters<ChunkState["applyEdits"]>[0], "operations" | "defaultSelector" | "defaultCrc"> {
|
||||
const parsedDefaultSelector = parseChunkEditSelector(params.defaultSelector);
|
||||
const operations = params.operations.map(operation =>
|
||||
toNativeEditOperation(operation, parsedDefaultSelector.region, encoding),
|
||||
);
|
||||
return {
|
||||
operations,
|
||||
defaultSelector: parsedDefaultSelector.selector,
|
||||
defaultCrc: params.defaultCrc ?? parsedDefaultSelector.crc,
|
||||
};
|
||||
}
|
||||
|
||||
function isChunkRegionEncodingError(error: unknown): error is Error {
|
||||
return (
|
||||
error instanceof Error &&
|
||||
/value `"(body|head|~|\^)"` does not match any variant of enum `ChunkRegion`/.test(error.message)
|
||||
);
|
||||
}
|
||||
|
||||
export function applyChunkEdits(params: {
|
||||
source: string;
|
||||
language?: string;
|
||||
cwd: string;
|
||||
filePath: string;
|
||||
operations: ChunkEditOperation[];
|
||||
defaultSelector?: string;
|
||||
defaultCrc?: string;
|
||||
anchorStyle?: ChunkAnchorStyle;
|
||||
}): ChunkEditResult {
|
||||
const normalizedSource = normalizeChunkSource(params.source);
|
||||
const applyNativeEdits = (encoding: NativeChunkRegionEncoding): ChunkEditResult => {
|
||||
const request = buildNativeChunkEditRequest(params, encoding);
|
||||
const state = ChunkState.parse(normalizedSource, normalizeLanguage(params.language));
|
||||
return buildChunkEditResult(
|
||||
state.applyEdits({
|
||||
operations: request.operations,
|
||||
normalizeIndent: resolveChunkAutoIndent(),
|
||||
defaultSelector: request.defaultSelector,
|
||||
defaultCrc: request.defaultCrc,
|
||||
anchorStyle: params.anchorStyle,
|
||||
cwd: params.cwd,
|
||||
filePath: params.filePath,
|
||||
}),
|
||||
);
|
||||
};
|
||||
|
||||
try {
|
||||
return applyNativeEdits("named");
|
||||
} catch (error) {
|
||||
if (isChunkRegionEncodingError(error)) {
|
||||
try {
|
||||
return applyNativeEdits("symbolic");
|
||||
} catch (fallbackError) {
|
||||
if (fallbackError instanceof Error) {
|
||||
throw new Error(normalizeChunkRegionSyntax(fallbackError.message));
|
||||
}
|
||||
throw fallbackError;
|
||||
}
|
||||
}
|
||||
if (error instanceof Error) {
|
||||
throw new Error(normalizeChunkRegionSyntax(error.message));
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
export async function getChunkInfoForFile(
|
||||
filePath: string,
|
||||
language: string | undefined,
|
||||
chunkPath: string,
|
||||
): Promise<ChunkInfo | undefined> {
|
||||
const { state } = await loadChunkStateForFile(filePath, language);
|
||||
return state.chunk(chunkPath) ?? undefined;
|
||||
}
|
||||
|
||||
export function missingChunkReadTarget(selector: string): ChunkReadTarget {
|
||||
return { status: ChunkReadStatus.NotFound, selector };
|
||||
}
|
||||
|
||||
export const chunkToolEditSchema = Type.Object(
|
||||
{
|
||||
path: Type.Optional(
|
||||
Type.String({
|
||||
description: "File path with chunk selector. Examples: 'src/app.ts:fn_foo#thth~', 'src/app.ts:class_Bar'.",
|
||||
}),
|
||||
),
|
||||
write: Type.Optional(
|
||||
Type.Union([Type.String(), Type.Null()], {
|
||||
description:
|
||||
"Write complete new content to the targeted region. Null is rejected; use delete: true for deletion.",
|
||||
}),
|
||||
),
|
||||
delete: Type.Optional(
|
||||
Type.Boolean({
|
||||
description: "Explicitly delete the targeted chunk. Must be true; include the current chunk ID.",
|
||||
}),
|
||||
),
|
||||
insert: Type.Optional(
|
||||
Type.Object(
|
||||
{
|
||||
loc: StringEnum(["append", "prepend"] as const),
|
||||
body: Type.String({ description: "Content to insert." }),
|
||||
},
|
||||
{ description: "Insert content relative to the chunk." },
|
||||
),
|
||||
),
|
||||
},
|
||||
{ additionalProperties: false },
|
||||
);
|
||||
export const chunkEditParamsSchema = Type.Object(
|
||||
{
|
||||
path: Type.Optional(Type.String({ description: "Default file path used when an edit omits its own `path`" })),
|
||||
edits: Type.Array(chunkToolEditSchema, {
|
||||
description: "Chunk edits",
|
||||
minItems: 1,
|
||||
}),
|
||||
},
|
||||
{ additionalProperties: false },
|
||||
);
|
||||
|
||||
export type ChunkToolEdit = Static<typeof chunkToolEditSchema>;
|
||||
export type ChunkParams = Static<typeof chunkEditParamsSchema>;
|
||||
|
||||
export interface ExecuteChunkSingleOptions {
|
||||
session: ToolSession;
|
||||
path: string;
|
||||
edits: ChunkToolEdit[];
|
||||
signal?: AbortSignal;
|
||||
batchRequest?: LspBatchRequest;
|
||||
writethrough: WritethroughCallback;
|
||||
beginDeferredDiagnosticsForPath: (path: string) => WritethroughDeferredHandle;
|
||||
}
|
||||
|
||||
/** Auto-correct indentation for content targeting a body region (`~`) when autoIndent is on.
|
||||
* Handles two patterns:
|
||||
* 1. Tab-based over-indentation: models include the function's base \t indent.
|
||||
* 2. Space-based indentation: models use literal spaces instead of \t.
|
||||
* Returns the corrected content and any warnings. */
|
||||
function autoCorrectBodyIndent(content: string, index: number): { content: string; warnings: string[] } {
|
||||
const warnings: string[] = [];
|
||||
if (!content || !resolveChunkAutoIndent()) return { content, warnings };
|
||||
const lines = content.split("\n");
|
||||
const nonEmpty = lines.filter(l => l.length > 0);
|
||||
if (nonEmpty.length <= 1) return { content, warnings };
|
||||
|
||||
// 1. Tab-based over-indentation: strip common leading tabs.
|
||||
const minTabs = Math.min(...nonEmpty.map(l => l.match(/^\t*/)?.[0].length ?? 0));
|
||||
if (minTabs >= 1) {
|
||||
const fixed = lines.map(l => (l.length === 0 ? l : l.slice(minTabs))).join("\n");
|
||||
warnings.push(
|
||||
`Edit ${index + 1}: auto-corrected body indentation \u2014 stripped ${minTabs} leading tab(s). When writing to \`~\`, write at column 0; the tool adds the function's base indent.`,
|
||||
);
|
||||
return { content: fixed, warnings };
|
||||
}
|
||||
|
||||
// 2. Space-based indentation: strip common leading spaces and convert to tabs.
|
||||
const spaceIndents = nonEmpty.map(l => l.match(/^ */)?.[0].length ?? 0);
|
||||
const minSpaces = Math.min(...spaceIndents);
|
||||
if (minSpaces >= 2) {
|
||||
const indentDiffs = spaceIndents.map(s => s - minSpaces).filter(d => d > 0);
|
||||
const indentUnit = indentDiffs.length > 0 ? Math.min(...indentDiffs) : 4;
|
||||
const unit = indentUnit >= 2 && indentUnit <= 8 ? indentUnit : 4;
|
||||
const fixed = lines
|
||||
.map(line => {
|
||||
if (line.length === 0) return line;
|
||||
const stripped = line.slice(minSpaces);
|
||||
const leadingSpaces = stripped.match(/^ */)?.[0].length ?? 0;
|
||||
const tabs = Math.floor(leadingSpaces / unit);
|
||||
const rem = leadingSpaces % unit;
|
||||
return "\t".repeat(tabs) + " ".repeat(rem) + stripped.slice(leadingSpaces);
|
||||
})
|
||||
.join("\n");
|
||||
warnings.push(
|
||||
`Edit ${index + 1}: auto-converted space indentation to tabs \u2014 stripped ${minSpaces} common leading spaces and converted ${unit}-space indent to tabs. When auto-indent is on, use \\t for indentation.`,
|
||||
);
|
||||
return { content: fixed, warnings };
|
||||
}
|
||||
|
||||
return { content, warnings };
|
||||
}
|
||||
|
||||
function chunkEditOperationFields(edit: ChunkToolEdit): string[] {
|
||||
const fields: string[] = [];
|
||||
if (edit.write !== undefined) fields.push("write");
|
||||
if (edit.insert != null) fields.push("insert");
|
||||
if (edit.delete === true) fields.push("delete");
|
||||
return fields;
|
||||
}
|
||||
|
||||
function assertSingleChunkOperation(edit: ChunkToolEdit, index: number): string {
|
||||
const fields = chunkEditOperationFields(edit);
|
||||
if (fields.length === 0) {
|
||||
throw new Error(
|
||||
`Edit ${index + 1}: no operation specified. Use write:"..." to replace, insert:{loc,body} to insert, or delete:true to delete. Use the open tool to inspect chunks.`,
|
||||
);
|
||||
}
|
||||
if (fields.length > 1) {
|
||||
throw new Error(
|
||||
`Edit ${index + 1}: multiple operation fields set (${fields.join(", ")}). Each chunk edit entry must have exactly one operation.`,
|
||||
);
|
||||
}
|
||||
return fields[0];
|
||||
}
|
||||
|
||||
function normalizeChunkEditOperations(edits: ChunkToolEdit[]): {
|
||||
operations: ChunkEditOperation[];
|
||||
warnings: string[];
|
||||
} {
|
||||
const warnings: string[] = [];
|
||||
const operations = edits.map((edit, index): ChunkEditOperation => {
|
||||
const { selector } = parseChunkEditPath(edit.path);
|
||||
const operation = assertSingleChunkOperation(edit, index);
|
||||
if (operation === "write") {
|
||||
if (edit.write === null) {
|
||||
throw new Error(
|
||||
`Edit ${index + 1}: write:null no longer deletes chunks. Use delete:true to delete, or open the chunk to inspect its content without modifying the file.`,
|
||||
);
|
||||
}
|
||||
if (typeof edit.write !== "string") {
|
||||
throw new Error(`Edit ${index + 1}: write must be a string.`);
|
||||
}
|
||||
if (edit.write.length === 0) {
|
||||
throw new Error(
|
||||
`Edit ${index + 1}: write:"" is a destructive empty replacement. Use delete:true to delete the chunk, or open the chunk to inspect its content without modifying the file.`,
|
||||
);
|
||||
}
|
||||
let writeContent = edit.write;
|
||||
if (selector?.endsWith("~")) {
|
||||
const corrected = autoCorrectBodyIndent(writeContent, index);
|
||||
writeContent = corrected.content;
|
||||
warnings.push(...corrected.warnings);
|
||||
}
|
||||
return { op: "put", sel: selector, content: writeContent };
|
||||
}
|
||||
if (operation === "insert") {
|
||||
if (edit.insert == null || typeof edit.insert.body !== "string" || edit.insert.body.length === 0) {
|
||||
throw new Error(`Edit ${index + 1}: insert.body must be a non-empty string.`);
|
||||
}
|
||||
const op = edit.insert.loc === "prepend" ? "before" : "after";
|
||||
let insertContent = edit.insert.body;
|
||||
if (selector?.endsWith("~")) {
|
||||
const corrected = autoCorrectBodyIndent(insertContent, index);
|
||||
insertContent = corrected.content;
|
||||
warnings.push(...corrected.warnings);
|
||||
}
|
||||
return { op, sel: selector, content: insertContent };
|
||||
}
|
||||
if (operation !== "delete") {
|
||||
throw new Error(`Edit ${index + 1}: unsupported chunk edit operation "${operation}".`);
|
||||
}
|
||||
return { op: "delete", sel: selector };
|
||||
});
|
||||
return { operations, warnings };
|
||||
}
|
||||
|
||||
async function writeChunkResult(params: {
|
||||
result: ChunkEditResult;
|
||||
resolvedPath: string;
|
||||
sourceFile: BunFile;
|
||||
sourceText: string;
|
||||
sourceExists: boolean;
|
||||
signal?: AbortSignal;
|
||||
batchRequest?: LspBatchRequest;
|
||||
writethrough: WritethroughCallback;
|
||||
beginDeferredDiagnosticsForPath: (path: string) => WritethroughDeferredHandle;
|
||||
}): Promise<AgentToolResult<EditToolDetails, typeof chunkEditParamsSchema>> {
|
||||
const {
|
||||
result,
|
||||
resolvedPath,
|
||||
sourceFile,
|
||||
sourceText,
|
||||
sourceExists,
|
||||
signal,
|
||||
batchRequest,
|
||||
writethrough,
|
||||
beginDeferredDiagnosticsForPath,
|
||||
} = params;
|
||||
|
||||
const { bom, text } = stripBom(sourceText);
|
||||
const originalEnding = detectLineEnding(text);
|
||||
const finalContent = bom + restoreLineEndings(result.diffSourceAfter, originalEnding);
|
||||
const diagnostics = await writethrough(resolvedPath, finalContent, signal, sourceFile, batchRequest, dst =>
|
||||
dst === resolvedPath ? beginDeferredDiagnosticsForPath(resolvedPath) : undefined,
|
||||
);
|
||||
invalidateFsScanAfterWrite(resolvedPath);
|
||||
|
||||
const diffResult = generateUnifiedDiffString(result.diffSourceBefore, result.diffSourceAfter);
|
||||
const warningsBlock = result.warnings.length > 0 ? `\n\nWarnings:\n${result.warnings.join("\n")}` : "";
|
||||
const meta = outputMeta()
|
||||
.diagnostics(diagnostics?.summary ?? "", diagnostics?.messages ?? [])
|
||||
.get();
|
||||
|
||||
return {
|
||||
content: [{ type: "text", text: `${result.responseText}${warningsBlock}` }],
|
||||
details: {
|
||||
diff: diffResult.diff,
|
||||
firstChangedLine: diffResult.firstChangedLine,
|
||||
diagnostics,
|
||||
op: sourceExists ? "update" : "create",
|
||||
meta,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
export async function executeChunkSingle(
|
||||
options: ExecuteChunkSingleOptions,
|
||||
): Promise<AgentToolResult<EditToolDetails, typeof chunkEditParamsSchema>> {
|
||||
const { session, path, edits, signal, batchRequest, writethrough, beginDeferredDiagnosticsForPath } = options;
|
||||
const { resolvedPath, sourceFile, sourceExists, rawContent, chunkLanguage } = await resolveChunkSourceContext(
|
||||
session,
|
||||
path,
|
||||
{ intent: "write" },
|
||||
);
|
||||
const parentDir = nodePath.dirname(resolvedPath);
|
||||
if (parentDir && parentDir !== ".") {
|
||||
await fs.mkdir(parentDir, { recursive: true });
|
||||
}
|
||||
const { operations: normalizedOperations, warnings: normWarnings } = normalizeChunkEditOperations(edits);
|
||||
|
||||
if (!sourceExists && normalizedOperations.some(op => op.sel)) {
|
||||
throw new Error(
|
||||
`File does not exist: ${path}. Cannot resolve chunk selectors on a non-existent file. Use the write tool to create a new file, or check the path for typos.`,
|
||||
);
|
||||
}
|
||||
|
||||
const chunkResult = applyChunkEdits({
|
||||
source: rawContent,
|
||||
language: chunkLanguage,
|
||||
cwd: session.cwd,
|
||||
filePath: resolvedPath,
|
||||
operations: normalizedOperations,
|
||||
anchorStyle: resolveAnchorStyle(session.settings),
|
||||
});
|
||||
chunkResult.warnings.push(...normWarnings);
|
||||
|
||||
if (!chunkResult.changed) {
|
||||
const warningsBlock = chunkResult.warnings.length > 0 ? `\n\nWarnings:\n${chunkResult.warnings.join("\n")}` : "";
|
||||
return {
|
||||
content: [{ type: "text", text: `[No changes needed — content already matches.]${warningsBlock}` }],
|
||||
details: {
|
||||
diff: "",
|
||||
op: sourceExists ? "update" : "create",
|
||||
meta: outputMeta().get(),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
return writeChunkResult({
|
||||
result: chunkResult,
|
||||
resolvedPath,
|
||||
sourceFile,
|
||||
sourceText: rawContent,
|
||||
sourceExists,
|
||||
signal,
|
||||
batchRequest,
|
||||
writethrough,
|
||||
beginDeferredDiagnosticsForPath,
|
||||
});
|
||||
}
|
||||
@@ -29,7 +29,6 @@ import type { VimToolDetails } from "../vim/types";
|
||||
import type { DiffError, DiffResult } from "./diff";
|
||||
import { expandApplyPatchToEntries, expandApplyPatchToPreviewEntries } from "./modes/apply-patch";
|
||||
import type { Operation, PatchEditEntry } from "./modes/patch";
|
||||
import type { PerFileDiffPreview } from "./streaming";
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════════════
|
||||
// LSP Batching
|
||||
@@ -94,7 +93,7 @@ interface EditRenderArgs {
|
||||
*/
|
||||
previewDiff?: string;
|
||||
__partialJson?: string;
|
||||
// Hashline / chunk mode fields
|
||||
// Hashline mode fields
|
||||
edits?: EditRenderEntry[];
|
||||
}
|
||||
|
||||
@@ -141,8 +140,6 @@ export interface EditRenderContext {
|
||||
editMode?: EditMode;
|
||||
/** Pre-computed diff preview (computed before tool executes) */
|
||||
editDiffPreview?: DiffResult | DiffError;
|
||||
/** Multi-file streaming diff preview (chunk edits spanning several files) */
|
||||
perFileDiffPreview?: PerFileDiffPreview[];
|
||||
/** Function to render diff text with syntax highlighting */
|
||||
renderDiff?: (diffText: string, options?: { filePath?: string }) => string;
|
||||
}
|
||||
@@ -151,11 +148,9 @@ const EDIT_STREAMING_PREVIEW_LINES = 12;
|
||||
const CALL_TEXT_PREVIEW_LINES = 6;
|
||||
const CALL_TEXT_PREVIEW_WIDTH = 80;
|
||||
|
||||
/** Extract file path from an edit entry's path (handles chunk's file:selector format). */
|
||||
/** Extract file path from an edit entry. */
|
||||
function filePathFromEditEntry(p: string | undefined): string | undefined {
|
||||
if (!p) return undefined;
|
||||
const ci = /^[a-zA-Z]:[/\\]/.test(p) ? p.indexOf(":", 2) : p.indexOf(":");
|
||||
return ci === -1 ? p : p.slice(0, ci);
|
||||
return p ?? undefined;
|
||||
}
|
||||
|
||||
function decodePartialJsonStringFragment(fragment: string): string {
|
||||
@@ -280,32 +275,7 @@ function formatMetadataLine(lineCount: number | null, language: string | undefin
|
||||
return uiTheme.fg("dim", `${icon}`);
|
||||
}
|
||||
|
||||
function formatMultiFileStreamingDiff(previews: PerFileDiffPreview[], uiTheme: Theme): string {
|
||||
const parts: string[] = [];
|
||||
for (const preview of previews) {
|
||||
if (!preview.diff && !preview.error) continue;
|
||||
const header = uiTheme.fg("dim", `\n\n── ${shortenPath(preview.path)} ──`);
|
||||
if (preview.error) {
|
||||
parts.push(`${header}\n${uiTheme.fg("error", replaceTabs(preview.error))}`);
|
||||
continue;
|
||||
}
|
||||
if (preview.diff) {
|
||||
parts.push(`${header}${formatStreamingDiff(preview.diff, preview.path, uiTheme, "preview")}`);
|
||||
}
|
||||
}
|
||||
return parts.join("");
|
||||
}
|
||||
|
||||
function getCallPreview(
|
||||
args: EditRenderArgs,
|
||||
rawPath: string,
|
||||
uiTheme: Theme,
|
||||
renderContext: EditRenderContext | undefined,
|
||||
): string {
|
||||
const multi = renderContext?.perFileDiffPreview;
|
||||
if (multi && multi.length > 0 && multi.some(p => p.diff || p.error)) {
|
||||
return formatMultiFileStreamingDiff(multi, uiTheme);
|
||||
}
|
||||
function getCallPreview(args: EditRenderArgs, rawPath: string, uiTheme: Theme): string {
|
||||
if (args.previewDiff) {
|
||||
return formatStreamingDiff(args.previewDiff, rawPath, uiTheme, "preview");
|
||||
}
|
||||
@@ -438,7 +408,7 @@ export const editToolRenderer = {
|
||||
if (fileCount > 1) {
|
||||
text += uiTheme.fg("dim", ` (+${fileCount - 1} more)`);
|
||||
}
|
||||
text += getCallPreview(editArgs, rawPath, uiTheme, renderContext);
|
||||
text += getCallPreview(editArgs, rawPath, uiTheme);
|
||||
if (applyPatchSummary?.error) {
|
||||
text += `\n\n${uiTheme.fg("error", truncateToWidth(replaceTabs(applyPatchSummary.error), CALL_TEXT_PREVIEW_WIDTH))}`;
|
||||
}
|
||||
|
||||
@@ -16,7 +16,6 @@ import type { Theme } from "../modes/theme/theme";
|
||||
import { type EditMode, resolveEditMode } from "../utils/edit-mode";
|
||||
import { computeEditDiff, type DiffError, type DiffResult } from "./diff";
|
||||
import { expandApplyPatchToEntries, expandApplyPatchToPreviewEntries } from "./modes/apply-patch";
|
||||
import { type ChunkToolEdit, computeChunkDiff, parseChunkEditPath } from "./modes/chunk";
|
||||
import { computeHashlineDiff, type HashlineToolEdit } from "./modes/hashline";
|
||||
import { computePatchDiff, type PatchEditEntry } from "./modes/patch";
|
||||
import type { ReplaceEditEntry } from "./modes/replace";
|
||||
@@ -226,70 +225,6 @@ const hashlineStrategy: EditStreamingStrategy<HashlineArgs> = {
|
||||
},
|
||||
};
|
||||
|
||||
interface ChunkArgs {
|
||||
path?: string;
|
||||
edits?: ChunkToolEdit[];
|
||||
__partialJson?: string;
|
||||
}
|
||||
|
||||
const chunkStrategy: EditStreamingStrategy<ChunkArgs> = {
|
||||
extractCompleteEdits(args, partialJson) {
|
||||
if (!args?.edits) return args;
|
||||
let edits = dropIncompleteLastEdit(args.edits, partialJson, "edits");
|
||||
// Extra guard: if partial JSON still contains `":nu` / `":nul` (partial
|
||||
// `null` literals), `partial-json` may have already surfaced the last
|
||||
// entry with `write === null`. When that entry's `}` hasn't closed
|
||||
// yet, it has already been dropped above. But if dropping was not
|
||||
// triggered (e.g. list still open and no new `{` after), also drop the
|
||||
// trailing null-write entry so the preview does not flicker with an
|
||||
// error for an incomplete string/null literal.
|
||||
if (partialJson && edits.length > 0) {
|
||||
const last = edits[edits.length - 1] as Partial<ChunkToolEdit> | undefined;
|
||||
const endsInPartialNull = /:\s*nu?l?\s*$/.test(partialJson.trimEnd());
|
||||
if (last && endsInPartialNull && last.write === null) {
|
||||
edits = edits.slice(0, -1);
|
||||
}
|
||||
}
|
||||
return { ...args, edits };
|
||||
},
|
||||
async computeDiffPreview(args, ctx) {
|
||||
const edits = args.edits ?? [];
|
||||
if (edits.length === 0) return null;
|
||||
// Group edits by file path
|
||||
const groups = new Map<string, ChunkToolEdit[]>();
|
||||
const fileOrder: string[] = [];
|
||||
for (const edit of edits) {
|
||||
if (!edit) continue;
|
||||
const editPath = edit.path ?? args.path;
|
||||
if (!editPath) continue;
|
||||
const { filePath } = parseChunkEditPath(editPath);
|
||||
if (!filePath) continue;
|
||||
let bucket = groups.get(filePath);
|
||||
if (!bucket) {
|
||||
bucket = [];
|
||||
groups.set(filePath, bucket);
|
||||
fileOrder.push(filePath);
|
||||
}
|
||||
bucket.push({ ...edit, path: editPath });
|
||||
}
|
||||
if (fileOrder.length === 0) return null;
|
||||
|
||||
const MAX_FILES = 5;
|
||||
const selected = fileOrder.slice(0, MAX_FILES);
|
||||
const previews: PerFileDiffPreview[] = [];
|
||||
for (const filePath of selected) {
|
||||
ctx.signal.throwIfAborted();
|
||||
const fileEdits = groups.get(filePath) ?? [];
|
||||
const result = await computeChunkDiff({ path: filePath, edits: fileEdits }, ctx.cwd, { signal: ctx.signal });
|
||||
previews.push(toPerFilePreview(filePath, result));
|
||||
}
|
||||
return previews;
|
||||
},
|
||||
renderStreamingFallback() {
|
||||
return "";
|
||||
},
|
||||
};
|
||||
|
||||
interface ApplyPatchArgs {
|
||||
input?: string;
|
||||
}
|
||||
@@ -365,7 +300,6 @@ export const EDIT_MODE_STRATEGIES: Record<EditMode, EditStreamingStrategy<unknow
|
||||
replace: replaceStrategy as EditStreamingStrategy<unknown>,
|
||||
patch: patchStrategy as EditStreamingStrategy<unknown>,
|
||||
hashline: hashlineStrategy as EditStreamingStrategy<unknown>,
|
||||
chunk: chunkStrategy as EditStreamingStrategy<unknown>,
|
||||
apply_patch: applyPatchStrategy as EditStreamingStrategy<unknown>,
|
||||
vim: vimStrategy,
|
||||
atom: atomStrategy as EditStreamingStrategy<unknown>,
|
||||
|
||||
@@ -298,11 +298,6 @@ const OPTION_PROVIDERS: Partial<Record<SettingPath, OptionProvider>> = {
|
||||
{ value: "1000", label: "1000 lines" },
|
||||
{ value: "5000", label: "5000 lines" },
|
||||
],
|
||||
"read.anchorstyle": [
|
||||
{ value: "full", label: "Full", description: "Show the kind prefix and identifier" },
|
||||
{ value: "kind", label: "Kind", description: "Show only the kind prefix plus checksum" },
|
||||
{ value: "bare", label: "Bare", description: "Show only the checksum" },
|
||||
],
|
||||
// Todo auto-clear delay
|
||||
"tasks.todoClearDelay": [
|
||||
{ value: "0", label: "Instant" },
|
||||
|
||||
@@ -109,7 +109,7 @@ export class ToolExecutionComponent extends Container {
|
||||
isError?: boolean;
|
||||
details?: any;
|
||||
};
|
||||
// Edit preview state (single-file for legacy modes, multi-file for chunk)
|
||||
// Edit preview state
|
||||
#editMode?: EditMode;
|
||||
#editDiffPreview?: PerFileDiffPreview[];
|
||||
#editDiffScheduleTimer?: NodeJS.Timeout;
|
||||
@@ -639,10 +639,7 @@ export class ToolExecutionComponent extends Container {
|
||||
return this.#args;
|
||||
}
|
||||
// Single-file previews feed the existing `previewDiff` channel consumed
|
||||
// by `formatStreamingDiff` in the renderer. Multi-file previews are
|
||||
// piped via `renderContext.perFileDiffPreview`, so the args we hand to
|
||||
// `renderCall` only need the first file's diff to preserve prior
|
||||
// single-file behavior.
|
||||
// by `formatStreamingDiff` in the renderer.
|
||||
const first = previews[0];
|
||||
if (!first?.diff) {
|
||||
return this.#args;
|
||||
@@ -687,9 +684,6 @@ export class ToolExecutionComponent extends Container {
|
||||
? { error: first.error }
|
||||
: { diff: first.diff ?? "", firstChangedLine: first.firstChangedLine };
|
||||
}
|
||||
if (previews.length > 1) {
|
||||
context.perFileDiffPreview = previews;
|
||||
}
|
||||
}
|
||||
context.renderDiff = renderDiff;
|
||||
}
|
||||
|
||||
@@ -1,158 +0,0 @@
|
||||
Edits files via syntax-aware chunks. Use `read(path="file.ts")` to read and discover chunks before editing.
|
||||
- `read` is the canonical read path for chunk source and `sel="?"` tree listings.
|
||||
- `write` rewrites the entire targeted region — best for most edits.
|
||||
- `insert` adds content before/after a chunk.
|
||||
- `delete` deletes a targeted chunk and must be explicit.
|
||||
|
||||
Call format: `{"edits": [{"path": "file:chunk#ID~", "write": "new body"}, …]}`
|
||||
|
||||
<rules>
|
||||
- **MUST** inspect first with `read`. Never invent chunk paths or IDs. Copy them from the latest `read` output or edit response.
|
||||
- `path` format: `file:selector` — e.g. `src/app.ts:fn_foo#thth~`. Append `~` for body, `^` for head, or nothing for the whole chunk. Include `#ID` for `write`/`delete`.
|
||||
- If the exact chunk path is unclear, run `read(path="file", sel="?")` and copy a selector from that listing.
|
||||
{{#if chunkAutoIndent}}
|
||||
- Use `\t` for indentation in `content`. Write content at indent-level 0 — the tool re-indents it to match the chunk's position in the file. For example, to replace `~` of a method, write the body starting at column 0:
|
||||
```
|
||||
content: "if (x) {\n\treturn true;\n}"
|
||||
```
|
||||
The tool adds the correct base indent automatically. Never manually pad with the chunk's own indentation.
|
||||
Multiple sibling body lines at the same level all start at column 0: `"print(a)\nprint(b)\nprint(c)\n"`. Only use `\t` when nesting deeper (e.g. `"if cond:\n\tinner\nouter\n"`).
|
||||
Before applying the target's base indent, the tool strips any common leading whitespace shared by all non-empty `write` lines as a safety net. Do not rely on that cleanup for mixed indentation; write `~` bodies at column 0 and use one `\t` per relative nesting level.
|
||||
Multi-line replacements use the same relative-indentation model: the replacement text is dedented, then re-indented to the matched source line. Do not include the chunk's base indentation in replacement text.
|
||||
**Common mistake** when replacing `~` of a function body: do NOT include the function's own indentation.
|
||||
Wrong: `"if b == 0:\n\t\treturn None\n\treturn a / b\n"` — adds the function's base `\t` to every line.
|
||||
Correct: `"if b == 0:\n\treturn None\nreturn a / b\n"` — `if` and `return a / b` at column 0, only `return None` gets `\t` for nesting.
|
||||
{{else}}
|
||||
- Match the file's literal tabs/spaces in `content`. Do not convert indentation to canonical `\t`.
|
||||
- Write content at indent-level 0 relative to the target region. For example, to replace `~` of a method, write:
|
||||
```
|
||||
content: "if (x) {\n return true;\n}"
|
||||
```
|
||||
The tool adds the correct base indent automatically, then preserves the tabs/spaces you used inside the snippet. Never manually pad with the chunk's own indentation.
|
||||
Before applying the target's base indent, the tool strips any common leading whitespace shared by all non-empty `write` lines as a safety net. Do not rely on that cleanup for mixed indentation; write `~` bodies at column 0.
|
||||
Multi-line replacements use the same relative-indentation model: the replacement text is dedented, then re-indented to the matched source line. Do not include the chunk's base indentation in replacement text.
|
||||
{{/if}}
|
||||
- Region suffixes only apply to chunks with a real head/body boundary (classes, functions, impl blocks, and similar containers). On code leaf chunks (enum variants, fields, single statements, and compound statements like `if`/`for`/`while`/`match`/`try`), `~` and `^` are rejected. Use the unsuffixed selector and supply the complete replacement content, or edit the parent container's `~` body.
|
||||
- Unsuffixed `write` on a leaf chunk uses your content verbatim after normal replacement; it is not a body-region rewrite. Include the exact indentation and punctuation the leaf needs in the file.
|
||||
- `^` head writes and `~` body writes use the same base-indent model: write content at column 0 relative to the target region, and the tool applies the chunk's file indentation.
|
||||
- `write` and `delete` require the current ID. `prepend`/`append` do not.
|
||||
- **IDs change after every edit.** The edit response always carries the new IDs — use those for the next call or run `read(path="file", sel="?")` to refresh. Never reuse an ID from before the latest edit.
|
||||
- Same-file edit batches are transactional: if any operation in that file fails, no changes from that file's batch are saved. Multi-file edit calls run per file, so a later file error does not roll back earlier files that already succeeded.
|
||||
</rules>
|
||||
|
||||
<critical>
|
||||
You **MUST** use the narrowest region that covers your change. Putting without a region overwrites the **entire chunk including leading comments, decorators, and attributes** — omitting them from `content` deletes them.
|
||||
|
||||
**`put` is total, not surgical.** The `content` you supply becomes the *complete* new content for the targeted region. Everything in the original region that you omit from `content` is deleted. Before using `put` on any chunk's `~`, verify the chunk does not contain children you intend to keep. If a chunk spans hundreds of lines and your change touches only a few, target a specific child chunk — not the parent.
|
||||
|
||||
**Group chunks (`stmts_*`, `imports_*`, `decls_*`) are containers.** They hold many sibling items (test functions, import statements, declarations). `put` on a group chunk's `~` overwrites **all** of its children. To edit one item inside a group, target that item's own chunk path. If no child chunk exists, use the specific child's chunk selector from `read` output — do not `put` the parent group.
|
||||
</critical>
|
||||
|
||||
<regions>
|
||||
In `read` output, lines marked `^` between the line number and `|` are **head** lines (doc comments, attributes/decorators, signature). Lines without `^` are **body** lines. Use this to decide which region to target:
|
||||
- `fn_foo#ID~` — **body only (the default choice for most edits).** Head lines (`^`) are preserved automatically — doc comments, attributes, and signature stay untouched. On code leaf chunks, this is rejected because there is no safe body boundary.
|
||||
- `fn_foo#ID^` — head only (decorators, attributes, doc comments, signature, opening delimiter). Body stays untouched.
|
||||
- `fn_foo#ID` — entire chunk including leading trivia. **You must include doc comments and attributes in `content`; omitting them deletes them.**
|
||||
- `chunk~` + `append`/`prepend` inserts *inside* the container. `chunk` + `append`/`prepend` inserts *outside*. Appending to a container without `~` emits a warning because it lands after the closing delimiter, not before it.
|
||||
|
||||
**Note on leading trivia:** whether a decorator/doc comment belongs to `^` depends on the parser. In Rust and Python, attributes and decorators are attached to the function chunk, so `^` covers them. In TypeScript/JavaScript, a `@decorator` + `/** jsdoc */` block immediately above a method often surfaces as a **separate sibling chunk** (shown as `chunk#ID` in the `?` listing) rather than as part of the function's `^`. JSDoc directly above a plain function is more likely to be absorbed into that function's `^`. If you need to rewrite a decorated member, run `read(path="file", sel="?")` and check for a sibling `chunk#ID` directly above your target.
|
||||
|
||||
**Python notes:** Python docstrings are body lines, not head lines. A `~` body write on a function that has a docstring deletes the docstring unless you include the docstring in `content`. Python enum members and nested functions/closures are often opaque inside their parent chunk and may not appear as addressable child chunks; rewrite the parent container body. Python decorated class/function `^` writes and Python `^` deletes are rejected because indentation-sensitive bodies can become attached to the wrong block while still parsing.
|
||||
|
||||
**Note on non-code formats:** for prose and data formats (markdown, YAML, JSON, frontmatter), unsupported `^` and `~` suffixes warn and fall back to whole-chunk editing. Always replace the entire chunk and include any delimiter syntax (fence backticks, `---` frontmatter markers, list markers, table rows, headings) in your `content` — omitting them deletes them. For markdown sections (`sect_*`), prefer unsuffixed whole-chunk replace because `^`/`~` on prose sections can replace the heading and child content too; if you only need the heading, target the heading child chunk shown in `sel="?"`. Fenced code blocks with a declared language are parsed again and can expose inner chunks such as `code_py#ID.fn_gre#ID`; target those inner chunks when available. Markdown root writes preserve fenced code indentation verbatim. Recognized pipe tables expose `row_N` children for row-level edits; table cells and list items are not independently addressable, so rewrite the whole list/table chunk for those structural changes. Appending a table-row-shaped string (`| value |`) to a table chunk inserts it before the trailing blank-line separator so it remains part of the table. Otherwise read with `raw` first and preserve the exact whitespace inside fences. To insert content after a markdown section heading, use `after` on the heading chunk (`sect_*.chunk` or `sect_*.chunk_1`) — not `before`/`prepend` on the section itself, which lands physically before the heading and gets absorbed by the preceding section on reparse.
|
||||
</regions>
|
||||
|
||||
<ops>
|
||||
Each edit entry has `path` (`file:selector`) plus **exactly one** operation field — `write`, `insert`, or `delete`. Never set more than one on the same entry. `write:null`, `write:""`, and bare `{path}` entries are rejected; they do not delete.
|
||||
|
||||
|fields|path (selector part)|effect|
|
||||
|---|---|---|
|
||||
|`write: "content"`|`file:chunk#ID`, `file:chunk#ID~`, or `file:chunk#ID^`|write complete new content to the region|
|
||||
|`delete: true`|`file:chunk#ID`|delete the chunk explicitly|
|
||||
|`insert: {loc, body}`|`file:chunk` or `file:chunk~`|insert before/after the chunk (`loc`: `"prepend"` or `"append"`)|
|
||||
</ops>
|
||||
|
||||
<examples>
|
||||
Given this `read` output for `counter.rs`:
|
||||
```
|
||||
| counter.rs·62L·rust·#anth
|
||||
|
|
||||
@imp#erhe
|
||||
1 |use std::fmt;
|
||||
|
|
||||
@struct_Counte#onat
|
||||
3^|/// A simple counter that tracks a value and its history.
|
||||
4^|#[derive(Debug, Clone)]
|
||||
5^|pub struct Counter {
|
||||
-@struct_Counte.field_value#enth
|
||||
6 | /// The current value.
|
||||
7 | value: i32,
|
||||
-@struct_Counte.field_max#seti
|
||||
8 | /// Maximum allowed value.
|
||||
9 | max: i32,
|
||||
10 |}
|
||||
|
|
||||
@impl_Counte#reha
|
||||
12^|impl Counter {
|
||||
-@impl_Counte.fn_new#ndas
|
||||
13^| /// Creates a new counter starting at zero.
|
||||
14^| pub fn new(max: i32) -> Self {
|
||||
15 | Self { value: 0, max }
|
||||
16 | }
|
||||
17 |
|
||||
-@impl_Counte.fn_increm#ouer
|
||||
18^| /// Increments the counter by one, clamping at max.
|
||||
19^| pub fn increment(&mut self) {
|
||||
20 | if self.value < self.max {
|
||||
21 | self.value += 1;
|
||||
22 | }
|
||||
23 | }
|
||||
24 |
|
||||
-@impl_Counte.fn_decrem#arve
|
||||
25^| /// Decrements the counter by one, clamping at zero.
|
||||
26^| pub fn decrement(&mut self) {
|
||||
27 | if self.value > 0 {
|
||||
28 | self.value -= 1;
|
||||
29 | }
|
||||
30 | }
|
||||
31 |
|
||||
-@impl_Counte.fn_get#arco
|
||||
32^| /// Returns the current value.
|
||||
33^| pub fn get(&self) -> i32 {
|
||||
34 | self.value
|
||||
35 | }
|
||||
36 |}
|
||||
|
|
||||
@impl_Displa#meha
|
||||
38^|impl fmt::Display for Counter {
|
||||
-@impl_Displa.fn_fmt#deri
|
||||
39^| fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
40 | write!(f, "Counter({}/{})", self.value, self.max)
|
||||
41 | }
|
||||
42 |}
|
||||
```
|
||||
Lines marked `^` between the line number and `|` are **head** lines (doc comments, attributes, signature). Lines without `^` are **body** lines. `~` replaces body lines only; `^` replaces head lines only.
|
||||
|
||||
# Put body (`~` — the common case)
|
||||
`{ "path": "counter.rs:impl_Counte.fn_increm#ouer~", "write": "self.value = (self.value + 1).min(self.max);\n" }`
|
||||
Only body changes; doc comment, signature, and closing `}` are preserved.
|
||||
# Write whole chunk (rewrite signature + doc + body)
|
||||
`{ "path": "counter.rs:impl_Counte.fn_increm#ouer", "write": "/// Increments by the given step, clamping at max.\npub fn increment(&mut self, step: i32) {\n\tself.value = (self.value + step).min(self.max);\n}\n" }`
|
||||
Everything is rewritten. Omitting the doc comment or signature deletes them.
|
||||
# Write head (`^` — attributes, doc comments, signature)
|
||||
`{ "path": "counter.rs:impl_Counte.fn_get#arco^", "write": "/// Returns the current counter value.\n#[inline]\npub fn get(&self) → i32 {\n" }`
|
||||
Head changes (all `^` lines + opening brace); body untouched.
|
||||
# Insert before a chunk (`prepend`)
|
||||
`{ "path": "counter.rs:impl_Counte.fn_get", "insert": { "loc": "prepend", "body": "/// Resets the counter to zero.\npub fn reset(&mut self) {\n\tself.value = 0;\n}\n\n" } }`
|
||||
# Insert after a chunk (`append`)
|
||||
`{ "path": "counter.rs:struct_Counte", "insert": { "loc": "append", "body": "\nimpl Default for Counter {\n\tfn default() → Self {\n\t\tSelf { value: 0, max: 100 }\n\t}\n}\n" } }`
|
||||
# Insert at start of container body (`~` + `prepend`)
|
||||
`{ "path": "counter.rs:impl_Counte~", "insert": { "loc": "prepend", "body": "/// Creates a counter starting at the given value.\npub fn with_value(value: i32, max: i32) → Self {\n\tSelf { value: value.min(max), max }\n}\n\n" } }`
|
||||
Lands at the top of the impl body, before existing methods.
|
||||
# Insert at end of container body (`~` + `append`)
|
||||
`{ "path": "counter.rs:impl_Counte~", "insert": { "loc": "append", "body": "\n/// Returns true if the counter is at its maximum.\npub fn is_maxed(&self) → bool {\n\tself.value ≥ self.max\n}\n" } }`
|
||||
Lands at the end of the impl body, before the closing `}`.
|
||||
# Delete a chunk
|
||||
`{ "path": "counter.rs:impl_Counte.fn_decrem#arve", "delete": true }`
|
||||
Removes the method including its doc comment and signature.
|
||||
</examples>
|
||||
@@ -14,14 +14,11 @@ Searches files using powerful regex matching.
|
||||
- Text output is line-number-prefixed
|
||||
{{/if}}
|
||||
{{/if}}
|
||||
{{#if IS_CHUNK_MODE}}
|
||||
- Text output is chunk-path-prefixed: `path:sel>123|content`
|
||||
{{/if}}
|
||||
</output>
|
||||
|
||||
<critical>
|
||||
- You **MUST** use the built-in Grep tool for any content search. Do **NOT** shell out to `grep`, `rg`, `ripgrep`, `ag`, `ack`, `git grep`, `awk`, `sed`-for-search, or any other CLI search via Bash — even for a single match, even "just to check quickly", even piped through other commands.
|
||||
- Bash `grep`/`rg` returns raw text without chunk paths, loses `.gitignore` semantics, bypasses result limits, and wastes tokens. The Grep tool is faster, structured, and already wired into the workspace — there is no scenario where Bash search is preferable.
|
||||
- Bash `grep`/`rg` loses `.gitignore` semantics, bypasses result limits, and wastes tokens. The Grep tool is faster, structured, and already wired into the workspace — there is no scenario where Bash search is preferable.
|
||||
- If you catch yourself typing `grep`, `rg`, or `| grep` in a Bash command, stop and re-issue the search through the Grep tool instead.
|
||||
- If the search is open-ended, requiring multiple rounds, you **MUST** use the Task tool with the explore subagent instead of chaining Grep calls yourself.
|
||||
</critical>
|
||||
|
||||
@@ -4,4 +4,4 @@ You **MUST** use the `poll` tool (in a loop, if necessary) instead of manually r
|
||||
|
||||
If the timeout elapses before any job changes state, it returns the current snapshot (still-running jobs and any already-completed deliveries) without erroring — call `poll` again to keep waiting.
|
||||
|
||||
You **MUST NOT** poll the same job repeatedly without evidence of progress. Between calls, inspect `read jobs://<id>` to confirm new output or activity. If a job is stalled, has hung, or is producing nothing useful, cancel it via `cancel_job` and try a different approach instead of waiting indefinitely.
|
||||
You **MUST NOT** poll the same job repeatedly without evidence of progress. Between calls, inspect `read jobs://<id>` to confirm new output or activity. If a job is stalled, has hung, or is producing nothing useful, cancel it via `cancel_job` and try a different approach instead of waiting indefinitely.
|
||||
|
||||
@@ -1,73 +0,0 @@
|
||||
Reads files using syntax-aware chunks. Also inspects directories, archives, SQLite databases, images, documents (PDF/DOCX/PPTX/XLSX/RTF/EPUB/ipynb), **and URLs**.
|
||||
|
||||
<instruction>
|
||||
The chunk-aware `read` variant returns AST-scoped chunks with current checksum IDs for structural editing, and otherwise behaves like `open` for non-code content.
|
||||
- You **MUST** parallelize calls when exploring related files
|
||||
- For URLs, `read` fetches the page and returns clean extracted text/markdown by default (reader-mode). It handles HTML pages, GitHub issues/PRs, Stack Overflow, Wikipedia, Reddit, NPM, arXiv, RSS/Atom, JSON endpoints, PDFs, etc. You **SHOULD** reach for `read` — not a browser/puppeteer tool — for fetching and inspecting web content.
|
||||
|
||||
## Parameters
|
||||
- `path` — file path or URL; may include `:selector` suffix (required)
|
||||
- `sel` — optional selector for chunks, line ranges, listing, or raw mode
|
||||
- `timeout` — seconds, for URLs only
|
||||
|
||||
## Selectors
|
||||
|
||||
|`sel` value|Behavior|
|
||||
|---|---|
|
||||
|*(omitted)*|Read full file as chunks (up to {{DEFAULT_LIMIT}} lines)|
|
||||
|`class_Foo`|Read a specific chunk|
|
||||
|`class_Foo.fn_bar#thth~`|Read a chunk region (body `~` / head `^`) by ID|
|
||||
|`?`|List all chunk paths with IDs|
|
||||
|`L50`|Read from line 50 onward (shorthand for L50 to EOF)|
|
||||
|`L50-L120`|Read lines 50 through 120|
|
||||
|`L20-L20`|Read exactly one line|
|
||||
|`raw`|Raw content without transformations (for URLs: untouched HTML)|
|
||||
|
||||
Max {{DEFAULT_MAX_LINES}} lines per call.
|
||||
|
||||
# Chunks
|
||||
Each anchor `@full.chunk.path#thth` (with `-` prefixes for nesting depth) in the output identifies a chunk. Use `full.chunk.path#thth` as-is to read truncated chunks.
|
||||
If you need a canonical target list, run `read(path="file", sel="?")`. That listing shows chunk paths with IDs and is the safest structural discovery mode. Summary lines in this listing are orientation hints; follow a selector with `read(path="file", sel="chunk#ID")` or use `raw` when you need exact source.
|
||||
Line numbers in the gutter are absolute file line numbers.
|
||||
|
||||
{{#if chunkAutoIndent}}
|
||||
Chunk reads normalize leading indentation so copied content round-trips cleanly into chunk edits.
|
||||
{{else}}
|
||||
Chunk reads preserve literal leading tabs/spaces from the file. When editing, keep the same whitespace characters you see here.
|
||||
{{/if}}
|
||||
`raw` shows the file's literal whitespace. Structured chunk views may normalize or display indentation for edit round-tripping, so use `raw` when exact tabs/spaces matter, especially inside markdown fenced code blocks.
|
||||
|
||||
IDs change after every edit. Use the new IDs from the edit response or refresh with `sel="?"` before the next `write`/`delete`. `insert` selectors may omit IDs, but still prefer fresh paths after structural edits.
|
||||
|
||||
Parser boundaries vary by language: TypeScript/JavaScript decorators and JSDoc above decorated methods may appear as sibling `chunk#ID` entries, Python decorators are part of the function/class head, Python docstrings are body lines, and Python enum members or nested closures may remain opaque inside their parent chunk. Decorated Python `^` writes and Python `^` deletes are rejected for safety.
|
||||
Markdown sections, lists, and tables are structural chunks. Recognized pipe tables expose `row_N` children for row-level edits; list items and table cells are not independently addressable. Fenced code blocks with a declared language are parsed again when possible, so functions inside a markdown fence can appear as addressable nested chunks.
|
||||
|
||||
Chunk trees: JS, TS, TSX, Python, Rust, Go. Others use blank-line fallback.
|
||||
# Inspection
|
||||
Extracts text from PDF, Word, PowerPoint, Excel, RTF, EPUB, and Jupyter notebook files. Can inspect images.
|
||||
|
||||
# Directories & Archives
|
||||
Directories and archive roots return a list of entries. Supports `.tar`, `.tar.gz`, `.tgz`, `.zip`. Use `archive.ext:path/inside/archive` to read contents.
|
||||
|
||||
# SQLite Databases
|
||||
When used against a SQLite database (`.sqlite`, `.sqlite3`, `.db`, `.db3`), returns structured database content.
|
||||
- `file.db` — list tables with row counts
|
||||
- `file.db:table` — table schema + sample rows
|
||||
- `file.db:table:key` — single row by primary key
|
||||
- `file.db:table?limit=50&offset=100` — paginated rows
|
||||
- `file.db:table?where=status='active'&order=created:desc` — filtered rows
|
||||
- `file.db?q=SELECT …` — read-only SELECT query
|
||||
|
||||
# URLs
|
||||
Extracts content from web pages, GitHub issues/PRs, Stack Overflow, Wikipedia, Reddit, NPM, arXiv, RSS/Atom feeds, JSON endpoints, PDFs at URLs, and similar text-based resources. Returns clean reader-mode text/markdown — no browser required. Use `sel="raw"` for untouched HTML; `timeout` to override the default request timeout. You **SHOULD** prefer `read` over a browser/puppeteer tool for fetching URL content; only use a browser when the page requires JS execution, authentication, or interactive actions (clicks, forms, scrolling).
|
||||
</instruction>
|
||||
|
||||
<critical>
|
||||
- You **MUST** `read` before editing — never invent chunk names or IDs.
|
||||
- Chunk names are truncated (e.g., `handleRequest` becomes `fn_handleRequ`). Always copy chunk paths from `read` or `?` output — never construct them from source identifiers.
|
||||
- You **MUST** use `read` (never bash `cat`/`head`/`tail`/`less`/`more`/`ls`/`tar`/`unzip`/`curl`/`wget`) for all file, directory, archive, and URL reads.
|
||||
- You **MUST NOT** reach for a browser/puppeteer tool to fetch static web content — `read` handles HTML, PDFs, JSON, feeds, and docs directly. Reserve browser tools for JS-heavy pages or interactive flows.
|
||||
- You **MUST** always include the `path` parameter; never call with `{}`.
|
||||
- For specific line ranges, use `sel`: `read(path="file", sel="L50-L150")` — not `cat -n file | sed`.
|
||||
- You **MAY** use `sel` with URL reads; the tool paginates cached fetched output.
|
||||
</critical>
|
||||
@@ -18,7 +18,7 @@ The `read` tool is multi-purpose and more capable than it looks — inspects fil
|
||||
|`L50`|Read from line 50 onward (shorthand for L50 to EOF)|
|
||||
|`L50-L120`|Read lines 50 through 120|
|
||||
|`L20-L20`|Read exactly one line|
|
||||
|`raw`|Skip line-numbering / hashline / chunking; return file content as plain text. For URLs: untouched HTML.|
|
||||
|`raw`|Skip line-numbering / hashline; return file content as plain text. For URLs: untouched HTML.|
|
||||
|
||||
Max {{DEFAULT_MAX_LINES}} lines per call.
|
||||
|
||||
|
||||
@@ -45,7 +45,7 @@ If `done`, `rm`, or `drop` omits both `task` and `phase`, it applies to all task
|
||||
|
||||
## Phase Anatomy
|
||||
- `name`: Short, human-readable noun phrase (1-3 words). Capitalize naturally.
|
||||
- Always prefix with a roman-numeral ordinal (`I.`, `II.`, `III.`, `IV.`, ...) to convey ordering — e.g. `I. Foundation`, `II. Auth`, `III. Routing`. Single-phase plans use `I.` too.
|
||||
- Always prefix with a roman-numeral ordinal (`I.`, `II.`, `III.`, `IV.`, …) to convey ordering — e.g. `I. Foundation`, `II. Auth`, `III. Routing`. Single-phase plans use `I.` too.
|
||||
- You **MUST NOT** use snake_case, `Phase1_*`, arabic numerals (`1.`), or letter prefixes (`A.`) — they render as ugly identifiers.
|
||||
|
||||
## Rules
|
||||
@@ -66,7 +66,7 @@ Create a todo list when:
|
||||
<examples>
|
||||
# Initial setup (multi-phase)
|
||||
`{"ops":[{"op":"replace","phases":[{"name":"I. Foundation","tasks":[{"content":"Scaffold crate"},{"content":"Wire workspace"}]},{"name":"II. Auth","tasks":[{"content":"Port credential store"},{"content":"Wire OAuth providers"}]},{"name":"III. Verification","tasks":[{"content":"Run cargo test"}]}]}]}`
|
||||
# Initial setup (single phase " still prefixed)
|
||||
# Initial setup (single phase — still prefixed)
|
||||
`{"ops":[{"op":"replace","phases":[{"name":"I. Implementation","tasks":[{"content":"Apply fix"},{"content":"Run tests"}]}]}]}`
|
||||
# Complete one task
|
||||
`{"ops":[{"op":"done","task":"task-2"}]}`
|
||||
|
||||
@@ -1,12 +1,10 @@
|
||||
import { invalidateFsScanCache } from "@oh-my-pi/pi-natives";
|
||||
import { invalidateChunkCache } from "../edit/modes/chunk";
|
||||
|
||||
/**
|
||||
* Invalidate shared filesystem scan caches after a content write/update.
|
||||
*/
|
||||
export function invalidateFsScanAfterWrite(path: string): void {
|
||||
invalidateFsScanCache(path);
|
||||
invalidateChunkCache(path);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -14,7 +12,6 @@ export function invalidateFsScanAfterWrite(path: string): void {
|
||||
*/
|
||||
export function invalidateFsScanAfterDelete(path: string): void {
|
||||
invalidateFsScanCache(path);
|
||||
invalidateChunkCache(path);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -25,9 +22,7 @@ export function invalidateFsScanAfterDelete(path: string): void {
|
||||
*/
|
||||
export function invalidateFsScanAfterRename(oldPath: string, newPath: string): void {
|
||||
invalidateFsScanCache(oldPath);
|
||||
invalidateChunkCache(oldPath);
|
||||
if (newPath !== oldPath) {
|
||||
invalidateFsScanCache(newPath);
|
||||
invalidateChunkCache(newPath);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,9 +6,8 @@ import type { Component } from "@oh-my-pi/pi-tui";
|
||||
import { Text } from "@oh-my-pi/pi-tui";
|
||||
import { prompt, untilAborted } from "@oh-my-pi/pi-utils";
|
||||
import { type Static, Type } from "@sinclair/typebox";
|
||||
import { type ChunkedGrepMatch, describeChunkedGrepMatch } from "../edit/modes/chunk";
|
||||
import type { RenderResultOptions } from "../extensibility/custom-tools/types";
|
||||
import { getLanguageFromPath, type Theme } from "../modes/theme/theme";
|
||||
import { type Theme } from "../modes/theme/theme";
|
||||
import grepDescription from "../prompts/tools/grep.md" with { type: "text" };
|
||||
import { DEFAULT_MAX_COLUMN, type TruncationResult, truncateHead } from "../session/streaming-output";
|
||||
import { Ellipsis, Hasher, type RenderCache, renderStatusLine, renderTreeList, truncateToWidth } from "../tui";
|
||||
@@ -83,7 +82,6 @@ export class GrepTool implements AgentTool<typeof grepSchema, GrepToolDetails> {
|
||||
this.description = prompt.render(grepDescription, {
|
||||
IS_HASHLINE_MODE: displayMode.hashLines,
|
||||
IS_LINE_NUMBER_MODE: !displayMode.hashLines && displayMode.lineNumbers,
|
||||
IS_CHUNK_MODE: displayMode.chunked,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -98,7 +96,6 @@ export class GrepTool implements AgentTool<typeof grepSchema, GrepToolDetails> {
|
||||
|
||||
return untilAborted(signal, async () => {
|
||||
const normalizedPattern = pattern.trim();
|
||||
const chunkMode = resolveEditMode(this.session) === "chunk";
|
||||
if (!normalizedPattern) {
|
||||
throw new ToolError("Pattern must not be empty");
|
||||
}
|
||||
@@ -297,124 +294,6 @@ export class GrepTool implements AgentTool<typeof grepSchema, GrepToolDetails> {
|
||||
}
|
||||
matchesByFile.get(relativePath)!.push(match);
|
||||
}
|
||||
if (chunkMode) {
|
||||
const annotatedMatches = await Promise.all(
|
||||
selectedMatches.map(match => {
|
||||
const relativePath = match.path.startsWith("/") ? match.path.slice(1) : match.path;
|
||||
const absoluteFilePath = isDirectory ? path.join(searchPath, relativePath) : searchPath;
|
||||
return describeChunkedGrepMatch({
|
||||
filePath: absoluteFilePath,
|
||||
lineNumber: match.lineNumber,
|
||||
line: match.line,
|
||||
cwd: this.session.cwd,
|
||||
language: getLanguageFromPath(absoluteFilePath),
|
||||
});
|
||||
}),
|
||||
);
|
||||
const chunkMatchesByFile = new Map<string, ChunkedGrepMatch[]>();
|
||||
for (const match of annotatedMatches) {
|
||||
recordFile(match.displayPath);
|
||||
if (!chunkMatchesByFile.has(match.displayPath)) {
|
||||
chunkMatchesByFile.set(match.displayPath, []);
|
||||
}
|
||||
chunkMatchesByFile.get(match.displayPath)!.push(match);
|
||||
}
|
||||
const renderChunkedMatchesForFile = (relativePath: string): string[] => {
|
||||
const renderedLines: string[] = [];
|
||||
const fileMatches = chunkMatchesByFile.get(relativePath) ?? [];
|
||||
if (fileMatches.length === 0) {
|
||||
return renderedLines;
|
||||
}
|
||||
const matchesByChunk = new Map<string, ChunkedGrepMatch[]>();
|
||||
for (const match of fileMatches) {
|
||||
const chunkKey = match.chunkPath ?? "";
|
||||
if (!matchesByChunk.has(chunkKey)) {
|
||||
matchesByChunk.set(chunkKey, []);
|
||||
}
|
||||
matchesByChunk.get(chunkKey)!.push(match);
|
||||
}
|
||||
for (const [chunkPath, chunkMatches] of matchesByChunk) {
|
||||
if (chunkPath) {
|
||||
const chunkChecksum = chunkMatches[0]?.chunkChecksum;
|
||||
const dashes = "-".repeat(chunkPath.split(".").length - 1);
|
||||
const anchor = chunkChecksum
|
||||
? `${dashes}@${chunkPath}#${chunkChecksum}`
|
||||
: `${dashes}@${chunkPath}`;
|
||||
renderedLines.push(anchor);
|
||||
}
|
||||
for (const match of chunkMatches) {
|
||||
renderedLines.push(` ${match.lineNumber}|${match.line}`);
|
||||
fileMatchCounts.set(relativePath, (fileMatchCounts.get(relativePath) ?? 0) + 1);
|
||||
}
|
||||
}
|
||||
return renderedLines;
|
||||
};
|
||||
if (isDirectory) {
|
||||
const filesByDirectory = new Map<string, string[]>();
|
||||
for (const relativePath of fileList) {
|
||||
const directory = path.dirname(relativePath).replace(/\\/g, "/");
|
||||
if (!filesByDirectory.has(directory)) {
|
||||
filesByDirectory.set(directory, []);
|
||||
}
|
||||
filesByDirectory.get(directory)!.push(relativePath);
|
||||
}
|
||||
for (const [directory, directoryFiles] of filesByDirectory) {
|
||||
if (directory === ".") {
|
||||
for (const relativePath of directoryFiles) {
|
||||
const renderedLines = renderChunkedMatchesForFile(relativePath);
|
||||
if (renderedLines.length === 0) continue;
|
||||
if (outputLines.length > 0) {
|
||||
outputLines.push("");
|
||||
}
|
||||
outputLines.push(`# ${path.basename(relativePath)}`);
|
||||
outputLines.push(...renderedLines);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
const renderedFiles = directoryFiles
|
||||
.map(relativePath => ({ relativePath, lines: renderChunkedMatchesForFile(relativePath) }))
|
||||
.filter(file => file.lines.length > 0);
|
||||
if (renderedFiles.length === 0) continue;
|
||||
if (outputLines.length > 0) {
|
||||
outputLines.push("");
|
||||
}
|
||||
outputLines.push(`# ${directory}`);
|
||||
for (const { relativePath, lines } of renderedFiles) {
|
||||
outputLines.push(`## └─ ${path.basename(relativePath)}`);
|
||||
outputLines.push(...lines);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (const relativePath of fileList) {
|
||||
outputLines.push(...renderChunkedMatchesForFile(relativePath));
|
||||
}
|
||||
}
|
||||
if (matchLimitReached || result.limitReached) {
|
||||
outputLines.push("", limitMessage);
|
||||
}
|
||||
const rawOutput = outputLines.join("\n");
|
||||
const truncation = truncateHead(rawOutput, { maxLines: Number.MAX_SAFE_INTEGER });
|
||||
const truncated = Boolean(matchLimitReached || result.limitReached || truncation.truncated);
|
||||
const details: GrepToolDetails = {
|
||||
scopePath,
|
||||
matchCount: selectedMatches.length,
|
||||
fileCount: fileList.length,
|
||||
files: fileList,
|
||||
fileMatches: fileList.map(path => ({
|
||||
path,
|
||||
count: fileMatchCounts.get(path) ?? 0,
|
||||
})),
|
||||
truncated,
|
||||
matchLimitReached: matchLimitReached ? effectiveLimit : undefined,
|
||||
resultLimitReached: result.limitReached ? internalLimit : undefined,
|
||||
};
|
||||
if (truncation.truncated) details.truncation = truncation;
|
||||
const resultBuilder = toolResult(details).text(truncation.content);
|
||||
if (truncation.truncated) {
|
||||
resultBuilder.truncation(truncation, { direction: "head" });
|
||||
}
|
||||
return resultBuilder.done();
|
||||
}
|
||||
const displayLines: string[] = [];
|
||||
const renderMatchesForFile = (relativePath: string): { model: string[]; display: string[] } => {
|
||||
const modelOut: string[] = [];
|
||||
|
||||
@@ -25,7 +25,7 @@ const WAIT_DURATION_MS: Record<string, number> = {
|
||||
};
|
||||
|
||||
function parseWaitDurationMs(value: string | undefined): number {
|
||||
return (value && WAIT_DURATION_MS[value]) ?? WAIT_DURATION_MS["30s"];
|
||||
return (value ? WAIT_DURATION_MS[value] : undefined) ?? WAIT_DURATION_MS["30s"];
|
||||
}
|
||||
|
||||
interface PollResult {
|
||||
|
||||
@@ -9,20 +9,11 @@ import { Text } from "@oh-my-pi/pi-tui";
|
||||
import { getRemoteDir, prompt, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils";
|
||||
import { type Static, Type } from "@sinclair/typebox";
|
||||
import { formatHashLines } from "../edit/line-hash";
|
||||
import {
|
||||
type ChunkReadTarget,
|
||||
formatChunkedRead,
|
||||
parseChunkReadPath,
|
||||
parseChunkSelector,
|
||||
resolveAnchorStyle,
|
||||
resolveChunkAutoIndent,
|
||||
} from "../edit/modes/chunk";
|
||||
import type { RenderResultOptions } from "../extensibility/custom-tools/types";
|
||||
import { parseInternalUrl } from "../internal-urls/parse";
|
||||
import type { InternalUrl } from "../internal-urls/types";
|
||||
import { getLanguageFromPath, type Theme } from "../modes/theme/theme";
|
||||
import readDescription from "../prompts/tools/read.md" with { type: "text" };
|
||||
import readChunkDescription from "../prompts/tools/read-chunk.md" with { type: "text" };
|
||||
import type { ToolSession } from "../sdk";
|
||||
import {
|
||||
DEFAULT_MAX_BYTES,
|
||||
@@ -71,12 +62,6 @@ import {
|
||||
import { ToolAbortError, ToolError, throwIfAborted } from "./tool-errors";
|
||||
import { toolResult } from "./tool-result";
|
||||
|
||||
const PROSE_LANGUAGES = new Set(["markdown", "text", "log", "asciidoc", "restructuredtext"]);
|
||||
|
||||
function isProseLanguage(language: string | undefined): boolean {
|
||||
return language !== undefined && PROSE_LANGUAGES.has(language);
|
||||
}
|
||||
|
||||
// Document types converted to markdown via markit.
|
||||
const CONVERTIBLE_EXTENSIONS = new Set([".pdf", ".doc", ".docx", ".ppt", ".pptx", ".xls", ".xlsx", ".rtf", ".epub"]);
|
||||
|
||||
@@ -370,7 +355,6 @@ export interface ReadToolDetails {
|
||||
isDirectory?: boolean;
|
||||
resolvedPath?: string;
|
||||
suffixResolution?: { from: string; to: string };
|
||||
chunk?: ChunkReadTarget;
|
||||
url?: string;
|
||||
finalUrl?: string;
|
||||
contentType?: string;
|
||||
@@ -389,16 +373,14 @@ type ReadParams = ReadToolInput;
|
||||
type ParsedSelector =
|
||||
| { kind: "none" }
|
||||
| { kind: "raw" }
|
||||
| { kind: "lines"; startLine: number; endLine: number | undefined }
|
||||
| { kind: "chunk"; selector: string };
|
||||
| { kind: "lines"; startLine: number; endLine: number | undefined };
|
||||
|
||||
const LINE_RANGE_RE = /^L(\d+)(?:-L?(\d+))?$/i;
|
||||
|
||||
function parseSel(sel: string | undefined): ParsedSelector {
|
||||
if (!sel || sel.length === 0) return { kind: "none" };
|
||||
const normalizedSelector = parseChunkSelector(sel).selector ?? sel;
|
||||
if (normalizedSelector === "raw") return { kind: "raw" };
|
||||
const lineMatch = LINE_RANGE_RE.exec(normalizedSelector);
|
||||
if (sel === "raw") return { kind: "raw" };
|
||||
const lineMatch = LINE_RANGE_RE.exec(sel);
|
||||
if (lineMatch) {
|
||||
const rawStart = Number.parseInt(lineMatch[1]!, 10);
|
||||
if (rawStart < 1) {
|
||||
@@ -410,7 +392,7 @@ function parseSel(sel: string | undefined): ParsedSelector {
|
||||
}
|
||||
return { kind: "lines", startLine: rawStart, endLine: rawEnd };
|
||||
}
|
||||
return { kind: "chunk", selector: normalizedSelector };
|
||||
throw new ToolError(`Invalid sel '${sel}'. Use 'raw' or a line range like 'L50' or 'L50-L120'.`);
|
||||
}
|
||||
|
||||
/** Convert a line-range selector to the offset/limit pair used by internal pagination. */
|
||||
@@ -477,18 +459,12 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
Math.min(session.settings.get("read.defaultLimit") ?? DEFAULT_MAX_LINES, DEFAULT_MAX_LINES),
|
||||
);
|
||||
this.#inspectImageEnabled = session.settings.get("inspect_image.enabled");
|
||||
this.description =
|
||||
resolveEditMode(session) === "chunk"
|
||||
? prompt.render(readChunkDescription, {
|
||||
anchorStyle: resolveAnchorStyle(session.settings),
|
||||
chunkAutoIndent: resolveChunkAutoIndent(),
|
||||
})
|
||||
: prompt.render(readDescription, {
|
||||
DEFAULT_LIMIT: String(this.#defaultLimit),
|
||||
DEFAULT_MAX_LINES: String(DEFAULT_MAX_LINES),
|
||||
IS_HASHLINE_MODE: displayMode.hashLines,
|
||||
IS_LINE_NUMBER_MODE: !displayMode.hashLines && displayMode.lineNumbers,
|
||||
});
|
||||
this.description = prompt.render(readDescription, {
|
||||
DEFAULT_LIMIT: String(this.#defaultLimit),
|
||||
DEFAULT_MAX_LINES: String(DEFAULT_MAX_LINES),
|
||||
IS_HASHLINE_MODE: displayMode.hashLines,
|
||||
IS_LINE_NUMBER_MODE: !displayMode.hashLines && displayMode.lineNumbers,
|
||||
});
|
||||
}
|
||||
|
||||
async #resolveArchiveReadPath(readPath: string, signal?: AbortSignal): Promise<ResolvedArchiveReadPath | null> {
|
||||
@@ -930,7 +906,6 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
readPath = expandPath(readPath);
|
||||
}
|
||||
const displayMode = resolveFileDisplayMode(this.session);
|
||||
const chunkMode = resolveEditMode(this.session) === "chunk";
|
||||
|
||||
// Handle internal URLs (agent://, artifact://, memory://, skill://, rule://, local://, mcp://)
|
||||
const internalRouter = this.session.internalRouter;
|
||||
@@ -964,13 +939,8 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
return executeReadUrl(this.session, { path: parsedUrlTarget.path, timeout, raw: parsedUrlTarget.raw }, signal);
|
||||
}
|
||||
|
||||
const parsedReadPath = chunkMode ? parseChunkReadPath(readPath) : { filePath: readPath };
|
||||
const localReadPath = parsedReadPath.filePath;
|
||||
const pathSelectorParsed = chunkMode ? parseSel(parsedReadPath.selector) : { kind: "none" as const };
|
||||
const pathChunkSelector = pathSelectorParsed.kind === "chunk" ? pathSelectorParsed.selector : undefined;
|
||||
const selectorInput = sel ?? parsedReadPath.selector;
|
||||
const rawSelectorInput = sel ?? parsedReadPath.selector;
|
||||
const parsed = parseSel(selectorInput);
|
||||
const localReadPath = readPath;
|
||||
const parsed = parseSel(sel);
|
||||
|
||||
const archivePath = await this.#resolveArchiveReadPath(localReadPath, signal);
|
||||
if (archivePath) {
|
||||
@@ -1032,52 +1002,8 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
const ext = path.extname(absolutePath).toLowerCase();
|
||||
const hasEditTool = this.session.hasEditTool ?? true;
|
||||
const language = getLanguageFromPath(absolutePath);
|
||||
const skipChunksForExplore = !hasEditTool && !this.session.settings.get("read.explorechunks");
|
||||
const skipChunksForProse = isProseLanguage(language) && !this.session.settings.get("read.prosechunks");
|
||||
const shouldConvertWithMarkit =
|
||||
CONVERTIBLE_EXTENSIONS.has(ext) || (ext === ".ipynb" && (parsed.kind === "raw" || !chunkMode));
|
||||
|
||||
if (chunkMode && parsed.kind !== "raw" && !skipChunksForExplore && !skipChunksForProse) {
|
||||
const absoluteLineRange =
|
||||
pathChunkSelector && parsed.kind === "lines"
|
||||
? { startLine: parsed.startLine, endLine: parsed.endLine }
|
||||
: undefined;
|
||||
// sel= wins over path:chunk when both are provided (explicit param > embedded path).
|
||||
const effectiveSelector = sel ? selectorInput : (pathChunkSelector ?? selectorInput);
|
||||
const rawEffectiveSelector = sel ? selectorInput : (rawSelectorInput ?? effectiveSelector);
|
||||
const chunkReadPath =
|
||||
parsed.kind === "chunk" || (pathChunkSelector && !sel)
|
||||
? rawEffectiveSelector
|
||||
? `${localReadPath}:${rawEffectiveSelector}`
|
||||
: localReadPath
|
||||
: parsed.kind === "lines"
|
||||
? parsed.endLine !== undefined
|
||||
? `${localReadPath}:L${parsed.startLine}-L${parsed.endLine}`
|
||||
: `${localReadPath}:L${parsed.startLine}`
|
||||
: localReadPath;
|
||||
const chunkResult = await formatChunkedRead({
|
||||
filePath: absolutePath,
|
||||
readPath: chunkReadPath,
|
||||
cwd: this.session.cwd,
|
||||
language,
|
||||
omitChecksum: !hasEditTool,
|
||||
anchorStyle: resolveAnchorStyle(this.session.settings),
|
||||
absoluteLineRange,
|
||||
});
|
||||
let text = chunkResult.text;
|
||||
if (suffixResolution) {
|
||||
text = prependSuffixResolutionNotice(text, suffixResolution);
|
||||
}
|
||||
return toolResult<ReadToolDetails>({
|
||||
resolvedPath: absolutePath,
|
||||
suffixResolution,
|
||||
chunk: chunkResult.chunk,
|
||||
})
|
||||
.text(text)
|
||||
.sourcePath(absolutePath)
|
||||
.done();
|
||||
}
|
||||
|
||||
CONVERTIBLE_EXTENSIONS.has(ext) || (ext === ".ipynb" && parsed.kind === "raw");
|
||||
// Read the file based on type
|
||||
let content: Array<TextContent | ImageContent>;
|
||||
let details: ReadToolDetails = {};
|
||||
@@ -1160,31 +1086,6 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
content = [{ type: "text", text: `[Cannot read ${ext} file: conversion failed]` }];
|
||||
}
|
||||
} else {
|
||||
// Chunk mode: dispatch to chunk tree unless raw or line range requested
|
||||
if (chunkMode && parsed.kind !== "raw" && parsed.kind !== "lines") {
|
||||
const chunkSel = parsed.kind === "chunk" ? parsed.selector : undefined;
|
||||
const chunkResult = await formatChunkedRead({
|
||||
filePath: absolutePath,
|
||||
readPath: chunkSel ? `${localReadPath}:${chunkSel}` : localReadPath,
|
||||
cwd: this.session.cwd,
|
||||
language: getLanguageFromPath(absolutePath),
|
||||
omitChecksum: !(this.session.hasEditTool ?? true),
|
||||
anchorStyle: resolveAnchorStyle(this.session.settings),
|
||||
});
|
||||
let text = chunkResult.text;
|
||||
if (suffixResolution) {
|
||||
text = prependSuffixResolutionNotice(text, suffixResolution);
|
||||
}
|
||||
return toolResult<ReadToolDetails>({
|
||||
resolvedPath: absolutePath,
|
||||
suffixResolution,
|
||||
chunk: chunkResult.chunk,
|
||||
})
|
||||
.text(text)
|
||||
.sourcePath(absolutePath)
|
||||
.done();
|
||||
}
|
||||
|
||||
// Raw text or line-range mode
|
||||
const { offset, limit } = selToOffsetLimit(parsed);
|
||||
const startLine = offset ? Math.max(0, offset - 1) : 0;
|
||||
|
||||
@@ -1,13 +1,12 @@
|
||||
import { $env, $flag } from "@oh-my-pi/pi-utils";
|
||||
|
||||
export type EditMode = "replace" | "patch" | "hashline" | "chunk" | "vim" | "apply_patch" | "atom";
|
||||
export type EditMode = "replace" | "patch" | "hashline" | "vim" | "apply_patch" | "atom";
|
||||
|
||||
export const DEFAULT_EDIT_MODE: EditMode = "hashline";
|
||||
|
||||
const EDIT_MODE_IDS = {
|
||||
apply_patch: "apply_patch",
|
||||
atom: "atom",
|
||||
chunk: "chunk",
|
||||
hashline: "hashline",
|
||||
patch: "patch",
|
||||
replace: "replace",
|
||||
|
||||
@@ -7,7 +7,6 @@ import { resolveEditMode } from "./edit-mode";
|
||||
export interface FileDisplayMode {
|
||||
lineNumbers: boolean;
|
||||
hashLines: boolean;
|
||||
chunked: boolean;
|
||||
}
|
||||
|
||||
/** Session-like object providing settings and tool availability for display mode resolution. */
|
||||
@@ -33,10 +32,8 @@ export function resolveFileDisplayMode(session: FileDisplayModeSession, options?
|
||||
const usesHashLineAnchors = editMode === "hashline" || editMode === "atom";
|
||||
const raw = options?.raw === true;
|
||||
const hashLines = !raw && hasEditTool && usesHashLineAnchors && settings.get("readHashLines") !== false;
|
||||
const chunked = !raw && hasEditTool && editMode === "chunk";
|
||||
return {
|
||||
hashLines,
|
||||
lineNumbers: !raw && (hashLines || settings.get("readLineNumbers") === true),
|
||||
chunked,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1,33 +0,0 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { runReadCommand } from "@oh-my-pi/pi-coding-agent/cli/read-cli";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types";
|
||||
|
||||
describe("runReadCommand URL handling", () => {
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
it("delegates URL inputs through the read tool pipeline", async () => {
|
||||
const cwd = path.join(os.tmpdir(), "read-cli-url-test");
|
||||
const settings = Settings.isolated({ "fetch.enabled": true });
|
||||
const pageUrl = "https://example.com/cli-read";
|
||||
const consoleLogSpy = vi.spyOn(console, "log").mockImplementation(() => {});
|
||||
vi.spyOn(Settings, "init").mockResolvedValue(settings);
|
||||
vi.spyOn(scrapers, "loadPage").mockResolvedValue({
|
||||
ok: true,
|
||||
status: 200,
|
||||
contentType: "text/plain",
|
||||
finalUrl: pageUrl,
|
||||
content: "CLI URL content",
|
||||
});
|
||||
const cwdSpy = vi.spyOn(process, "cwd").mockReturnValue(cwd);
|
||||
|
||||
await runReadCommand({ path: pageUrl });
|
||||
|
||||
expect(cwdSpy).toHaveBeenCalled();
|
||||
expect(consoleLogSpy).toHaveBeenCalledWith(expect.stringContaining("CLI URL content"));
|
||||
});
|
||||
});
|
||||
@@ -4,14 +4,10 @@ import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import {
|
||||
adjustIndentation,
|
||||
computeChunkDiff,
|
||||
computeEditDiff,
|
||||
computeHashlineDiff,
|
||||
DEFAULT_FUZZY_THRESHOLD,
|
||||
findMatch,
|
||||
loadChunkSource,
|
||||
parseChunkEditPath,
|
||||
parseChunkReadPath,
|
||||
} from "@oh-my-pi/pi-coding-agent/edit";
|
||||
|
||||
describe("findMatch", () => {
|
||||
@@ -162,127 +158,6 @@ describe("findMatch", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("computeChunkDiff", () => {
|
||||
let tmpDir: string;
|
||||
beforeEach(async () => {
|
||||
tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "compute-chunk-"));
|
||||
});
|
||||
afterEach(async () => {
|
||||
await fs.rm(tmpDir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test("returns { error } when chunk selector cannot resolve", async () => {
|
||||
const file = path.join(tmpDir, "c.ts");
|
||||
await fs.writeFile(file, "export const x = 1;\n");
|
||||
const result = await computeChunkDiff(
|
||||
{
|
||||
path: "c.ts:fn_does_not_exist#ABCD",
|
||||
edits: [
|
||||
{
|
||||
path: "c.ts:fn_does_not_exist#ABCD",
|
||||
write: "console.log('replaced')\n",
|
||||
},
|
||||
],
|
||||
},
|
||||
tmpDir,
|
||||
);
|
||||
expect("error" in result).toBe(true);
|
||||
});
|
||||
|
||||
test("returns { error } when path is empty", async () => {
|
||||
const result = await computeChunkDiff({ path: "", edits: [{ path: "", write: "x\n" }] }, tmpDir);
|
||||
expect("error" in result).toBe(true);
|
||||
});
|
||||
|
||||
test("rejects write:null instead of previewing a delete", async () => {
|
||||
const file = path.join(tmpDir, "null-delete.ts");
|
||||
await fs.writeFile(file, "export const x = 1;\n");
|
||||
const result = await computeChunkDiff(
|
||||
{
|
||||
path: "null-delete.ts",
|
||||
edits: [{ path: "null-delete.ts", write: null }],
|
||||
},
|
||||
tmpDir,
|
||||
);
|
||||
expect("error" in result).toBe(true);
|
||||
if ("error" in result) {
|
||||
expect(result.error).toContain("write:null no longer deletes chunks");
|
||||
}
|
||||
});
|
||||
|
||||
test("rejects bare chunk edit entries instead of treating them as deletes", async () => {
|
||||
const file = path.join(tmpDir, "bare.ts");
|
||||
await fs.writeFile(file, "export const x = 1;\n");
|
||||
const result = await computeChunkDiff(
|
||||
{
|
||||
path: "bare.ts",
|
||||
edits: [{ path: "bare.ts" }],
|
||||
},
|
||||
tmpDir,
|
||||
);
|
||||
expect("error" in result).toBe(true);
|
||||
if ("error" in result) {
|
||||
expect(result.error).toContain("no operation specified");
|
||||
}
|
||||
});
|
||||
|
||||
test("rejects write empty string instead of previewing a destructive empty replacement", async () => {
|
||||
const file = path.join(tmpDir, "empty-write.ts");
|
||||
await fs.writeFile(file, "export const x = 1;\n");
|
||||
const result = await computeChunkDiff(
|
||||
{
|
||||
path: "empty-write.ts",
|
||||
edits: [{ path: "empty-write.ts", write: "" }],
|
||||
},
|
||||
tmpDir,
|
||||
);
|
||||
expect("error" in result).toBe(true);
|
||||
if ("error" in result) {
|
||||
expect(result.error).toContain('write:"" is a destructive empty replacement');
|
||||
}
|
||||
});
|
||||
|
||||
test("aborts when signal fires before compute completes", async () => {
|
||||
const controller = new AbortController();
|
||||
controller.abort();
|
||||
const result = await computeChunkDiff(
|
||||
{
|
||||
path: "d.ts",
|
||||
edits: [{ path: "d.ts", write: "foo\n" }],
|
||||
},
|
||||
tmpDir,
|
||||
{ signal: controller.signal },
|
||||
);
|
||||
expect("error" in result).toBe(true);
|
||||
});
|
||||
|
||||
test("computes diff for a root chunk replacement with valid checksum", async () => {
|
||||
const file = path.join(tmpDir, "e.ts");
|
||||
await fs.writeFile(file, "export const x = 1;\n");
|
||||
// Read the file once via loadChunkSource so the test does not depend on
|
||||
// knowing the internal chunk checksum scheme.
|
||||
const loaded = await loadChunkSource({ cwd: tmpDir, path: "e.ts" });
|
||||
expect(loaded.exists).toBe(true);
|
||||
expect(loaded.rawContent).toContain("export const x");
|
||||
});
|
||||
});
|
||||
|
||||
describe("chunk path parsing", () => {
|
||||
test("splits local plan URLs with chunk selectors after the URL path", () => {
|
||||
expect(parseChunkEditPath("local://PLAN.md:sct_0_T#SRJJ")).toEqual({
|
||||
filePath: "local://PLAN.md",
|
||||
selector: "sct_0_T#SRJJ",
|
||||
});
|
||||
expect(parseChunkReadPath("local://PLAN.md:sct_6_R.sct_6_u#MZKS")).toEqual({
|
||||
filePath: "local://PLAN.md",
|
||||
selector: "sct_6_R.sct_6_u#MZKS",
|
||||
});
|
||||
});
|
||||
|
||||
test("does not treat the local URL scheme colon as a chunk selector separator", () => {
|
||||
expect(parseChunkEditPath("local://PLAN.md")).toEqual({ filePath: "local://PLAN.md" });
|
||||
});
|
||||
});
|
||||
|
||||
describe("adjustIndentation", () => {
|
||||
test("adds indentation when actualText is more indented than oldText", () => {
|
||||
|
||||
@@ -30,45 +30,6 @@ describe("dropIncompleteLastEdit", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("chunk extractCompleteEdits", () => {
|
||||
const strategy = EDIT_MODE_STRATEGIES.chunk;
|
||||
|
||||
test("passes through a single complete entry", () => {
|
||||
const args = {
|
||||
edits: [{ path: "a.ts", write: "foo" }],
|
||||
__partialJson: '{"edits":[{"path":"a.ts","write":"foo"}]}',
|
||||
};
|
||||
const out = strategy.extractCompleteEdits(args, args.__partialJson) as typeof args;
|
||||
expect(out.edits).toHaveLength(1);
|
||||
});
|
||||
|
||||
test("drops trailing entry when partial JSON has open-brace after last close", () => {
|
||||
const args = {
|
||||
edits: [{ path: "a.ts", write: "foo" }, { path: "b.ts" }],
|
||||
__partialJson: '{"edits":[{"path":"a.ts","write":"foo"},{"path":"b.ts"',
|
||||
};
|
||||
const out = strategy.extractCompleteEdits(args, args.__partialJson) as typeof args;
|
||||
expect(out.edits).toHaveLength(1);
|
||||
expect(out.edits[0].path).toBe("a.ts");
|
||||
});
|
||||
|
||||
test("drops trailing entry when partial JSON ends in ':nu' (write: null guard)", () => {
|
||||
const args = {
|
||||
edits: [
|
||||
{ path: "a.ts", write: "foo" },
|
||||
{ path: "b.ts", write: null },
|
||||
],
|
||||
// simulates partial-json coercing the in-flight `nu` to `null`
|
||||
__partialJson: '{"edits":[{"path":"a.ts","write":"foo"},{"path":"b.ts","write":nu',
|
||||
};
|
||||
const out = strategy.extractCompleteEdits(args, args.__partialJson) as typeof args;
|
||||
// Last entry should be dropped because its `}` hasn't arrived yet, so
|
||||
// incomplete null-write errors are suppressed while streaming.
|
||||
expect(out.edits).toHaveLength(1);
|
||||
expect(out.edits[0].path).toBe("a.ts");
|
||||
});
|
||||
});
|
||||
|
||||
describe("apply_patch extractCompleteEdits", () => {
|
||||
const strategy = EDIT_MODE_STRATEGIES.apply_patch;
|
||||
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the `chunk` napi module (`ChunkState`, chunk schema, chunk rendering, chunk edit) and dropped `generate_chunk_schema()` from the build script
|
||||
|
||||
## [14.3.0] - 2026-04-25
|
||||
### Added
|
||||
|
||||
|
||||
Vendored
-311
@@ -1,69 +1,5 @@
|
||||
/* auto-generated by NAPI-RS */
|
||||
/* eslint-disable */
|
||||
/**
|
||||
* Parsed file as a chunk tree: query nodes, render views, format grep hits,
|
||||
* and apply edits.
|
||||
*/
|
||||
export declare class ChunkState {
|
||||
/**
|
||||
* Build chunk state by parsing `source` with the given `language` id (e.g.
|
||||
* `typescript`).
|
||||
*/
|
||||
static parse(source: string, language: string): ChunkState
|
||||
/** Normalized language identifier used for the tree-sitter parse. */
|
||||
get language(): string
|
||||
/** Full source text for this file. */
|
||||
get source(): string
|
||||
/** Stable checksum for the entire file contents. */
|
||||
get checksum(): string
|
||||
/** Line count of the source buffer. */
|
||||
get lineCount(): number
|
||||
/** Count of tree-sitter error nodes seen while building the tree. */
|
||||
get parseErrors(): number
|
||||
/** True when a fallback classifier produced the tree. */
|
||||
get fallback(): boolean
|
||||
/** Selector path string for the synthetic root (often empty). */
|
||||
get rootPath(): string
|
||||
/** Top-level child chunk paths under the root. */
|
||||
get rootChildren(): Array<string>
|
||||
/** Total number of chunk nodes. */
|
||||
get chunkCount(): number
|
||||
/** True when the parsed file contains unresolved merge conflicts. */
|
||||
hasConflicts(): boolean
|
||||
/** Count of unresolved merge conflicts represented in the chunk tree. */
|
||||
conflictCount(): number
|
||||
/** Summary for the root chunk, if it exists. */
|
||||
root(): ChunkInfo | null
|
||||
/** Look up [`ChunkInfo`] for a chunk selector path. */
|
||||
chunk(chunkPath: string): ChunkInfo | null
|
||||
/** Every chunk node as a [`ChunkInfo`] list. */
|
||||
chunks(): Array<ChunkInfo>
|
||||
/**
|
||||
* Direct children of `chunkPath` (use empty or omit for root); errors if
|
||||
* the path is missing.
|
||||
*/
|
||||
children(chunkPath?: string | undefined | null): Array<ChunkInfo>
|
||||
/** Chunk selector path that contains 1-based source line `line`, if any. */
|
||||
lineToContainingChunkPath(line: number): string | null
|
||||
/** Render a chunk subtree or listing as UTF-8 text for tools. */
|
||||
render(params: RenderParams): string
|
||||
/**
|
||||
* Parse `readPath` (selector, line scope, etc.) and return rendered text or
|
||||
* errors.
|
||||
*/
|
||||
renderRead(params: ReadRenderParams): ReadResult
|
||||
/**
|
||||
* Prefix a grep line with `display_path` and the chunk path for
|
||||
* `line_number`, when known.
|
||||
*/
|
||||
formatGrepLine(displayPath: string, lineNumber: number, line: string): string
|
||||
/**
|
||||
* Apply batch edits, re-parse, write files, and return updated state and
|
||||
* messaging.
|
||||
*/
|
||||
applyEdits(params: EditParams): EditResult
|
||||
}
|
||||
|
||||
/**
|
||||
* Long-lived macOS appearance observer.
|
||||
*
|
||||
@@ -347,95 +283,6 @@ export interface AstReplaceResult {
|
||||
parseErrors?: Array<string>
|
||||
}
|
||||
|
||||
/**
|
||||
* How chunk anchors are formatted in rendered output (name and checksum
|
||||
* visibility).
|
||||
*/
|
||||
export declare enum ChunkAnchorStyle {
|
||||
/** `[.name#crc]` style anchor. */
|
||||
Full = 'full',
|
||||
/** `[.kind#crc]` style anchor (kind is the name prefix before `_`). */
|
||||
Kind = 'kind',
|
||||
/** `[#crc]` style anchor. */
|
||||
Bare = 'bare',
|
||||
/** `[.name]` without checksum. */
|
||||
FullOmit = 'full-omit',
|
||||
/** `[.kind]` without checksum. */
|
||||
KindOmit = 'kind-omit',
|
||||
/** Minimal anchor without name or checksum. */
|
||||
None = 'none'
|
||||
}
|
||||
|
||||
/** Structural edit to apply relative to a chunk anchor. */
|
||||
export declare enum ChunkEditOp {
|
||||
/** Put new content into the targeted region. */
|
||||
Put = 'put',
|
||||
/** Find and replace a literal substring within the targeted region. */
|
||||
Replace = 'replace',
|
||||
/** Remove the targeted region. */
|
||||
Delete = 'delete',
|
||||
/** Insert `content` before the targeted region span. */
|
||||
Before = 'before',
|
||||
/** Insert `content` after the targeted region span. */
|
||||
After = 'after',
|
||||
/** Insert `content` at the start inside the targeted region. */
|
||||
Prepend = 'prepend',
|
||||
/** Insert `content` at the end inside the targeted region. */
|
||||
Append = 'append'
|
||||
}
|
||||
|
||||
/** How a chunk participates in a focus-scoped render pass. */
|
||||
export declare enum ChunkFocusMode {
|
||||
/** Emit full content and recurse normally. */
|
||||
Expanded = 'expanded',
|
||||
/** Emit just the opening anchor; do not recurse or emit body. */
|
||||
Collapsed = 'collapsed',
|
||||
/**
|
||||
* Emit opening + closing anchors; recurse into focused children only.
|
||||
* Interior gap lines between children are suppressed.
|
||||
*/
|
||||
Container = 'container'
|
||||
}
|
||||
|
||||
/** Summary of a single chunk node for tool output and navigation. */
|
||||
export interface ChunkInfo {
|
||||
/** Chunk selector path within the tree. */
|
||||
path: string
|
||||
/** Bare chunk identifier (without kind prefix), if available. */
|
||||
identifier?: string
|
||||
/** Stable checksum anchor for this chunk. */
|
||||
checksum: string
|
||||
/** 1-based start line in the source file (inclusive). */
|
||||
startLine: number
|
||||
/** 1-based end line in the source file (inclusive). */
|
||||
endLine: number
|
||||
/** Whether this node is a leaf (no child chunks). */
|
||||
leaf: boolean
|
||||
}
|
||||
|
||||
/** Result of resolving a chunk read request against the tree. */
|
||||
export declare enum ChunkReadStatus {
|
||||
/** Selector matched a chunk and content was produced. */
|
||||
Ok = 'ok',
|
||||
/** No chunk matched the requested selector. */
|
||||
NotFound = 'not_found',
|
||||
/** Chunk matched but does not support the requested region. */
|
||||
UnsupportedRegion = 'unsupported_region'
|
||||
}
|
||||
|
||||
/** Outcome of resolving which chunk was read for a `renderRead`-style request. */
|
||||
export interface ChunkReadTarget {
|
||||
/** Whether the selector matched. */
|
||||
status: ChunkReadStatus
|
||||
/** Sanitized selector string that was applied. */
|
||||
selector: string
|
||||
}
|
||||
|
||||
export declare enum ChunkRegion {
|
||||
Head = '^',
|
||||
Body = '~'
|
||||
}
|
||||
|
||||
/** Clipboard image payload encoded as PNG bytes. */
|
||||
export interface ClipboardImage {
|
||||
/** PNG-encoded image bytes. */
|
||||
@@ -469,78 +316,6 @@ export declare function copyToClipboard(text: string): void
|
||||
*/
|
||||
export declare function detectMacOSAppearance(): MacOSAppearance | null
|
||||
|
||||
/**
|
||||
* One edit in a batch; targets a chunk via `sel`/`crc` (with params-level
|
||||
* defaults).
|
||||
*/
|
||||
export interface EditOperation {
|
||||
/** Edit kind (replace, delete, insert relative to anchor). */
|
||||
op: ChunkEditOp
|
||||
/**
|
||||
* Chunk selector path; falls back to `EditParams.defaultSelector` when
|
||||
* omitted.
|
||||
*/
|
||||
sel?: string
|
||||
/**
|
||||
* Optional checksum anchor; falls back to `EditParams.defaultCrc` when
|
||||
* omitted.
|
||||
*/
|
||||
crc?: string
|
||||
/** Region to target. When omitted, targets the full chunk. */
|
||||
region?: ChunkRegion
|
||||
/** Replacement or inserted text (meaning depends on `op`). */
|
||||
content?: string
|
||||
/**
|
||||
* For `replace` op: literal substring to find inside the target chunk.
|
||||
* Must match exactly once.
|
||||
*/
|
||||
find?: string
|
||||
}
|
||||
|
||||
/** Arguments for applying a batch of chunk edits to a file. */
|
||||
export interface EditParams {
|
||||
/** Edits to apply in order. */
|
||||
operations: Array<EditOperation>
|
||||
/**
|
||||
* When true, normalize indentation for response rendering and inserted
|
||||
* content. When false, preserve literal tabs/spaces.
|
||||
*/
|
||||
normalizeIndent?: boolean
|
||||
/** Default chunk selector when an `EditOperation` omits `sel`. */
|
||||
defaultSelector?: string
|
||||
/** Default checksum when an `EditOperation` omits `crc`. */
|
||||
defaultCrc?: string
|
||||
/** Anchor formatting for rendered response text. */
|
||||
anchorStyle?: ChunkAnchorStyle
|
||||
/** Working directory used to resolve `filePath` and display paths. */
|
||||
cwd: string
|
||||
/** Path to the source file to edit (often relative to `cwd`). */
|
||||
filePath: string
|
||||
}
|
||||
|
||||
/**
|
||||
* Result of applying edits: new parse state plus before/after source and
|
||||
* messaging.
|
||||
*/
|
||||
export interface EditResult {
|
||||
/** Chunk tree state after applying edits and re-parsing. */
|
||||
state: ChunkState
|
||||
/** Full file text before edits. */
|
||||
diffBefore: string
|
||||
/** Full file text after edits. */
|
||||
diffAfter: string
|
||||
/** Rendered summary for tooling (hunks, anchors), driven by `anchorStyle`. */
|
||||
responseText: string
|
||||
/** Whether the on-disk source changed. */
|
||||
changed: boolean
|
||||
/** Whether the updated source re-parsed without fatal issues. */
|
||||
parseValid: boolean
|
||||
/** Absolute or normalized paths that were written or touched. */
|
||||
touchedPaths: Array<string>
|
||||
/** Non-fatal issues (e.g. selector warnings) collected during apply. */
|
||||
warnings: Array<string>
|
||||
}
|
||||
|
||||
/** Ellipsis strategy for [`truncate_to_width`]. */
|
||||
export declare enum Ellipsis {
|
||||
/** Use a single Unicode ellipsis character ("…"). */
|
||||
@@ -601,18 +376,6 @@ export declare enum FileType {
|
||||
Symlink = 3
|
||||
}
|
||||
|
||||
/** Path + focus mode pair for the N-API boundary (`HashMap` doesn't cross FFI). */
|
||||
export interface FocusedPath {
|
||||
path: string
|
||||
mode: ChunkFocusMode
|
||||
}
|
||||
|
||||
/**
|
||||
* Format one chunk anchor string for a node at `depth` using `style` and
|
||||
* optional checksum omission.
|
||||
*/
|
||||
export declare function formatAnchor(name: string, checksum: string, style: ChunkAnchorStyle, omitChecksum?: boolean | undefined | null): string
|
||||
|
||||
/** Fuzzy file path search for autocomplete. */
|
||||
export declare function fuzzyFind(options: FuzzyFindOptions): Promise<FuzzyFindResult>
|
||||
|
||||
@@ -1157,69 +920,6 @@ export interface PtyStartOptions {
|
||||
*/
|
||||
export declare function readImageFromClipboard(): Promise<ClipboardImage | undefined | null>
|
||||
|
||||
/**
|
||||
* Options for `ChunkState.renderRead`: selector path, display path, and
|
||||
* optional line scoping.
|
||||
*/
|
||||
export interface ReadRenderParams {
|
||||
/** Read selector (`sel=...` path, line range, or empty for whole tree). */
|
||||
readPath: string
|
||||
/** Path shown in titles and error messages (often the file path). */
|
||||
displayPath: string
|
||||
/** Optional language label for the rendered block. */
|
||||
languageTag?: string
|
||||
/** Hide checksums in rendered anchors. */
|
||||
omitChecksum: boolean
|
||||
/** Anchor formatting style. */
|
||||
anchorStyle?: ChunkAnchorStyle
|
||||
/** Optional absolute file line range to intersect with the resolved chunk. */
|
||||
absoluteLineRange?: VisibleLineRange
|
||||
/** Replace tabs in embedded previews. */
|
||||
tabReplacement?: string
|
||||
/** When true, normalize displayed indentation to canonical tabs. */
|
||||
normalizeIndent?: boolean
|
||||
}
|
||||
|
||||
/** Rendered chunk text plus optional resolution metadata for the read request. */
|
||||
export interface ReadResult {
|
||||
/** Rendered UTF-8 text (chunk tree, notice, or error message). */
|
||||
text: string
|
||||
/** When a selector was used, whether it matched and which selector applied. */
|
||||
chunk?: ChunkReadTarget
|
||||
}
|
||||
|
||||
/**
|
||||
* Options for `ChunkState.render`: which subtree to show and how anchors
|
||||
* appear.
|
||||
*/
|
||||
export interface RenderParams {
|
||||
/** Path of the chunk to render; `None` uses the tree root. */
|
||||
chunkPath?: string
|
||||
/** Title line shown above the tree (often the file path). */
|
||||
title: string
|
||||
/** Optional language label for the header block. */
|
||||
languageTag?: string
|
||||
/** Restrict output to an inclusive line range of the file. */
|
||||
visibleRange?: VisibleLineRange
|
||||
/** When true, list only direct children instead of a full subtree. */
|
||||
renderChildrenOnly: boolean
|
||||
/** Hide checksums in anchors when true. */
|
||||
omitChecksum: boolean
|
||||
/** Anchor formatting style for chunk headers. */
|
||||
anchorStyle?: ChunkAnchorStyle
|
||||
/** Include a one-line preview for leaf chunks. */
|
||||
showLeafPreview: boolean
|
||||
/** Replace tab characters in displayed previews (e.g. two spaces). */
|
||||
tabReplacement?: string
|
||||
/** When true, normalize displayed indentation to canonical tabs. */
|
||||
normalizeIndent?: boolean
|
||||
/**
|
||||
* When set, restrict rendering to these chunks with their specified focus
|
||||
* modes. Everything not in this list is skipped.
|
||||
*/
|
||||
focusedPaths?: Array<FocusedPath>
|
||||
}
|
||||
|
||||
/** Sampling filter for resize operations. */
|
||||
export declare enum SamplingFilter {
|
||||
/** Nearest-neighbor sampling (fast, low quality). */
|
||||
@@ -1395,17 +1095,6 @@ export declare function supportsLanguage(lang: string): boolean
|
||||
*/
|
||||
export declare function truncateToWidth(text: string, maxWidth: number, ellipsisKind: Ellipsis | undefined | null, pad: boolean | undefined | null, tabWidth: number): string
|
||||
|
||||
/**
|
||||
* Inclusive 1-based line range within a source file (used for scoped chunk
|
||||
* rendering).
|
||||
*/
|
||||
export interface VisibleLineRange {
|
||||
/** First line to include. */
|
||||
startLine: number
|
||||
/** Last line to include. */
|
||||
endLine: number
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculate visible width of text, excluding ANSI escape sequences.
|
||||
*
|
||||
|
||||
@@ -233,37 +233,6 @@ module.exports.AstMatchStrictness = {
|
||||
Signature: 'signature',
|
||||
Template: 'template',
|
||||
};
|
||||
module.exports.ChunkAnchorStyle = {
|
||||
Full: 'full',
|
||||
Kind: 'kind',
|
||||
Bare: 'bare',
|
||||
FullOmit: 'full-omit',
|
||||
KindOmit: 'kind-omit',
|
||||
None: 'none',
|
||||
};
|
||||
module.exports.ChunkEditOp = {
|
||||
Put: 'put',
|
||||
Replace: 'replace',
|
||||
Delete: 'delete',
|
||||
Before: 'before',
|
||||
After: 'after',
|
||||
Prepend: 'prepend',
|
||||
Append: 'append',
|
||||
};
|
||||
module.exports.ChunkFocusMode = {
|
||||
Expanded: 'expanded',
|
||||
Collapsed: 'collapsed',
|
||||
Container: 'container',
|
||||
};
|
||||
module.exports.ChunkReadStatus = {
|
||||
Ok: 'ok',
|
||||
NotFound: 'not_found',
|
||||
UnsupportedRegion: 'unsupported_region',
|
||||
};
|
||||
module.exports.ChunkRegion = {
|
||||
Head: '^',
|
||||
Body: '~',
|
||||
};
|
||||
module.exports.Ellipsis = {
|
||||
Unicode: 0,
|
||||
Ascii: 1,
|
||||
|
||||
@@ -98,7 +98,7 @@ Options:
|
||||
--tasks <ids> Comma-separated task IDs to run (default: all)
|
||||
--max-tasks <n> Max tasks to sample (default: 80, 0 = all)
|
||||
--fixtures <path> Fixtures directory or .tar.gz archive (default: built-in)
|
||||
--edit-variant <v> Edit variant: any string (e.g. replace, patch, hashline, chunk, vim, atom, apply_patch), or auto (default: auto)
|
||||
--edit-variant <v> Edit variant: any string (e.g. replace, patch, hashline, vim, atom, apply_patch), or auto (default: auto)
|
||||
--edit-fuzzy <bool> Fuzzy matching: true, false, auto (default: auto)
|
||||
--edit-fuzzy-threshold <n> Fuzzy threshold 0-1 or auto (default: auto)
|
||||
--auto-format Auto-format output files after verify (debug only)
|
||||
|
||||
@@ -136,7 +136,7 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
if (typeof summary.mutationIntentMatchRate === "number") {
|
||||
lines.push(`| Mutation Intent Match Rate | ${formatPercent(summary.mutationIntentMatchRate)} |`);
|
||||
}
|
||||
if (config.editVariant === "patch" || config.editVariant === "hashline" || config.editVariant === "chunk") {
|
||||
if (config.editVariant === "patch" || config.editVariant === "hashline") {
|
||||
lines.push(`| Patch Failure Rate | ${formatRate(totalEditFailures, totalEditAttempts)} |`);
|
||||
}
|
||||
lines.push(`| Tasks All Passing | ${summary.tasksWithAllPassing} |`);
|
||||
@@ -188,24 +188,6 @@ export function generateReport(result: BenchmarkResult): string {
|
||||
}
|
||||
}
|
||||
|
||||
if (summary.chunkEditSubtypes) {
|
||||
const order = ["append", "prepend", "replace", "delete"] as const;
|
||||
const total = order.reduce((sum, key) => sum + (summary.chunkEditSubtypes?.[key] ?? 0), 0);
|
||||
if (total > 0) {
|
||||
lines.push("### Chunk Edit Subtypes");
|
||||
lines.push("");
|
||||
lines.push("| Operation | Count | % |");
|
||||
lines.push("|-----------|-------|---|");
|
||||
for (const key of order) {
|
||||
const count = summary.chunkEditSubtypes[key] ?? 0;
|
||||
const pct = formatPercent(count / total);
|
||||
lines.push(`| ${key} | ${count} | ${pct} |`);
|
||||
}
|
||||
lines.push(`| **Total** | **${total}** | 100% |`);
|
||||
lines.push("");
|
||||
}
|
||||
}
|
||||
|
||||
lines.push("## Task Results");
|
||||
lines.push("");
|
||||
lines.push("| Task | File | Success | Edit Hit | R/E/W | Tokens (In/Out) | Time | Indent |");
|
||||
|
||||
@@ -192,8 +192,6 @@ function getEditPathFromArgs(args: unknown): string | null {
|
||||
|
||||
const HASHLINE_SUBTYPES = ["set", "set_range", "insert"] as const;
|
||||
|
||||
const CHUNK_OP_SUBTYPES = ["append", "prepend", "replace", "delete"] as const;
|
||||
|
||||
const BENCHMARK_TOOL_NAMES = ["read", "edit", "vim", "write", "apply_patch"] as const;
|
||||
const EDIT_TOOL_NAMES = ["edit", "vim", "apply_patch"] as const;
|
||||
|
||||
@@ -205,21 +203,6 @@ function isMutationTool(toolName: unknown): boolean {
|
||||
return isEditTool(toolName) || toolName === "write";
|
||||
}
|
||||
|
||||
function countChunkEditSubtypes(args: unknown): Record<string, number> {
|
||||
const counts: Record<string, number> = Object.fromEntries(CHUNK_OP_SUBTYPES.map(k => [k, 0]));
|
||||
if (!args || typeof args !== "object") return counts;
|
||||
const operations = (args as { operations?: unknown[] }).operations;
|
||||
if (!Array.isArray(operations)) return counts;
|
||||
for (const operation of operations) {
|
||||
if (!operation || typeof operation !== "object") continue;
|
||||
const op = (operation as { op?: string }).op;
|
||||
if (typeof op === "string" && op in counts) {
|
||||
counts[op]++;
|
||||
}
|
||||
}
|
||||
return counts;
|
||||
}
|
||||
|
||||
function countHashlineEditSubtypes(args: unknown): Record<string, number> {
|
||||
const counts: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0]));
|
||||
if (!args || typeof args !== "object") return counts;
|
||||
@@ -771,8 +754,6 @@ export interface TaskRunResult {
|
||||
editAutocorrectCount: number;
|
||||
/** Hashline edit subtype counts (replaceLine, replaceLines, etc.) — only when editVariant is hashline */
|
||||
hashlineEditSubtypes?: Record<string, number>;
|
||||
/** Chunk edit subtype counts — only when editVariant is chunk */
|
||||
chunkEditSubtypes?: Record<string, number>;
|
||||
mutationIntentMatched?: boolean;
|
||||
mutationIntentReason?: string;
|
||||
timeoutTelemetry?: PromptAttemptTelemetry;
|
||||
@@ -838,8 +819,6 @@ export interface BenchmarkSummary {
|
||||
mutationIntentMatchRate?: number;
|
||||
/** Hashline edit subtype totals — only when editVariant is hashline */
|
||||
hashlineEditSubtypes?: Record<string, number>;
|
||||
/** Chunk edit subtype totals — only when editVariant is chunk */
|
||||
chunkEditSubtypes?: Record<string, number>;
|
||||
}
|
||||
|
||||
export interface BenchmarkResult {
|
||||
@@ -931,7 +910,6 @@ async function runSingleTask(
|
||||
totalInputChars: 0,
|
||||
};
|
||||
const hashlineSubtypes: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0]));
|
||||
const chunkSubtypes: Record<string, number> = Object.fromEntries(CHUNK_OP_SUBTYPES.map(k => [k, 0]));
|
||||
|
||||
const logFile = path.join(TMP, `run-${task.id}-${runIndex}.jsonl`);
|
||||
const logEvent = async (event: unknown) => {
|
||||
@@ -1154,12 +1132,6 @@ async function runSingleTask(
|
||||
hashlineSubtypes[key] += counts[key];
|
||||
}
|
||||
}
|
||||
if (config.editVariant === "chunk" && args) {
|
||||
const counts = countChunkEditSubtypes(args);
|
||||
for (const key of CHUNK_OP_SUBTYPES) {
|
||||
chunkSubtypes[key] += counts[key];
|
||||
}
|
||||
}
|
||||
if (e.isError) {
|
||||
toolStats.editFailures++;
|
||||
const error = await appendNoChangeMutationHint(
|
||||
@@ -1286,7 +1258,6 @@ async function runSingleTask(
|
||||
editWarnings,
|
||||
editAutocorrectCount,
|
||||
hashlineEditSubtypes: config.editVariant === "hashline" ? hashlineSubtypes : undefined,
|
||||
chunkEditSubtypes: config.editVariant === "chunk" ? chunkSubtypes : undefined,
|
||||
mutationIntentMatched: mutationIntentValidation?.matched,
|
||||
mutationIntentReason: mutationIntentValidation?.reason,
|
||||
timeoutTelemetry,
|
||||
@@ -1335,7 +1306,6 @@ async function _runRpcBenchmarkRun(
|
||||
totalInputChars: 0,
|
||||
};
|
||||
const hashlineSubtypes: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0]));
|
||||
const chunkSubtypes: Record<string, number> = Object.fromEntries(CHUNK_OP_SUBTYPES.map(k => [k, 0]));
|
||||
|
||||
const logFile = path.join(sessionDir, `run-${task.id}-${runIndex}.jsonl`);
|
||||
const logEvent = async (event: unknown) => {
|
||||
@@ -1483,12 +1453,6 @@ async function _runRpcBenchmarkRun(
|
||||
hashlineSubtypes[key] += counts[key];
|
||||
}
|
||||
}
|
||||
if (config.editVariant === "chunk" && args) {
|
||||
const counts = countChunkEditSubtypes(args);
|
||||
for (const key of CHUNK_OP_SUBTYPES) {
|
||||
chunkSubtypes[key] += counts[key];
|
||||
}
|
||||
}
|
||||
if (e.isError) {
|
||||
toolStats.editFailures++;
|
||||
const toolError = await appendNoChangeMutationHint(
|
||||
@@ -1609,7 +1573,6 @@ async function _runRpcBenchmarkRun(
|
||||
editWarnings,
|
||||
editAutocorrectCount,
|
||||
hashlineEditSubtypes: config.editVariant === "hashline" ? hashlineSubtypes : undefined,
|
||||
chunkEditSubtypes: config.editVariant === "chunk" ? chunkSubtypes : undefined,
|
||||
mutationIntentMatched: mutationIntentValidation?.matched,
|
||||
mutationIntentReason: mutationIntentValidation?.reason,
|
||||
timeoutTelemetry,
|
||||
@@ -2161,15 +2124,6 @@ export async function runBenchmark(
|
||||
)
|
||||
: undefined;
|
||||
|
||||
const chunkEditSubtypes: Record<string, number> | undefined =
|
||||
config.editVariant === "chunk"
|
||||
? Object.fromEntries(
|
||||
CHUNK_OP_SUBTYPES.map(key => [
|
||||
key,
|
||||
allRuns.reduce((sum, r) => sum + (r.chunkEditSubtypes?.[key] ?? 0), 0),
|
||||
]),
|
||||
)
|
||||
: undefined;
|
||||
|
||||
const denom = effectiveRuns || 1;
|
||||
const summary: BenchmarkSummary = {
|
||||
@@ -2212,7 +2166,6 @@ export async function runBenchmark(
|
||||
transportFailureRuns,
|
||||
mutationIntentMatchRate,
|
||||
hashlineEditSubtypes,
|
||||
chunkEditSubtypes,
|
||||
};
|
||||
|
||||
return {
|
||||
|
||||
@@ -2,11 +2,10 @@
|
||||
"""
|
||||
Edit benchmark: tests the edit tool across models with a simple edit task.
|
||||
|
||||
Select the edit variant via the PI_EDIT_VARIANT env var (e.g. `chunk`, `vim`,
|
||||
Select the edit variant via the PI_EDIT_VARIANT env var (e.g. `vim`,
|
||||
`hashline`, `replace`, `patch`, `apply_patch`) or `--variant`.
|
||||
|
||||
Examples:
|
||||
PI_EDIT_VARIANT=chunk scripts/edit-benchmark.py
|
||||
PI_EDIT_VARIANT=vim scripts/edit-benchmark.py
|
||||
scripts/edit-benchmark.py --variant hashline
|
||||
"""
|
||||
@@ -54,11 +53,7 @@ def build_spec(variant: str) -> BenchmarkSpec:
|
||||
f"```rust\n"
|
||||
f"{EXPECTED_CONTENT}```\n"
|
||||
)
|
||||
retry = (
|
||||
'Use `read(path="test.rs")` to refresh chunk selectors if needed, then try again using the edit tool.'
|
||||
if variant == "chunk"
|
||||
else f"Please try again using the edit tool {mode_phrase}."
|
||||
)
|
||||
retry = f"Please try again using the edit tool {mode_phrase}."
|
||||
return BenchmarkSpec(
|
||||
description=f"Benchmark edit tool in {variant} mode across models with simple edit tasks.",
|
||||
workspace_prefix=f"{variant}-benchmark",
|
||||
|
||||
@@ -720,7 +720,7 @@ MARKDOWN_FIXTURE = (
|
||||
|
||||
| Surface | Expected stress |
|
||||
| --- | --- |
|
||||
| `main.ts` | Structural chunk addressing |
|
||||
| `main.ts` | Type/interface and class member edits |
|
||||
| `main.rs` | Enum and impl member edits |
|
||||
| `main.py` | Indentation-sensitive blocks |
|
||||
| `main.md` | Prose and block-level text edits |
|
||||
|
||||
Reference in New Issue
Block a user