feat: removed chunk-mode modules and read/edit entrypoints from pi-natives

- Removed `pi-natives` chunk language classifier modules and all core chunk subsystems (kind, state, render, edit, resolve).
- Removed chunk-mode CLI/read/edit entrypoints, including `read` command and chunk mode registration/prompt tooling.
- Removed chunk selectors from `read` and `grep` tools, switching behavior to raw/L-range handling.
- Fixed poll wait parsing to keep defaulting to `30s` when the provided value is empty.
This commit is contained in:
can1357
2026-04-26 08:19:02 +02:00
parent a227adad1b
commit 5ea1d55e56
88 changed files with 49 additions and 30753 deletions
-829
View File
@@ -1,373 +1,12 @@
use std::{
collections::{BTreeMap, BTreeSet, HashMap},
env,
fmt::Write as _,
fs,
path::{Path, PathBuf},
};
use serde::Deserialize;
const IDENTIFIER_FIELD_PRIORITY: &[&str] = &[
"name",
"identifier",
"attrpath",
"key",
"label",
"alias",
"field",
"member",
"property",
"tag",
"target",
"variable",
];
const BODY_FIELD_PRIORITY: &[&str] = &["body", "value", "declaration_list", "block", "members"];
const PROMOTION_FIELD_PRIORITY: &[&str] = &["definition", "declaration", "item", "member"];
#[derive(Clone, Copy)]
struct GrammarSpec {
language: &'static str,
package: &'static str,
node_types_rel: &'static str,
}
struct LockedPackage {
version: String,
source: Option<String>,
}
#[derive(Deserialize)]
struct RawTypeRef {
#[serde(rename = "type")]
kind: Option<String>,
named: bool,
}
#[derive(Deserialize)]
struct RawFieldSpec {
#[serde(default)]
multiple: bool,
#[serde(default)]
types: Vec<RawTypeRef>,
}
#[derive(Deserialize)]
struct RawNodeType {
#[serde(rename = "type")]
kind: Option<String>,
fields: Option<BTreeMap<String, RawFieldSpec>>,
children: Option<RawFieldSpec>,
subtypes: Option<Vec<RawTypeRef>>,
}
#[derive(serde::Serialize)]
struct GeneratedSchema {
languages: BTreeMap<String, BTreeMap<String, GeneratedNodeTypeSchema>>,
}
#[derive(serde::Serialize)]
struct GeneratedNodeTypeSchema {
identifier_fields: Vec<String>,
body_fields: Vec<String>,
promotion_fields: Vec<String>,
container_child_kinds: Vec<String>,
is_supertype: bool,
has_structural_children: bool,
}
const GRAMMARS: &[GrammarSpec] = &[
GrammarSpec {
language: "astro",
package: "tree-sitter-astro-next",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "bash",
package: "tree-sitter-bash",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "c",
package: "tree-sitter-c",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "clojure",
package: "tree-sitter-clojure",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "cmake",
package: "tree-sitter-cmake",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "cpp",
package: "tree-sitter-cpp",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "csharp",
package: "tree-sitter-c-sharp",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "dart",
package: "tree-sitter-dart",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "css",
package: "tree-sitter-css",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "diff",
package: "tree-sitter-diff",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "dockerfile",
package: "tree-sitter-dockerfile-updated",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "elixir",
package: "tree-sitter-elixir",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "erlang",
package: "tree-sitter-erlang",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "go",
package: "tree-sitter-go",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "graphql",
package: "tree-sitter-graphql",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "handlebars",
package: "tree-sitter-glimmer",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "haskell",
package: "tree-sitter-haskell",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "hcl",
package: "tree-sitter-hcl",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "html",
package: "tree-sitter-html",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "ini",
package: "tree-sitter-ini",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "java",
package: "tree-sitter-java",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "javascript",
package: "tree-sitter-javascript",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "json",
package: "tree-sitter-json",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "toml",
package: "tree-sitter-toml-ng",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "just",
package: "tree-sitter-just",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "julia",
package: "tree-sitter-julia",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "kotlin",
package: "tree-sitter-kotlin-sg",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "lua",
package: "tree-sitter-lua",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "make",
package: "tree-sitter-make",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "markdown",
package: "tree-sitter-md",
node_types_rel: "tree-sitter-markdown/src/node-types.json",
},
GrammarSpec {
language: "nix",
package: "tree-sitter-nix",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "objc",
package: "tree-sitter-objc",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "odin",
package: "tree-sitter-odin",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "perl",
package: "tree-sitter-perl-next",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "php",
package: "tree-sitter-php",
node_types_rel: "php/src/node-types.json",
},
GrammarSpec {
language: "powershell",
package: "tree-sitter-powershell",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "protobuf",
package: "tree-sitter-proto",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "python",
package: "tree-sitter-python",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "r",
package: "tree-sitter-r",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "regex",
package: "tree-sitter-regex",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "ruby",
package: "tree-sitter-ruby",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "rust",
package: "tree-sitter-rust",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "scala",
package: "tree-sitter-scala",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "solidity",
package: "tree-sitter-solidity",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "sql",
package: "tree-sitter-sequel",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "starlark",
package: "tree-sitter-starlark",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "svelte",
package: "tree-sitter-svelte-next",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "swift",
package: "tree-sitter-swift",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "toml",
package: "tree-sitter-toml-ng",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "tlaplus",
package: "tree-sitter-tlaplus",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "tsx",
package: "tree-sitter-typescript",
node_types_rel: "tsx/src/node-types.json",
},
GrammarSpec {
language: "typescript",
package: "tree-sitter-typescript",
node_types_rel: "typescript/src/node-types.json",
},
GrammarSpec {
language: "verilog",
package: "tree-sitter-verilog",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "vue",
package: "tree-sitter-vue-next",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "xml",
package: "tree-sitter-xml",
node_types_rel: "xml/src/node-types.json",
},
GrammarSpec {
language: "yaml",
package: "tree-sitter-yaml",
node_types_rel: "src/node-types.json",
},
GrammarSpec {
language: "zig",
package: "tree-sitter-zig",
node_types_rel: "src/node-types.json",
},
];
fn main() {
napi_build::setup();
generate_chunk_schema();
generate_minimizer_builtin_filters();
}
@@ -423,471 +62,3 @@ fn generate_minimizer_builtin_filters() {
fs::write(&output_path, concatenated)
.unwrap_or_else(|e| panic!("failed to write {}: {e}", output_path.display()));
}
fn generate_chunk_schema() {
let manifest_dir = env::var("CARGO_MANIFEST_DIR").expect("CARGO_MANIFEST_DIR should be set");
let workspace_root = Path::new(&manifest_dir)
.parent()
.and_then(Path::parent)
.expect("pi-natives should live under the workspace root");
let out_dir = PathBuf::from(env::var("OUT_DIR").expect("OUT_DIR should be set"));
let output_path = out_dir.join("chunk_schema.json");
let locked_packages = locked_packages(&workspace_root.join("Cargo.lock"));
let registry_roots = cargo_registry_roots();
let git_roots = cargo_git_checkout_roots();
let mut languages = BTreeMap::new();
for grammar in GRAMMARS {
let Some(locked) = locked_packages.get(grammar.package) else {
continue;
};
let Some(package_dir) =
find_locked_package_dir(&registry_roots, &git_roots, grammar.package, locked)
else {
continue;
};
let node_types_path = package_dir.join(grammar.node_types_rel);
if !node_types_path.exists() {
continue;
}
println!("cargo:rerun-if-changed={}", node_types_path.display());
let source =
fs::read_to_string(&node_types_path).expect("node-types.json should be readable");
let raw_nodes: Vec<RawNodeType> =
serde_json::from_str(&source).expect("node-types.json should parse");
let schemas = build_language_schema(raw_nodes);
if !schemas.is_empty() {
languages.insert(grammar.language.to_string(), schemas);
}
}
let generated = GeneratedSchema { languages };
let json = serde_json::to_string(&generated).expect("schema JSON should serialize");
fs::write(output_path, json).expect("schema JSON should write");
}
fn build_language_schema(raw_nodes: Vec<RawNodeType>) -> BTreeMap<String, GeneratedNodeTypeSchema> {
let mut raw_by_kind = HashMap::new();
for raw in raw_nodes {
let Some(kind) = raw.kind.clone() else {
continue;
};
raw_by_kind.insert(kind, raw);
}
let structural_state = compute_structural_state(&raw_by_kind);
let mut out = BTreeMap::new();
for (kind, raw) in &raw_by_kind {
let identifier_fields = pick_priority_fields(raw.fields.as_ref(), IDENTIFIER_FIELD_PRIORITY);
let body_fields = pick_priority_fields(raw.fields.as_ref(), BODY_FIELD_PRIORITY);
let promotion_fields =
collect_promotion_fields(raw, &structural_state, &identifier_fields, &body_fields);
let container_child_kinds = collect_child_container_kinds(raw, &structural_state);
let is_supertype = is_supertype(raw);
let has_structural_children = structural_state
.get(kind)
.is_some_and(|state| state.has_structural_children);
if identifier_fields.is_empty()
&& body_fields.is_empty()
&& promotion_fields.is_empty()
&& container_child_kinds.is_empty()
&& !is_supertype
&& !has_structural_children
{
continue;
}
out.insert(kind.clone(), GeneratedNodeTypeSchema {
identifier_fields,
body_fields,
promotion_fields,
container_child_kinds,
is_supertype,
has_structural_children,
});
}
out
}
fn pick_priority_fields(
fields: Option<&BTreeMap<String, RawFieldSpec>>,
priority: &[&str],
) -> Vec<String> {
let Some(fields) = fields else {
return Vec::new();
};
priority
.iter()
.filter(|field| fields.contains_key(**field))
.map(|field| (*field).to_string())
.collect()
}
fn collect_child_container_kinds(
raw: &RawNodeType,
structural_state: &HashMap<String, StructuralState>,
) -> Vec<String> {
let mut kinds = BTreeSet::new();
let child_types = raw
.children
.as_ref()
.map(|children| children.types.as_slice())
.unwrap_or_default();
for child in child_types {
if !child.named {
continue;
}
let Some(kind) = child.kind.as_deref() else {
continue;
};
if structural_state
.get(kind)
.copied()
.is_some_and(StructuralState::is_structural)
{
kinds.insert(kind.to_string());
}
}
kinds.into_iter().collect()
}
fn collect_promotion_fields(
raw: &RawNodeType,
structural_state: &HashMap<String, StructuralState>,
identifier_fields: &[String],
body_fields: &[String],
) -> Vec<String> {
let Some(fields) = raw.fields.as_ref() else {
return Vec::new();
};
PROMOTION_FIELD_PRIORITY
.iter()
.filter_map(|field_name| {
let spec = fields.get(*field_name)?;
if spec.multiple
|| identifier_fields.iter().any(|field| field == field_name)
|| body_fields.iter().any(|field| field == field_name)
{
return None;
}
let has_structural_type = spec.types.iter().any(|field_type| {
field_type.named
&& field_type
.kind
.as_deref()
.and_then(|kind| structural_state.get(kind))
.copied()
.is_some_and(StructuralState::is_structural)
});
has_structural_type.then(|| (*field_name).to_string())
})
.collect()
}
#[derive(Clone, Copy, Default)]
struct StructuralState {
is_structural: bool,
has_structural_children: bool,
}
impl StructuralState {
const fn is_structural(self) -> bool {
self.is_structural
}
}
fn compute_structural_state(
raw_by_kind: &HashMap<String, RawNodeType>,
) -> HashMap<String, StructuralState> {
let mut state = raw_by_kind
.iter()
.map(|(kind, raw)| {
let base_structural = is_supertype(raw)
|| raw.fields.as_ref().is_some_and(|fields| !fields.is_empty())
|| !named_child_type_kinds(raw).is_empty();
(kind.clone(), StructuralState {
is_structural: base_structural,
has_structural_children: false,
})
})
.collect::<HashMap<_, _>>();
loop {
let mut changed = false;
for (kind, raw) in raw_by_kind {
let next_has_structural_children =
named_child_type_kinds(raw).into_iter().any(|child_kind| {
state
.get(child_kind.as_str())
.copied()
.is_some_and(StructuralState::is_structural)
});
let entry = state
.get_mut(kind.as_str())
.expect("every raw node should have structural state");
let next_is_structural = entry.is_structural || next_has_structural_children;
if next_is_structural != entry.is_structural
|| next_has_structural_children != entry.has_structural_children
{
entry.is_structural = next_is_structural;
entry.has_structural_children = next_has_structural_children;
changed = true;
}
}
if !changed {
break;
}
}
state
}
fn is_supertype(raw: &RawNodeType) -> bool {
raw.subtypes
.as_ref()
.is_some_and(|subtypes| !subtypes.is_empty())
}
fn named_child_type_kinds(raw: &RawNodeType) -> BTreeSet<String> {
let mut kinds = BTreeSet::new();
if let Some(fields) = raw.fields.as_ref() {
for field in fields.values() {
for field_type in &field.types {
if field_type.named
&& let Some(kind) = &field_type.kind
{
kinds.insert(kind.clone());
}
}
}
}
if let Some(children) = raw.children.as_ref() {
for child in &children.types {
if child.named
&& let Some(kind) = &child.kind
{
kinds.insert(kind.clone());
}
}
}
kinds
}
fn cargo_registry_roots() -> Vec<PathBuf> {
let mut roots = Vec::new();
if let Some(cargo_home) = env::var_os("CARGO_HOME") {
roots.push(PathBuf::from(cargo_home).join("registry").join("src"));
}
if let Some(home) = env::var_os("HOME") {
roots.push(
PathBuf::from(home)
.join(".cargo")
.join("registry")
.join("src"),
);
}
roots
}
fn cargo_git_checkout_roots() -> Vec<PathBuf> {
let mut roots = Vec::new();
if let Some(cargo_home) = env::var_os("CARGO_HOME") {
roots.push(PathBuf::from(cargo_home).join("git").join("checkouts"));
}
if let Some(home) = env::var_os("HOME") {
roots.push(
PathBuf::from(home)
.join(".cargo")
.join("git")
.join("checkouts"),
);
}
roots
}
fn find_locked_package_dir(
registry_roots: &[PathBuf],
git_roots: &[PathBuf],
package: &str,
locked: &LockedPackage,
) -> Option<PathBuf> {
match locked.source.as_deref() {
Some(source) if source.starts_with("git+") => {
find_git_package_dir(git_roots, package, &locked.version, git_revision(source))
},
_ => find_registry_package_dir(registry_roots, package, &locked.version),
}
}
fn find_registry_package_dir(
registry_roots: &[PathBuf],
package: &str,
version: &str,
) -> Option<PathBuf> {
for registry_root in registry_roots {
let Ok(registry_dirs) = fs::read_dir(registry_root) else {
continue;
};
for registry_dir in registry_dirs.flatten() {
let candidate = registry_dir.path().join(format!("{package}-{version}"));
if candidate.exists() {
return Some(candidate);
}
}
}
None
}
fn find_git_package_dir(
git_roots: &[PathBuf],
package: &str,
version: &str,
revision: Option<&str>,
) -> Option<PathBuf> {
for git_root in git_roots {
let Ok(checkout_dirs) = fs::read_dir(git_root) else {
continue;
};
for checkout_dir in checkout_dirs.flatten() {
let Ok(revision_dirs) = fs::read_dir(checkout_dir.path()) else {
continue;
};
for revision_dir in revision_dirs.flatten() {
let revision_path = revision_dir.path();
let Some(revision_name) = revision_path.file_name().and_then(|name| name.to_str())
else {
continue;
};
if !revision_matches(revision_name, revision) {
continue;
}
if let Some(package_dir) = find_manifest_package_dir(&revision_path, package, version) {
return Some(package_dir);
}
}
}
}
None
}
fn revision_matches(revision_name: &str, revision: Option<&str>) -> bool {
revision.is_none_or(|revision| {
revision.starts_with(revision_name) || revision_name.starts_with(revision)
})
}
fn find_manifest_package_dir(root: &Path, package: &str, version: &str) -> Option<PathBuf> {
if manifest_matches_package(&root.join("Cargo.toml"), package, version) {
return Some(root.to_path_buf());
}
let Ok(entries) = fs::read_dir(root) else {
return None;
};
for entry in entries.flatten() {
let candidate = entry.path();
if candidate.is_dir()
&& manifest_matches_package(&candidate.join("Cargo.toml"), package, version)
{
return Some(candidate);
}
}
None
}
fn manifest_matches_package(manifest_path: &Path, package: &str, version: &str) -> bool {
let Ok(source) = fs::read_to_string(manifest_path) else {
return false;
};
let mut in_package = false;
let mut name_matches = false;
let mut version_matches = false;
for line in source.lines() {
let trimmed = line.trim();
if trimmed.starts_with('[') {
in_package = trimmed == "[package]";
continue;
}
if !in_package {
continue;
}
if let Some(value) = toml_string_value(trimmed, "name") {
name_matches = value == package;
continue;
}
if let Some(value) = toml_string_value(trimmed, "version") {
version_matches = value == version;
}
}
name_matches && version_matches
}
fn git_revision(source: &str) -> Option<&str> {
source.rsplit_once('#').and_then(|(_, revision)| {
if revision.is_empty() {
None
} else {
Some(revision)
}
})
}
fn locked_packages(lock_path: &Path) -> HashMap<String, LockedPackage> {
let source = fs::read_to_string(lock_path).expect("Cargo.lock should be readable");
let mut packages = HashMap::new();
let mut current_name = None;
let mut current_version = None;
let mut current_source = None;
for line in source.lines() {
let trimmed = line.trim();
if trimmed == "[[package]]" {
if let (Some(name), Some(version)) = (current_name.take(), current_version.take()) {
packages.insert(name, LockedPackage { version, source: current_source.take() });
}
current_source = None;
continue;
}
if let Some(value) = toml_string_value(trimmed, "name") {
current_name = Some(value.to_string());
continue;
}
if let Some(value) = toml_string_value(trimmed, "version") {
current_version = Some(value.to_string());
continue;
}
if let Some(value) = toml_string_value(trimmed, "source") {
current_source = Some(value.to_string());
}
}
if let (Some(name), Some(version)) = (current_name, current_version) {
packages.insert(name, LockedPackage { version, source: current_source });
}
packages
}
fn toml_string_value<'a>(line: &'a str, key: &str) -> Option<&'a str> {
let value = line
.strip_prefix(key)?
.trim_start()
.strip_prefix('=')?
.trim_start()
.strip_prefix('"')?;
let end = value.find('"')?;
Some(&value[..end])
}
-231
View File
@@ -1,231 +0,0 @@
//! Language-specific chunk classifiers for Astro.
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
common::*,
kind::ChunkKind,
};
use crate::language::SupportLang;
pub struct AstroClassifier;
impl LangClassifier for AstroClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: StructuralOverrides {
extra_trivia: &[],
preserved_trivia: &[],
extra_root_wrappers: &["document"],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
&TABLES
}
fn classify_override<'t>(
&self,
_context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
classify_astro_node(node, source)
}
}
fn classify_astro_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"frontmatter" => Some(classify_frontmatter(node, source)),
"frontmatter_js_block" => Some(group_candidate(node, ChunkKind::Code, source)),
"element" => classify_element(node, source),
"script_element" => Some(classify_script_element(node, source)),
"style_element" => Some(classify_style_element(node, source)),
"html_interpolation" => Some(classify_html_interpolation(node, source)),
"attribute_interpolation" => Some(classify_attribute_interpolation(node, source)),
"attribute_js_expr" => Some(group_candidate(node, ChunkKind::Expression, source)),
"text" => Some(group_candidate(node, ChunkKind::Text, source)),
_ => None,
}
}
fn classify_frontmatter<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let Some(content_node) = child_by_kind(node, &["frontmatter_js_block"]) else {
return make_kind_chunk(node, ChunkKind::Frontmatter, None, source, None);
};
let candidate = with_region_node(
make_kind_chunk(node, ChunkKind::Frontmatter, None, source, None),
Some(content_node),
);
with_injected_subtree(candidate, SupportLang::TypeScript, content_node)
}
fn classify_element<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let tag_name = extract_tag_name(node, source)?;
let recurse = Some(recurse_self(node, ChunkContext::ClassBody));
if is_component_name(tag_name.as_str()) {
Some(force_container(make_explicit_candidate(
node,
ChunkKind::Tag,
format!("component_{tag_name}"),
source,
recurse,
)))
} else {
Some(force_container(make_container_chunk(
node,
ChunkKind::Tag,
Some(tag_name),
source,
recurse,
)))
}
}
fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let identifier = has_attribute(node, "is:inline", source).then_some("inline".to_string());
classify_raw_text_block(node, ChunkKind::Script, identifier, source, SupportLang::TypeScript)
}
fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let identifier = if has_attribute(node, "define:vars", source) {
Some("vars".to_string())
} else if has_attribute(node, "is:global", source) {
Some("global".to_string())
} else {
None
};
classify_raw_text_block(node, ChunkKind::Style, identifier, source, SupportLang::Css)
}
fn classify_html_interpolation<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let identifier = child_by_kind(node, &["permissible_text"])
.and_then(|expr| sanitize_identifier(node_text(source, expr.start_byte(), expr.end_byte())));
if let Some(nested_element) =
child_by_kind(node, &["element", "script_element", "style_element"])
{
force_container(make_container_chunk(
node,
ChunkKind::Expression,
identifier,
source,
Some(recurse_self(nested_element, ChunkContext::ClassBody)),
))
} else {
make_kind_chunk(node, ChunkKind::Expression, identifier, source, None)
}
}
fn classify_attribute_interpolation<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let identifier = child_by_kind(node, &["attribute_js_expr"])
.and_then(|expr| sanitize_identifier(node_text(source, expr.start_byte(), expr.end_byte())))
.map_or_else(|| "expr".to_string(), |expr| format!("expr_{expr}"));
make_kind_chunk(node, ChunkKind::Attr, Some(identifier), source, None)
}
fn make_explicit_candidate<'t>(
node: Node<'t>,
kind: ChunkKind,
identifier: impl Into<Option<String>>,
source: &str,
recurse: Option<RecurseSpec<'t>>,
) -> RawChunkCandidate<'t> {
make_candidate(
node,
kind,
identifier,
NameStyle::Named,
signature_for_node(node, source),
recurse,
source,
)
}
const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> {
candidate.force_recurse = true;
candidate
}
fn classify_raw_text_block<'t>(
node: Node<'t>,
kind: ChunkKind,
identifier: Option<String>,
source: &str,
default_language: SupportLang,
) -> RawChunkCandidate<'t> {
let Some(content_node) = child_by_kind(node, &["raw_text"]) else {
return make_kind_chunk(node, kind, identifier, source, None);
};
let candidate =
with_region_node(make_kind_chunk(node, kind, identifier, source, None), Some(content_node));
match resolve_embedded_language(node, source, default_language) {
Some(language) => with_injected_subtree(candidate, language, content_node),
None => candidate,
}
}
fn extract_tag_name(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["start_tag", "self_closing_tag"])
.and_then(|tag| child_by_kind(tag, &["tag_name"]))
.and_then(|tag_name| {
sanitize_identifier(node_text(source, tag_name.start_byte(), tag_name.end_byte()))
})
}
fn has_attribute(node: Node<'_>, name: &str, source: &str) -> bool {
child_by_kind(node, &["start_tag", "self_closing_tag"])
.into_iter()
.flat_map(named_children)
.filter(|child| child.kind() == "attribute")
.filter_map(|attr| extract_attribute_name(attr, source))
.any(|attr_name| attr_name == name)
}
fn extract_attribute_name(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["attribute_name"]).map(|name| {
node_text(source, name.start_byte(), name.end_byte())
.trim()
.to_string()
})
}
fn resolve_embedded_language(
node: Node<'_>,
source: &str,
default_language: SupportLang,
) -> Option<SupportLang> {
if let Some(language) = attribute_value(node, "lang", source) {
return SupportLang::from_alias(language.as_str());
}
Some(default_language)
}
fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option<String> {
let start = child_by_kind(node, &["start_tag", "self_closing_tag"])?;
for child in named_children(start) {
if child.kind() != "attribute" {
continue;
}
if extract_attribute_name(child, source).as_deref() != Some(name) {
continue;
}
if let Some(value) = child_by_kind(child, &["attribute_value", "quoted_attribute_value"]) {
return sanitize_identifier(&unquote_text(node_text(
source,
value.start_byte(),
value.end_byte(),
)));
}
return Some(name.to_string());
}
None
}
fn is_component_name(tag_name: &str) -> bool {
tag_name.chars().next().is_some_and(char::is_uppercase)
}
@@ -1,275 +0,0 @@
//! Language-specific chunk classifiers for Bash, Make, and Diff.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct ShellBuildClassifier;
impl ShellBuildClassifier {
/// Extract a Make rule target name (child node of kind `targets`).
fn extract_rule_target(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["targets"])
.and_then(|t| sanitize_identifier(node_text(source, t.start_byte(), t.end_byte())))
}
/// Extract a Make variable/define name (field `name`).
fn extract_var_name(node: Node<'_>, source: &str) -> Option<String> {
node
.child_by_field_name("name")
.and_then(|n| sanitize_identifier(node_text(source, n.start_byte(), n.end_byte())))
}
/// Strip the conventional `a/` or `b/` prefix from git diff paths.
fn strip_ab_prefix(path: &str) -> &str {
path
.strip_prefix("a/")
.or_else(|| path.strip_prefix("b/"))
.unwrap_or(path)
}
/// Extract the file path from a diff `block` node.
///
/// Extraction priority:
/// 1. `new_file` child -> `filename` child text (skip if `/dev/null`)
/// 2. `old_file` child -> `filename` child text (skip if `/dev/null`)
/// 3. `command` child -> parse `a/path b/path` from filename children
fn extract_diff_filename(node: Node<'_>, source: &str) -> Option<String> {
// Try new_file first (most diffs have it)
if let Some(new_file) = child_by_kind(node, &["new_file"])
&& let Some(filename) = child_by_kind(new_file, &["filename"])
{
let text = node_text(source, filename.start_byte(), filename.end_byte()).trim();
if text != "/dev/null" {
return sanitize_identifier(Self::strip_ab_prefix(text));
}
}
// Fall back to old_file (deleted files)
if let Some(old_file) = child_by_kind(node, &["old_file"])
&& let Some(filename) = child_by_kind(old_file, &["filename"])
{
let text = node_text(source, filename.start_byte(), filename.end_byte()).trim();
if text != "/dev/null" {
return sanitize_identifier(Self::strip_ab_prefix(text));
}
}
// Last resort: extract from the `command` line ("diff --git a/path b/path").
// The grammar's `filename` rule is `repeat1(/\S+/)`, so it captures both
// paths as a single node like "a/foo.ts b/foo.ts". Take the last
// space-delimited segment (the b-side path).
if let Some(command) = child_by_kind(node, &["command"])
&& let Some(filename) = child_by_kind(command, &["filename"])
{
let text = node_text(source, filename.start_byte(), filename.end_byte()).trim();
let b_side = text.rsplit_once(' ').map_or(text, |(_, b)| b);
return sanitize_identifier(Self::strip_ab_prefix(b_side));
}
None
}
}
impl LangClassifier for ShellBuildClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[
semantic_rule(
"conditional",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"command",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"pipeline",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"if_statement",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"case_statement",
ChunkKind::Switch,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"while_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"for_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"function_definition",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"hunks",
ChunkKind::Hunks,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
],
class: &[semantic_rule(
"hunk",
ChunkKind::Hunk,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
)],
function: &[
semantic_rule(
"if_statement",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"case_statement",
ChunkKind::Switch,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"while_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"for_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"command",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"pipeline",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"subshell",
ChunkKind::Block,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
],
structural_overrides: StructuralOverrides {
extra_trivia: &[],
preserved_trivia: &[],
extra_root_wrappers: &["makefile"],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_root_custom(node, source),
_ => None,
}
}
fn preserve_children(
&self,
_parent: &RawChunkCandidate<'_>,
children: &[RawChunkCandidate<'_>],
) -> bool {
// Diff file blocks should always preserve hunk children
children.iter().any(|c| c.kind == ChunkKind::Hunk)
}
}
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"rule" => {
let name = ShellBuildClassifier::extract_rule_target(node, source)
.unwrap_or_else(|| "anonymous".to_string());
Some(make_container_chunk(
node,
ChunkKind::Rule,
Some(name),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["recipe"]),
))
},
"variable_assignment" | "shell_assignment" => {
let name = ShellBuildClassifier::extract_var_name(node, source)
.unwrap_or_else(|| "anonymous".to_string());
Some(make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None))
},
"define_directive" => {
let name = ShellBuildClassifier::extract_var_name(node, source)
.unwrap_or_else(|| "anonymous".to_string());
Some(make_kind_chunk(node, ChunkKind::Define, Some(name), source, None))
},
"block" => {
let identifier = ShellBuildClassifier::extract_diff_filename(node, source);
let recurse = recurse_into(node, ChunkContext::ClassBody, &[], &["hunks"]);
let mut candidate =
make_container_chunk(node, ChunkKind::File, identifier, source, recurse);
// Always expand hunks so individual @@ sections are addressable,
// even for small diffs below the leaf threshold.
candidate.force_recurse = recurse.is_some();
Some(candidate)
},
_ => None,
}
}
@@ -1,431 +0,0 @@
//! Language-specific chunk classifiers for C, C++, and Objective-C.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
defaults::classify_var_decl,
kind::ChunkKind,
};
pub struct CCppClassifier;
/// Extract the function name from a C/C++ `function_definition` or
/// `function_declaration` node by traversing into the `declarator` chain.
fn extract_c_function_name(node: Node<'_>, source: &str) -> Option<String> {
let decl = node.child_by_field_name("declarator")?;
extract_c_declarator_name(decl, source)
}
/// Recursively resolve a C/C++ declarator to its leaf identifier.
/// Handles `function_declarator`, `pointer_declarator`, `reference_declarator`,
/// `qualified_identifier`, `destructor_name`, `template_function`, etc.
fn extract_c_declarator_name(node: Node<'_>, source: &str) -> Option<String> {
match node.kind() {
"identifier" | "field_identifier" | "type_identifier" => {
sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()))
},
"destructor_name" => {
// ~ClassName
sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()))
},
"qualified_identifier" | "scoped_identifier" => {
// e.g. Entity::update — extract the "name" field or last identifier
node
.child_by_field_name("name")
.and_then(|n| extract_c_declarator_name(n, source))
.or_else(|| {
named_children(node)
.into_iter()
.rev()
.find(|c| {
matches!(
c.kind(),
"identifier" | "destructor_name" | "template_function" | "field_identifier"
)
})
.and_then(|c| extract_c_declarator_name(c, source))
})
},
"template_function" => {
// template_function has a "name" field or direct identifier child
node
.child_by_field_name("name")
.and_then(|n| sanitize_identifier(node_text(source, n.start_byte(), n.end_byte())))
.or_else(|| {
named_children(node)
.into_iter()
.find(|c| c.kind() == "identifier")
.and_then(|c| {
sanitize_identifier(node_text(source, c.start_byte(), c.end_byte()))
})
})
},
_ => {
// function_declarator, pointer_declarator, reference_declarator, etc.
// recurse into the "declarator" field
node
.child_by_field_name("declarator")
.and_then(|inner| extract_c_declarator_name(inner, source))
.or_else(|| {
// fallback: look for direct identifier-like child
named_children(node)
.into_iter()
.find(|c| {
matches!(
c.kind(),
"identifier"
| "field_identifier"
| "qualified_identifier"
| "scoped_identifier"
| "destructor_name"
| "template_function"
)
})
.and_then(|c| extract_c_declarator_name(c, source))
})
},
}
}
/// Extract the field name from a C/C++ `field_declaration` node.
/// The name sits in the `declarator` field which may be a plain
/// `field_identifier`, or a `function_declarator` / `pointer_declarator` etc.
fn extract_c_field_name(node: Node<'_>, source: &str) -> Option<String> {
let decl = node.child_by_field_name("declarator")?;
extract_c_declarator_name(decl, source)
}
const C_CPP_ROOT_RULES: &[super::classify::SemanticRule] = &[
// ── Imports ──
semantic_rule(
"include_directive",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"preproc_include",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"using_directive",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"using_statement",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"import_declaration",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"module_import",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
// ── Statements ──
semantic_rule(
"expression_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const C_CPP_TABLES: ClassifierTables = ClassifierTables {
root: C_CPP_ROOT_RULES,
class: &[],
function: &[],
structural_overrides: StructuralOverrides::EMPTY,
};
impl LangClassifier for CCppClassifier {
fn tables(&self) -> &'static ClassifierTables {
&C_CPP_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_root_custom(self, node, source),
ChunkContext::ClassBody => classify_class_custom(self, node, source),
ChunkContext::FunctionBody => Some(classify_function_c(node, source)),
}
}
}
fn classify_root_custom<'t>(
classifier: &CCppClassifier,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Functions ──
"function_definition" | "function_declaration" => Some(make_kind_chunk(
node,
ChunkKind::Function,
extract_c_function_name(node, source),
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
"constructor_definition" => Some(make_kind_chunk(
node,
ChunkKind::Constructor,
None,
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Templates (unwrap to find the inner declaration) ──
"template_declaration" => {
// Find the inner function_definition / class_specifier / etc.
let inner = named_children(node).into_iter().find(|c| {
matches!(
c.kind(),
"function_definition"
| "function_declaration"
| "class_specifier"
| "struct_specifier"
| "type_alias_declaration"
)
});
match inner {
Some(inner) => {
let mut candidate =
classifier.classify_override(ChunkContext::Root, inner, source)?;
// Expand range to include the template<...> prefix
candidate.range_start_byte = node.start_byte();
candidate.range_start_line = node.start_position().row + 1;
candidate.checksum_start_byte = node.start_byte();
Some(candidate)
},
None => Some(make_candidate(
node,
ChunkKind::Template,
None,
NameStyle::Named,
signature_for_node(node, source),
recurse_body(node, ChunkContext::FunctionBody),
source,
)),
}
},
// ── Containers ──
"class_specifier" | "class_declaration" | "class_interface" | "class_implementation" => {
Some(container_candidate(node, ChunkKind::Class, source, recurse_class(node)))
},
"struct_specifier" | "struct_declaration" => {
Some(container_candidate(node, ChunkKind::Struct, source, recurse_class(node)))
},
"enum_specifier" | "enum_declaration" => {
Some(container_candidate(node, ChunkKind::Enum, source, recurse_enum(node)))
},
"namespace_definition" => {
Some(container_candidate(node, ChunkKind::Module, source, recurse_class(node)))
},
"union_declaration" => {
Some(container_candidate(node, ChunkKind::Union, source, recurse_class(node)))
},
// ── Types ──
"type_alias_declaration" | "user_defined_type_definition" => {
Some(named_candidate(node, ChunkKind::Type, source, recurse_class(node)))
},
// ── Variables / assignments ──
"variable_declaration" => Some(classify_var_decl(node, source)),
"assignment_statement" | "property_declaration" => {
Some(group_candidate(node, ChunkKind::Declarations, source))
},
// ── Macros ──
"macro_definition" => Some(named_candidate(
node,
ChunkKind::Macro,
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Control flow (top-level scripts) ──
"if_statement" | "switch_statement" | "for_statement" | "while_statement"
| "do_statement" | "try_block" => Some(classify_function_c(node, source)),
_ => None,
}
}
fn classify_class_custom<'t>(
classifier: &CCppClassifier,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Methods ──
"function_definition" | "function_declaration" | "method_declaration" => {
let name = extract_c_function_name(node, source)
.or_else(|| extract_identifier(node, source))
.unwrap_or_else(|| "anonymous".to_string());
if name == "constructor" {
Some(make_kind_chunk(
node,
ChunkKind::Constructor,
None,
source,
recurse_body(node, ChunkContext::FunctionBody),
))
} else {
Some(make_kind_chunk(
node,
ChunkKind::Function,
Some(name),
source,
recurse_body(node, ChunkContext::FunctionBody),
))
}
},
// ── Constructors ──
"constructor_definition" | "constructor_declaration" => Some(make_kind_chunk(
node,
ChunkKind::Constructor,
None,
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Fields ──
"field_declaration" => Some(match extract_c_field_name(node, source) {
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
None => group_candidate(node, ChunkKind::Fields, source),
}),
// ── Enum variants ──
"enum_constant" => Some(match extract_identifier(node, source) {
Some(name) => make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None),
None => group_candidate(node, ChunkKind::Variants, source),
}),
// ── Nested containers ──
"class_specifier" | "class_declaration" | "class_interface" | "class_implementation" => {
Some(container_candidate(node, ChunkKind::Class, source, recurse_class(node)))
},
"struct_specifier" | "struct_declaration" => {
Some(container_candidate(node, ChunkKind::Struct, source, recurse_class(node)))
},
"enum_specifier" | "enum_declaration" => {
Some(container_candidate(node, ChunkKind::Enum, source, recurse_enum(node)))
},
"union_declaration" => {
Some(container_candidate(node, ChunkKind::Union, source, recurse_class(node)))
},
"namespace_definition" => {
Some(container_candidate(node, ChunkKind::Module, source, recurse_class(node)))
},
// ── Templates (class body) ──
"template_declaration" => {
let inner = named_children(node).into_iter().find(|c| {
matches!(
c.kind(),
"function_definition"
| "function_declaration"
| "class_specifier"
| "struct_specifier"
| "type_alias_declaration"
)
});
match inner {
Some(inner) => {
let mut candidate =
classifier.classify_override(ChunkContext::ClassBody, inner, source)?;
candidate.range_start_byte = node.start_byte();
candidate.range_start_line = node.start_position().row + 1;
candidate.checksum_start_byte = node.start_byte();
Some(candidate)
},
None => Some(make_candidate(
node,
ChunkKind::Template,
None,
NameStyle::Named,
signature_for_node(node, source),
recurse_body(node, ChunkContext::FunctionBody),
source,
)),
}
},
// ── Types ──
"type_alias_declaration" => Some(named_candidate(node, ChunkKind::Type, source, None)),
_ => None,
}
}
fn classify_function_c<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
match node.kind() {
"if_statement" => {
make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source)
},
"switch_statement" => {
make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source)
},
"try_block" | "catch_clause" | "finally_clause" => {
make_candidate(node, ChunkKind::Try, None, NameStyle::Named, None, fn_recurse(), source)
},
"for_statement" => {
make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source)
},
"while_statement" => {
make_candidate(node, ChunkKind::While, None, NameStyle::Named, None, fn_recurse(), source)
},
"do_statement" => {
make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source)
},
"variable_declaration" => {
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
if span > 1 {
if let Some(name) = extract_single_declarator_name(node, source) {
make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)
} else {
group_candidate(node, ChunkKind::Variable, source)
}
} else {
group_candidate(node, ChunkKind::Variable, source)
}
},
_ => {
let kind_name = sanitize_node_kind(node.kind());
let kind = ChunkKind::from_sanitized_kind(kind_name);
group_candidate(node, kind, source)
},
}
}
@@ -1,78 +0,0 @@
//! Language-specific chunk classifier for Clojure.
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier},
common::*,
kind::ChunkKind,
};
pub struct ClojureClassifier;
/// Extract the head symbol of a Clojure list form (first
/// `sym_lit`/`kwd_lit`/`symbol`/`word`).
fn form_head(node: Node<'_>, source: &str) -> Option<String> {
named_children(node).into_iter().find_map(|child| {
matches!(child.kind(), "sym_lit" | "kwd_lit" | "symbol" | "word")
.then(|| node_text(source, child.start_byte(), child.end_byte()).to_string())
})
}
/// Extract the name from a Clojure form: the second named child after the head
/// symbol (e.g. `greet` in `(defn greet [x] x)`).
fn form_name(node: Node<'_>, source: &str) -> Option<String> {
let mut children = named_children(node).into_iter();
let _head = children.next()?;
children.find_map(|child| {
sanitize_identifier(node_text(source, child.start_byte(), child.end_byte()))
})
}
/// Classify a `list_lit` Clojure form based on its head symbol.
fn classify_form<'t>(node: Node<'t>, source: &str, at_root: bool) -> RawChunkCandidate<'t> {
let Some(head) = form_head(node, source) else {
return positional_candidate(node, ChunkKind::Form, source);
};
match head.as_str() {
"ns" | "require" | "use" | "import" | "refer-clojure" => {
group_candidate(node, ChunkKind::Imports, source)
},
"defn" | "defn-" | "defmacro" | "defmulti" | "defmethod" => {
make_kind_chunk(node, ChunkKind::Function, form_name(node, source), source, None)
},
"def" | "defonce" => {
make_kind_chunk(node, ChunkKind::Decl, form_name(node, source), source, None)
},
"defprotocol" => {
make_container_chunk(node, ChunkKind::Proto, form_name(node, source), source, None)
},
"deftype" | "defrecord" | "extend-type" | "extend-protocol" => {
make_container_chunk(node, ChunkKind::Type, form_name(node, source), source, None)
},
_ if at_root => positional_candidate(node, ChunkKind::Form, source),
_ => group_candidate(node, ChunkKind::Block, source),
}
}
impl LangClassifier for ClojureClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: super::classify::StructuralOverrides::EMPTY,
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
(node.kind() == "list_lit")
.then(|| classify_form(node, source, matches!(context, ChunkContext::Root)))
}
}
-174
View File
@@ -1,174 +0,0 @@
//! CMake-specific chunk classifier.
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
common::*,
kind::ChunkKind,
};
pub struct CMakeClassifier;
fn child_text<'a>(source: &'a str, node: Node<'_>) -> &'a str {
node_text(source, node.start_byte(), node.end_byte())
}
fn first_named_child(node: Node<'_>) -> Option<Node<'_>> {
named_children(node).into_iter().next()
}
fn first_named_child_of_kind<'t>(node: Node<'t>, kind: &str) -> Option<Node<'t>> {
named_children(node)
.into_iter()
.find(|child| child.kind() == kind)
}
fn command_name(node: Node<'_>, source: &str) -> Option<String> {
first_named_child(node).and_then(|child| sanitize_identifier(child_text(source, child)))
}
fn argument_nodes(node: Node<'_>) -> Vec<Node<'_>> {
first_named_child_of_kind(node, "argument_list")
.map(named_children)
.unwrap_or_default()
.into_iter()
.filter(|child| child.kind() == "argument")
.collect()
}
fn nth_argument_name(node: Node<'_>, index: usize, source: &str) -> Option<String> {
argument_nodes(node)
.into_iter()
.nth(index)
.and_then(|arg| sanitize_identifier(child_text(source, arg)))
}
fn classify_definition<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"function_def" => {
let header = first_named_child_of_kind(node, "function_command")?;
let name = nth_argument_name(header, 0, source).unwrap_or_else(|| "anonymous".to_string());
Some(make_container_chunk(
node,
ChunkKind::Function,
Some(name),
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["body"]),
))
},
"macro_def" => {
let header = first_named_child_of_kind(node, "macro_command")?;
let name = nth_argument_name(header, 0, source).unwrap_or_else(|| "anonymous".to_string());
Some(make_container_chunk(
node,
ChunkKind::Macro,
Some(name),
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["body"]),
))
},
"if_condition" => Some(make_container_chunk(
node,
ChunkKind::If,
None,
source,
Some(recurse_self(node, ChunkContext::FunctionBody)),
)),
"foreach_loop" | "while_loop" => Some(make_container_chunk(
node,
ChunkKind::Loop,
None,
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["body"]),
)),
_ => None,
}
}
fn classify_command<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
if node.kind() != "normal_command" {
return None;
}
let command = command_name(node, source)?;
Some(match command.as_str() {
"cmake_minimum_required" => make_kind_chunk(node, ChunkKind::VersionGate, None, source, None),
"project" => {
let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string());
make_kind_chunk(node, ChunkKind::Project, Some(name), source, None)
},
"include" | "find_package" => group_candidate(node, ChunkKind::Imports, source),
"option" => {
let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string());
make_kind_chunk(node, ChunkKind::Option, Some(name), source, None)
},
"set" => {
let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string());
make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)
},
"add_library" | "add_executable" | "add_custom_target" => {
let name = nth_argument_name(node, 0, source).unwrap_or_else(|| "anonymous".to_string());
make_kind_chunk(node, ChunkKind::Target, Some(name), source, None)
},
"install" | "export" => group_candidate(node, ChunkKind::Install, source),
other => make_kind_chunk(node, ChunkKind::Cmd, Some(other.to_string()), source, None),
})
}
fn classify_if_child<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"if_command" => Some(group_candidate(node, ChunkKind::Cond, source)),
"elseif_command" => Some(positional_candidate(node, ChunkKind::Elif, source)),
"else_command" => Some(positional_candidate(node, ChunkKind::Else, source)),
"body" => Some(make_container_chunk(
node,
ChunkKind::Block,
None,
source,
Some(recurse_self(node, ChunkContext::FunctionBody)),
)),
_ => None,
}
}
impl LangClassifier for CMakeClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: StructuralOverrides {
extra_trivia: &[
"endif_command",
"endforeach_command",
"endwhile_command",
"endfunction_command",
"endmacro_command",
],
preserved_trivia: &[],
extra_root_wrappers: &[],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => {
classify_definition(node, source).or_else(|| classify_command(node, source))
},
ChunkContext::FunctionBody => classify_definition(node, source)
.or_else(|| classify_if_child(node, source))
.or_else(|| classify_command(node, source)),
ChunkContext::ClassBody => None,
}
}
}
@@ -1,372 +0,0 @@
//! Language-specific chunk classifiers for C# and Java.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
defaults::classify_var_decl,
kind::ChunkKind,
};
pub struct CSharpJavaClassifier;
const CSHARP_JAVA_ROOT_RULES: &[super::classify::SemanticRule] = &[
// ── Imports ──
semantic_rule(
"import_declaration",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"using_directive",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"package_declaration",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"namespace_statement",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
// ── Functions ──
semantic_rule(
"method_declaration",
ChunkKind::Method,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"function_declaration",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"function_definition",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
// ── Constructors ──
semantic_rule(
"constructor_declaration",
ChunkKind::Constructor,
RuleStyle::Named,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
// ── Containers ──
semantic_rule(
"class_declaration",
ChunkKind::Class,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"interface_declaration",
ChunkKind::Iface,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"enum_declaration",
ChunkKind::Enum,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"struct_declaration",
ChunkKind::Struct,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"record_declaration",
ChunkKind::Struct,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"namespace_declaration",
ChunkKind::Module,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"file_scoped_namespace_declaration",
ChunkKind::Module,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
// ── Types ──
semantic_rule(
"type_alias_declaration",
ChunkKind::Type,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
// ── Declarations ──
semantic_rule(
"property_declaration",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"state_variable_declaration",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
// ── Statements ──
semantic_rule(
"expression_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const CSHARP_JAVA_CLASS_RULES: &[super::classify::SemanticRule] = &[
// ── Containers ──
semantic_rule(
"class_declaration",
ChunkKind::Class,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"interface_declaration",
ChunkKind::Iface,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"enum_declaration",
ChunkKind::Enum,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"struct_declaration",
ChunkKind::Struct,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"record_declaration",
ChunkKind::Struct,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"namespace_declaration",
ChunkKind::Module,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"file_scoped_namespace_declaration",
ChunkKind::Module,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
// ── Static blocks ──
semantic_rule(
"class_static_block",
ChunkKind::StaticInit,
RuleStyle::Named,
NamingMode::None,
RecurseMode::None,
),
];
const CSHARP_JAVA_TABLES: ClassifierTables = ClassifierTables {
root: CSHARP_JAVA_ROOT_RULES,
class: CSHARP_JAVA_CLASS_RULES,
function: &[],
structural_overrides: StructuralOverrides::EMPTY,
};
impl LangClassifier for CSharpJavaClassifier {
fn tables(&self) -> &'static ClassifierTables {
&CSHARP_JAVA_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => match node.kind() {
// ── Variables / assignments ──
"variable_declaration" | "lexical_declaration" => Some(classify_var_decl(node, source)),
// ── Control flow (top-level scripts) ──
"if_statement" | "switch_statement" | "switch_expression" | "for_statement"
| "foreach_statement" | "while_statement" | "do_statement" | "try_statement" => {
Some(classify_function_csharp_java(node, source))
},
_ => None,
},
ChunkContext::ClassBody => match node.kind() {
// ── Methods (conditional constructor detection) ──
"method_declaration" | "function_declaration" | "function_definition" => {
let name =
extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
if name == "constructor" {
Some(make_kind_chunk(
node,
ChunkKind::Constructor,
None,
source,
recurse_body(node, ChunkContext::FunctionBody),
))
} else {
Some(make_kind_chunk(
node,
ChunkKind::Function,
Some(name),
source,
recurse_body(node, ChunkContext::FunctionBody),
))
}
},
// ── Constructors ──
"constructor_declaration" | "secondary_constructor" => Some(make_kind_chunk(
node,
ChunkKind::Constructor,
None,
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Fields ──
"field_declaration"
| "property_declaration"
| "constant_declaration"
| "event_field_declaration" => Some(match extract_field_name(node, source) {
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
None => group_candidate(node, ChunkKind::Fields, source),
}),
// ── Enum members ──
"enum_member_declaration" | "enum_constant" | "enum_entry" => {
Some(match extract_identifier(node, source) {
Some(name) => make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None),
None => group_candidate(node, ChunkKind::Variants, source),
})
},
_ => None,
},
ChunkContext::FunctionBody => Some(classify_function_csharp_java(node, source)),
}
}
}
/// Extract the variable name from a field/constant declaration.
///
/// Java `field_declaration` has the structure:
/// `field_declaration` { modifiers, type: `type_identifier`, declarator:
/// `variable_declarator` { name: identifier } }
///
/// `extract_identifier` would find `type_identifier` first, so we look into
/// `variable_declarator` children for the actual variable name.
fn extract_field_name(node: Node<'_>, source: &str) -> Option<String> {
for child in named_children(node) {
if child.kind() == "variable_declarator" {
return extract_identifier(child, source);
}
}
extract_identifier(node, source)
}
fn classify_function_csharp_java<'tree>(
node: Node<'tree>,
source: &str,
) -> RawChunkCandidate<'tree> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
match node.kind() {
"if_statement" => {
make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source)
},
"switch_statement" | "switch_expression" => {
make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source)
},
"try_statement" | "catch_clause" | "finally_clause" => {
make_candidate(node, ChunkKind::Try, None, NameStyle::Named, None, fn_recurse(), source)
},
"for_statement" => {
make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source)
},
"foreach_statement" => {
make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source)
},
"while_statement" => {
make_candidate(node, ChunkKind::While, None, NameStyle::Named, None, fn_recurse(), source)
},
"do_statement" => {
make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source)
},
"variable_declaration" | "lexical_declaration" => {
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
if span > 1 {
if let Some(name) = extract_single_declarator_name(node, source) {
make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)
} else {
group_from_sanitized(node, source)
}
} else {
group_from_sanitized(node, source)
}
},
_ => group_from_sanitized(node, source),
}
}
fn group_from_sanitized<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let sanitized = sanitize_node_kind(node.kind());
let kind = ChunkKind::from_sanitized_kind(sanitized);
let identifier = if kind == ChunkKind::Chunk {
Some(sanitized.to_string())
} else {
None
};
make_candidate(node, kind, identifier, NameStyle::Group, None, None, source)
}
-124
View File
@@ -1,124 +0,0 @@
//! Language-specific chunk classifiers for CSS and SCSS.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct CssClassifier;
const CSS_SHARED_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"keyframe_block",
ChunkKind::Frame,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"declaration",
ChunkKind::Fields,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const CSS_TABLES: ClassifierTables = ClassifierTables {
root: CSS_SHARED_RULES,
class: CSS_SHARED_RULES,
function: &[],
structural_overrides: StructuralOverrides {
extra_trivia: &[],
preserved_trivia: &[],
extra_root_wrappers: &["stylesheet"],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
/// Extract a CSS selector name from a `rule_set` or `at_rule` node.
///
/// Tries known child kinds first (`selectors`, `selector_query`, `identifier`),
/// then falls back to parsing the normalised header text.
fn extract_css_selector(node: Node<'_>, source: &str) -> Option<String> {
if let Some(sel) = child_by_kind(node, &["selectors", "selector_query", "identifier"]) {
return sanitize_identifier(node_text(source, sel.start_byte(), sel.end_byte()));
}
let header = normalized_header(source, node.start_byte(), node.end_byte());
let selector = header
.trim_start_matches('@')
.split('{')
.next()
.unwrap_or(header.as_str())
.trim();
sanitize_identifier(selector)
}
/// Classify a CSS `rule_set` as a named container.
fn classify_rule_set<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let name = extract_css_selector(node, source).unwrap_or_else(|| "anonymous".to_string());
make_container_chunk(
node,
ChunkKind::Rule,
Some(name),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["block"]),
)
}
/// Classify a CSS at-rule (`@media`, `@keyframes`, `@supports`, etc.) as a
/// named container.
fn classify_at_rule<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let name = extract_css_selector(node, source).unwrap_or_else(|| "rule".to_string());
make_container_chunk(
node,
ChunkKind::At,
Some(name),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["block", "keyframe_block_list"]),
)
}
/// Shared dispatch for CSS node kinds, used in both root and class-body
/// contexts.
fn classify_css_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"rule_set" => Some(classify_rule_set(node, source)),
"at_rule" | "media_statement" | "keyframes_statement" | "supports_statement" => {
Some(classify_at_rule(node, source))
},
"keyframe_block" => Some(named_candidate(
node,
ChunkKind::Frame,
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
)),
"declaration" => Some(group_candidate(node, ChunkKind::Fields, source)),
_ => None,
}
}
impl LangClassifier for CssClassifier {
fn tables(&self) -> &'static ClassifierTables {
&CSS_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
if matches!(context, ChunkContext::Root | ChunkContext::ClassBody) {
return classify_css_node(node, source);
}
None
}
}
@@ -1,278 +0,0 @@
//! Chunk classifiers for data formats: JSON, TOML, YAML.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct DataFormatsClassifier;
const DATA_FORMAT_STRUCTURAL_OVERRIDES: StructuralOverrides = StructuralOverrides {
extra_trivia: &["bare_key", "quoted_key", "dotted_key"],
preserved_trivia: &[],
extra_root_wrappers: &[
"array",
"block_mapping",
"block_node",
"block_sequence",
"document",
"flow_mapping",
"flow_node",
"flow_sequence",
"object",
"stream",
],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
};
const DATA_FORMAT_ROOT_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"inline_table",
ChunkKind::Table,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"object",
ChunkKind::Object,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"array",
ChunkKind::Array,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"block_mapping",
ChunkKind::Map,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"flow_mapping",
ChunkKind::Map,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"block_sequence",
ChunkKind::List,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"flow_sequence",
ChunkKind::List,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"attribute",
ChunkKind::Attr,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::ValueContainer,
),
];
const DATA_FORMAT_CLASS_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"inline_table",
ChunkKind::Table,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"object",
ChunkKind::Object,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"array",
ChunkKind::Array,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"block_mapping",
ChunkKind::Map,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"flow_mapping",
ChunkKind::Map,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"block_sequence",
ChunkKind::List,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"flow_sequence",
ChunkKind::List,
RuleStyle::Named,
NamingMode::None,
RecurseMode::SelfNode(ChunkContext::ClassBody),
),
semantic_rule(
"block_sequence_item",
ChunkKind::Item,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"attribute",
ChunkKind::Attr,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::ValueContainer,
),
];
const DATA_FORMAT_TABLES: ClassifierTables = ClassifierTables {
root: DATA_FORMAT_ROOT_RULES,
class: DATA_FORMAT_CLASS_RULES,
function: &[],
structural_overrides: DATA_FORMAT_STRUCTURAL_OVERRIDES,
};
impl LangClassifier for DataFormatsClassifier {
fn tables(&self) -> &'static ClassifierTables {
&DATA_FORMAT_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_data_node(node, source, true),
ChunkContext::ClassBody => classify_data_node(node, source, false),
ChunkContext::FunctionBody => None,
}
}
fn preserve_children(
&self,
parent: &RawChunkCandidate<'_>,
_children: &[RawChunkCandidate<'_>],
) -> bool {
// YAML keys with container values should always expose sub-chunks
// so that deeply nested keys are individually addressable.
parent.force_recurse && parent.kind == ChunkKind::Key
}
}
fn classify_data_node<'t>(
node: Node<'t>,
source: &str,
is_root: bool,
) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// Key-value pairs (JSON pairs, YAML mappings)
"pair" => {
let name = extract_pair_key(node, source).unwrap_or_else(|| "anonymous".to_string());
Some(make_kind_chunk(
node,
ChunkKind::Key,
Some(name),
source,
recurse_value_container(node),
))
},
"block_mapping_pair" | "flow_pair" => {
let name = extract_yaml_key(node, source).unwrap_or_else(|| "anonymous".to_string());
let recurse = recurse_value_container(node);
let mut candidate = make_kind_chunk(node, ChunkKind::Key, Some(name), source, recurse);
// YAML structure is inherently hierarchical. Keys whose value is a
// container (mapping/sequence) should always produce sub-chunks so
// that deeply nested keys are individually addressable.
if candidate.recurse.is_some() {
candidate.force_recurse = true;
}
Some(candidate)
},
// TOML tables
"table" => {
let name = extract_toml_table_name(node, source);
Some(make_container_chunk(
node,
ChunkKind::Table,
name,
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
))
},
// TOML array tables
"table_array_element" => Some(make_candidate(
node,
ChunkKind::Table,
extract_toml_table_name(node, source).unwrap_or_else(|| "table_array".to_string()),
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::ClassBody)),
source,
)),
// YAML sequence items (only when nested, not at root level)
"block_sequence_item" if !is_root => {
Some(positional_candidate(node, ChunkKind::Item, source))
},
_ => None,
}
}
/// Extract key from a `pair` node (JSON or TOML).
/// JSON pairs have a `"key"` field; TOML pairs have no field names, so we fall
/// back to looking for the first `bare_key`, `quoted_key`, or `dotted_key`
/// child.
fn extract_pair_key(node: Node<'_>, source: &str) -> Option<String> {
let key = node
.child_by_field_name("key")
.or_else(|| child_by_kind(node, &["bare_key", "quoted_key", "dotted_key"]))?;
sanitize_identifier(unquote_text(node_text(source, key.start_byte(), key.end_byte())).as_str())
}
fn extract_toml_table_name(node: Node<'_>, source: &str) -> Option<String> {
let key = child_by_kind(node, &["dotted_key", "bare_key", "quoted_key"])?;
sanitize_identifier(node_text(source, key.start_byte(), key.end_byte()))
}
/// Extract key from a YAML `block_mapping_pair` or `flow_pair` node.
/// Descends into the key to find the first scalar child for complex keys.
fn extract_yaml_key(node: Node<'_>, source: &str) -> Option<String> {
let key = node.child_by_field_name("key")?;
let key_node = first_scalar_child(key).unwrap_or(key);
sanitize_identifier(
unquote_text(node_text(source, key_node.start_byte(), key_node.end_byte())).as_str(),
)
}
@@ -1,184 +0,0 @@
//! Chunk classifier for Dockerfile syntax.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct DockerfileClassifier;
const DOCKERFILE_ROOT_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"run_instruction",
ChunkKind::Cmd,
RuleStyle::Named,
NamingMode::SanitizedKind,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"cmd_instruction",
ChunkKind::Cmd,
RuleStyle::Named,
NamingMode::SanitizedKind,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"entrypoint_instruction",
ChunkKind::Cmd,
RuleStyle::Named,
NamingMode::SanitizedKind,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"copy_instruction",
ChunkKind::Copy,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"add_instruction",
ChunkKind::Add,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"workdir_instruction",
ChunkKind::Workdir,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"expose_instruction",
ChunkKind::Expose,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"user_instruction",
ChunkKind::User,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const DOCKERFILE_FUNCTION_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"cmd_instruction",
ChunkKind::Cmd,
RuleStyle::Named,
NamingMode::SanitizedKind,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"shell_command",
ChunkKind::Shell,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"json_string_array",
ChunkKind::Argv,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const DOCKERFILE_TABLES: ClassifierTables = ClassifierTables {
root: DOCKERFILE_ROOT_RULES,
class: &[],
function: DOCKERFILE_FUNCTION_RULES,
structural_overrides: super::classify::StructuralOverrides::EMPTY,
};
fn child_text<'a>(source: &'a str, node: Node<'_>) -> &'a str {
node_text(source, node.start_byte(), node.end_byte())
}
fn first_named_child(node: Node<'_>) -> Option<Node<'_>> {
named_children(node).into_iter().next()
}
fn first_named_child_of_kind<'t>(node: Node<'t>, kind: &str) -> Option<Node<'t>> {
named_children(node)
.into_iter()
.find(|child| child.kind() == kind)
}
fn extract_stage_name(node: Node<'_>, source: &str) -> Option<String> {
if let Some(alias) = child_by_kind(node, &["image_alias"]) {
return sanitize_identifier(child_text(source, alias));
}
child_by_kind(node, &["image_spec"]).and_then(|image| {
let image_name = child_by_kind(image, &["image_name"]).unwrap_or(image);
sanitize_identifier(child_text(source, image_name))
})
}
fn extract_pair_key(node: Node<'_>, pair_kind: &str, source: &str) -> Option<String> {
first_named_child_of_kind(node, pair_kind)
.and_then(first_named_child)
.and_then(|key| sanitize_identifier(unquote_text(child_text(source, key)).as_str()))
}
fn extract_arg_name(node: Node<'_>, source: &str) -> Option<String> {
first_named_child(node).and_then(|name| sanitize_identifier(child_text(source, name)))
}
impl LangClassifier for DockerfileClassifier {
fn tables(&self) -> &'static ClassifierTables {
&DOCKERFILE_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => match node.kind() {
"from_instruction" => {
let name =
extract_stage_name(node, source).unwrap_or_else(|| "anonymous".to_string());
Some(make_kind_chunk(node, ChunkKind::Stage, Some(name), source, None))
},
"arg_instruction" => {
let name = extract_arg_name(node, source).unwrap_or_else(|| "anonymous".to_string());
Some(make_kind_chunk(node, ChunkKind::Arg, Some(name), source, None))
},
"env_instruction" => {
let name = extract_pair_key(node, "env_pair", source)
.unwrap_or_else(|| "anonymous".to_string());
Some(make_kind_chunk(node, ChunkKind::Env, Some(name), source, None))
},
"label_instruction" => {
let name = extract_pair_key(node, "label_pair", source)
.unwrap_or_else(|| "anonymous".to_string());
Some(make_kind_chunk(node, ChunkKind::Label, Some(name), source, None))
},
"healthcheck_instruction" => Some(make_container_chunk(
node,
ChunkKind::Healthcheck,
None,
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["cmd_instruction"]),
)),
_ => None,
},
_ => None,
}
}
}
-159
View File
@@ -1,159 +0,0 @@
//! Language-specific chunk classifier for Elixir.
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
common::*,
kind::ChunkKind,
};
pub struct ElixirClassifier;
/// Extract the call target: `target` field, or first named child.
fn call_target(node: Node<'_>, source: &str) -> Option<String> {
node
.child_by_field_name("target")
.or_else(|| named_children(node).into_iter().next())
.map(|n| node_text(source, n.start_byte(), n.end_byte()).to_string())
}
/// Classify an Elixir `call` node based on its target keyword.
fn classify_call<'t>(node: Node<'t>, source: &str, at_root: bool) -> RawChunkCandidate<'t> {
let target = call_target(node, source).unwrap_or_default();
let name = || call_name(node, source).unwrap_or_else(|| "anonymous".to_string());
match target.as_str() {
"defmodule" => make_container_chunk(
node,
ChunkKind::Module,
Some(name()),
source,
recurse_body(node, ChunkContext::ClassBody),
),
"defprotocol" => make_container_chunk(
node,
ChunkKind::Proto,
Some(name()),
source,
recurse_body(node, ChunkContext::ClassBody),
),
"defimpl" => make_container_chunk(
node,
ChunkKind::Impl,
Some(name()),
source,
recurse_body(node, ChunkContext::ClassBody),
),
"def" | "defp" | "defdelegate" | "defguard" | "defguardp" | "defn" | "defnp" => {
make_kind_chunk(
node,
ChunkKind::Function,
Some(name()),
source,
recurse_body(node, ChunkContext::FunctionBody),
)
},
"defmacro" | "defmacrop" => make_kind_chunk(
node,
ChunkKind::Macro,
Some(name()),
source,
recurse_body(node, ChunkContext::FunctionBody),
),
"alias" | "import" | "require" | "use" => group_candidate(node, ChunkKind::Imports, source),
"defstruct" | "defexception" => group_candidate(node, ChunkKind::Declarations, source),
"if" | "unless" => positional_candidate(node, ChunkKind::If, source),
"case" | "cond" | "receive" => positional_candidate(node, ChunkKind::Switch, source),
"for" => positional_candidate(node, ChunkKind::For, source),
"try" | "with" => positional_candidate(node, ChunkKind::Block, source),
_ if at_root => group_candidate(node, ChunkKind::Statements, source),
_ => group_candidate(node, ChunkKind::Block, source),
}
}
/// Extract the name from an Elixir `call` node.
///
/// Skips keyword-only calls (imports, control flow) that have no meaningful
/// identifier, then returns the first non-`do_block` named child after the
/// target.
fn call_name(node: Node<'_>, source: &str) -> Option<String> {
let target = call_target(node, source)?;
if matches!(
target.as_str(),
"alias"
| "import"
| "require"
| "use"
| "if" | "case"
| "cond"
| "for"
| "try"
| "with"
| "unless"
| "receive"
) {
return None;
}
// The first named child after the target is typically `arguments`.
// For `def run(x)`, arguments contains a `call` node whose target is `run`.
// For `defmodule App`, arguments contains an `alias` node with text `App`.
// For `def run(x) when is_integer(x)`, arguments contains a `binary_operator`
// with the call on the left and the guard on the right.
// Extract the meaningful name, not the full text with parameters.
named_children(node).into_iter().skip(1).find_map(|child| {
if child.kind() == "do_block" {
return None;
}
if child.kind() == "arguments" {
// Dig into arguments to find the actual name.
return named_children(child).into_iter().next().and_then(|arg| {
if arg.kind() == "call" {
// `def run(x)` → arguments has call(target=run), extract target name
call_target(arg, source).and_then(|t| sanitize_identifier(&t))
} else if arg.kind() == "binary_operator" {
// `def run(x) when guard` → binary_operator(left=call, right=guard)
// Extract name from the left side (the actual function call).
arg.child_by_field_name("left").and_then(|left| {
if left.kind() == "call" {
call_target(left, source).and_then(|t| sanitize_identifier(&t))
} else {
sanitize_identifier(node_text(source, left.start_byte(), left.end_byte()))
}
})
} else {
// `defmodule App` → arguments has alias("App")
sanitize_identifier(node_text(source, arg.start_byte(), arg.end_byte()))
}
});
}
sanitize_identifier(node_text(source, child.start_byte(), child.end_byte()))
})
}
impl LangClassifier for ElixirClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: StructuralOverrides {
extra_trivia: &["unary_operator"],
preserved_trivia: &[],
extra_root_wrappers: &[],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
(node.kind() == "call").then(|| classify_call(node, source, context == ChunkContext::Root))
}
}
-242
View File
@@ -1,242 +0,0 @@
//! Language-specific chunk classifier for Erlang.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct ErlangClassifier;
fn find_named_descendant_by_kind<'t>(node: Node<'t>, kinds: &[&str]) -> Option<Node<'t>> {
if kinds.iter().any(|kind| node.kind() == *kind) {
return Some(node);
}
for child in named_children(node) {
if let Some(found) = find_named_descendant_by_kind(child, kinds) {
return Some(found);
}
}
None
}
fn named_text(node: Node<'_>, source: &str) -> Option<String> {
sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()))
}
fn erlang_name(node: Node<'_>, source: &str) -> Option<String> {
let name_node = match node.kind() {
"module_attribute" | "record_decl" | "record_field" => {
child_by_field_or_kind(node, &["name"], &["atom"])
},
"type_alias" => node
.child_by_field_name("name")
.and_then(|name| find_named_descendant_by_kind(name, &["atom"])),
"spec" => child_by_field_or_kind(node, &["fun"], &["atom"]),
"pp_define" => node
.child_by_field_name("lhs")
.and_then(|lhs| find_named_descendant_by_kind(lhs, &["var"])),
"fun_decl" => node
.child_by_field_name("clause")
.and_then(|clause| child_by_field_or_kind(clause, &["name"], &["atom"])),
"function_clause" => child_by_field_or_kind(node, &["name"], &["atom"]),
_ => child_by_kind(node, &["atom", "var"]),
}?;
named_text(name_node, source)
}
fn recurse_clause_body(node: Node<'_>) -> Option<RecurseSpec<'_>> {
recurse_into(node, ChunkContext::FunctionBody, &["body"], &["clause_body"])
}
impl LangClassifier for ErlangClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[
semantic_rule(
"export_attribute",
ChunkKind::Exports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"export_type_attribute",
ChunkKind::Exports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"import_attribute",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"pp_include",
ChunkKind::Includes,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"pp_include_lib",
ChunkKind::Includes,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
],
class: &[],
function: &[],
structural_overrides: StructuralOverrides {
extra_trivia: &[],
preserved_trivia: &[],
extra_root_wrappers: &[],
preserved_root_wrappers: &[],
absorbable_attrs: &["spec"],
},
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_erlang_root(node, source),
ChunkContext::ClassBody => classify_erlang_class(node, source),
ChunkContext::FunctionBody => classify_erlang_function(node, source),
}
}
}
fn classify_erlang_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"module_attribute" => {
make_kind_chunk(node, ChunkKind::Module, erlang_name(node, source), source, None)
},
"pp_define" => {
make_kind_chunk(node, ChunkKind::Macro, erlang_name(node, source), source, None)
},
"record_decl" => make_candidate(
node,
ChunkKind::Struct,
format!("record_{}", erlang_name(node, source)?),
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::ClassBody)),
source,
),
"type_alias" => {
make_kind_chunk(node, ChunkKind::Type, erlang_name(node, source), source, None)
},
// The Erlang grammar exposes each top-level clause as its own `fun_decl`.
// Keep that shape instead of inventing a synthetic merged function node.
"fun_decl" => make_kind_chunk(
node,
ChunkKind::Function,
erlang_name(node, source),
source,
Some(recurse_self(node, ChunkContext::FunctionBody)),
),
"spec" => return None,
_ => return None,
})
}
fn classify_erlang_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"record_field" => {
make_kind_chunk(node, ChunkKind::Field, erlang_name(node, source), source, None)
},
_ => return None,
})
}
fn classify_erlang_function<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"function_clause" => make_kind_chunk(
node,
ChunkKind::Clause,
erlang_name(node, source),
source,
recurse_clause_body(node),
),
"fun_clause" | "cr_clause" => make_candidate(
node,
ChunkKind::Clause,
None,
NameStyle::Named,
signature_for_node(node, source),
recurse_clause_body(node),
source,
),
"receive_after" => make_candidate(
node,
ChunkKind::After,
None,
NameStyle::Named,
signature_for_node(node, source),
recurse_clause_body(node),
source,
),
"catch_clause" => make_candidate(
node,
ChunkKind::Catch,
None,
NameStyle::Named,
signature_for_node(node, source),
recurse_clause_body(node),
source,
),
"receive_expr" => make_candidate(
node,
ChunkKind::Receive,
None,
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::FunctionBody)),
source,
),
"case_expr" => make_candidate(
node,
ChunkKind::Case,
None,
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::FunctionBody)),
source,
),
"try_expr" => make_candidate(
node,
ChunkKind::Try,
None,
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::FunctionBody)),
source,
),
"anonymous_fun" => make_kind_chunk(
node,
ChunkKind::Function,
Some("anonymous".to_string()),
source,
Some(recurse_self(node, ChunkContext::FunctionBody)),
),
_ => return None,
})
}
-320
View File
@@ -1,320 +0,0 @@
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct GoClassifier;
const ROOT_RULES: &[super::classify::SemanticRule] = &[
// ── Imports / package ──
semantic_rule(
"package_clause",
ChunkKind::Module,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::None,
),
semantic_rule(
"import_declaration",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
// ── Functions ──
semantic_rule(
"function_declaration",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"method_declaration",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
// ── Statements ──
semantic_rule(
"expression_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"go_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"defer_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"send_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const CLASS_RULES: &[super::classify::SemanticRule] = &[
// ── Methods ──
semantic_rule(
"method_spec",
ChunkKind::Method,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::None,
),
// ── Field / method lists ──
semantic_rule(
"field_declaration_list",
ChunkKind::Fields,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"method_spec_list",
ChunkKind::Methods,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const FUNCTION_RULES: &[super::classify::SemanticRule] = &[
// ── Control flow ──
semantic_rule(
"if_statement",
ChunkKind::If,
RuleStyle::Named,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"for_statement",
ChunkKind::For,
RuleStyle::Named,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"switch_statement",
ChunkKind::Switch,
RuleStyle::Named,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"expression_switch_statement",
ChunkKind::Switch,
RuleStyle::Named,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"type_switch_statement",
ChunkKind::Switch,
RuleStyle::Named,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"select_statement",
ChunkKind::Switch,
RuleStyle::Named,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
// ── Statements ──
semantic_rule(
"go_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"defer_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"send_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const GO_TABLES: ClassifierTables = ClassifierTables {
root: ROOT_RULES,
class: CLASS_RULES,
function: FUNCTION_RULES,
structural_overrides: StructuralOverrides::EMPTY,
};
impl LangClassifier for GoClassifier {
fn tables(&self) -> &'static ClassifierTables {
&GO_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_root_custom(node, source),
ChunkContext::ClassBody => classify_class_custom(node, source),
ChunkContext::FunctionBody => classify_function_custom(node, source),
}
}
}
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Variables ──
"const_declaration" | "var_declaration" | "short_var_declaration" => {
Some(match extract_identifier(node, source) {
Some(name) => make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None),
None => group_candidate(node, ChunkKind::Declarations, source),
})
},
// ── Containers ──
"type_declaration" => Some(classify_type_decl(node, source)),
// ── Control flow (top-level scripts) ──
"if_statement"
| "switch_statement"
| "expression_switch_statement"
| "type_switch_statement"
| "select_statement"
| "for_statement" => Some(classify_function_go(node, source)),
_ => None,
}
}
fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Fields ──
"field_declaration" | "embedded_field" => Some(match extract_identifier(node, source) {
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
None => group_candidate(node, ChunkKind::Fields, source),
}),
_ => None,
}
}
fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Variables ──
"short_var_declaration" | "var_declaration" | "const_declaration" => {
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
Some(if span > 1 {
if let Some(name) = extract_identifier(node, source) {
make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)
} else {
group_from_sanitized(node, source)
}
} else {
group_from_sanitized(node, source)
})
},
_ => None,
}
}
/// Classify Go function-level nodes (reused for top-level control flow
/// delegation).
fn classify_function_go<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
match node.kind() {
"if_statement" => {
make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source)
},
"switch_statement"
| "expression_switch_statement"
| "type_switch_statement"
| "select_statement" => {
make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source)
},
"for_statement" => {
make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source)
},
_ => group_candidate(node, ChunkKind::Statements, source),
}
}
/// Classify Go `type_declaration` nodes.
///
/// A single `type_spec` with a struct/interface body becomes a container;
/// a single `type_spec` without one becomes a named leaf.
/// Multiple `type_spec` children (type group) become a group.
fn classify_type_decl<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let specs: Vec<Node<'t>> = named_children(node)
.into_iter()
.filter(|c| c.kind() == "type_spec")
.collect();
if specs.len() == 1 {
let spec = specs[0];
let name = extract_identifier(spec, source).unwrap_or_else(|| "anonymous".to_string());
if let Some(recurse) = recurse_type_spec(spec) {
return make_container_chunk_from(
node,
spec,
ChunkKind::Type,
Some(name),
source,
Some(recurse),
);
}
return make_kind_chunk_from(node, spec, ChunkKind::Type, Some(name), source, None);
}
group_candidate(node, ChunkKind::Declarations, source)
}
fn group_from_sanitized<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let sanitized = sanitize_node_kind(node.kind());
let kind = ChunkKind::from_sanitized_kind(sanitized);
let identifier = if kind == ChunkKind::Chunk {
Some(sanitized.to_string())
} else {
None
};
make_candidate(node, kind, identifier, NameStyle::Group, None, None, source)
}
/// For a `type_spec`, find a `struct_type` or `interface_type` child and return
/// its body (`field_declaration_list` or `method_spec_list`) as a recurse spec.
fn recurse_type_spec(node: Node<'_>) -> Option<RecurseSpec<'_>> {
let container = child_by_kind(node, &["struct_type", "interface_type"])?;
let body = child_by_kind(container, &["field_declaration_list", "method_spec_list"])
.unwrap_or(container);
Some(RecurseSpec { node: body, context: ChunkContext::ClassBody })
}
-294
View File
@@ -1,294 +0,0 @@
//! GraphQL-specific chunk classifier.
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
common::*,
kind::ChunkKind,
};
pub struct GraphqlClassifier;
impl LangClassifier for GraphqlClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: StructuralOverrides {
extra_trivia: &["comma"],
preserved_trivia: &[],
extra_root_wrappers: &[
"document",
"definition",
"type_system_definition",
"type_definition",
"executable_definition",
],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_graphql_root(node, source),
ChunkContext::ClassBody => classify_graphql_class(node, source),
ChunkContext::FunctionBody => classify_graphql_function(node, source),
}
}
}
fn classify_graphql_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"schema_definition" => Some(make_container_chunk(
node,
ChunkKind::Schema,
None,
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
)),
"directive_definition" => Some(make_named_graphql_chunk(
node,
ChunkKind::Directive,
extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string()),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["arguments_definition"]),
)),
"scalar_type_definition" => Some(make_named_graphql_chunk(
node,
ChunkKind::Type,
format!(
"scalar_{}",
extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string())
),
source,
None,
)),
"object_type_definition" => Some(make_container_chunk(
node,
ChunkKind::Type,
extract_graphql_name(node, source),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["fields_definition"]),
)),
"interface_type_definition" => Some(make_container_chunk(
node,
ChunkKind::Interface,
extract_graphql_name(node, source),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["fields_definition"]),
)),
"union_type_definition" => Some(make_kind_chunk(
node,
ChunkKind::Union,
extract_graphql_name(node, source),
source,
None,
)),
"enum_type_definition" => Some(make_container_chunk(
node,
ChunkKind::Enum,
extract_graphql_name(node, source),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["enum_values_definition"]),
)),
"input_object_type_definition" => Some(make_named_graphql_chunk(
node,
ChunkKind::Type,
format!(
"input_{}",
extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string())
),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["input_fields_definition"]),
)),
"operation_definition" => Some(make_named_graphql_chunk(
node,
ChunkKind::Operation,
extract_graphql_operation_chunk_name(node, source),
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["selection_set"]),
)),
"fragment_definition" => Some(make_named_graphql_chunk(
node,
ChunkKind::Operation,
format!(
"fragment_{}",
extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string())
),
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["selection_set"]),
)),
_ => None,
}
}
fn classify_graphql_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"root_operation_type_definition" => Some(make_named_graphql_chunk(
node,
ChunkKind::Root,
extract_graphql_operation_type(node, source).unwrap_or_else(|| "anonymous".to_string()),
source,
None,
)),
"field_definition" => {
let name = extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string());
let recurse = recurse_into(node, ChunkContext::ClassBody, &[], &["arguments_definition"]);
Some(match recurse {
Some(recurse) => {
make_container_chunk(node, ChunkKind::Field, Some(name), source, Some(recurse))
},
None => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
})
},
"input_value_definition" => Some(classify_graphql_input_value(node, source)),
"enum_value_definition" => Some(make_named_graphql_chunk(
node,
ChunkKind::Variant,
format!(
"value_{}",
extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string())
),
source,
None,
)),
_ => None,
}
}
fn classify_graphql_function<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"selection" => classify_graphql_selection(node, source),
_ => None,
}
}
fn classify_graphql_selection<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let child = first_named_child(node)?;
match child.kind() {
"field" => {
let name = extract_graphql_name(child, source).unwrap_or_else(|| "anonymous".to_string());
let recurse = recurse_into(child, ChunkContext::FunctionBody, &[], &["selection_set"]);
Some(match recurse {
Some(recurse) => make_container_chunk_from(
node,
child,
ChunkKind::Field,
Some(name),
source,
Some(recurse),
),
None => make_kind_chunk_from(node, child, ChunkKind::Field, Some(name), source, None),
})
},
"fragment_spread" => Some(make_named_graphql_chunk_from(
node,
child,
ChunkKind::Operation,
format!(
"spread_{}",
extract_graphql_name(child, source).unwrap_or_else(|| "anonymous".to_string())
),
source,
None,
)),
"inline_fragment" => Some(make_container_chunk_from(
node,
child,
ChunkKind::InlineFragment,
None,
source,
recurse_into(child, ChunkContext::FunctionBody, &[], &["selection_set"]),
)),
_ => None,
}
}
fn classify_graphql_input_value<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let name = extract_graphql_name(node, source).unwrap_or_else(|| "anonymous".to_string());
match node.parent().map(|parent| parent.kind()) {
Some("input_fields_definition") => {
make_kind_chunk(node, ChunkKind::Field, Some(name), source, None)
},
_ => make_kind_chunk(node, ChunkKind::Arg, Some(name), source, None),
}
}
fn make_named_graphql_chunk<'t>(
node: Node<'t>,
kind: ChunkKind,
identifier: impl Into<Option<String>>,
source: &str,
recurse: Option<RecurseSpec<'t>>,
) -> RawChunkCandidate<'t> {
make_candidate(
node,
kind,
identifier,
NameStyle::Named,
signature_for_node(node, source),
recurse,
source,
)
}
fn make_named_graphql_chunk_from<'t>(
range_node: Node<'t>,
signature_node: Node<'t>,
kind: ChunkKind,
identifier: impl Into<Option<String>>,
source: &str,
recurse: Option<RecurseSpec<'t>>,
) -> RawChunkCandidate<'t> {
make_candidate(
range_node,
kind,
identifier,
NameStyle::Named,
signature_for_node(signature_node, source),
recurse,
source,
)
}
fn extract_graphql_name(node: Node<'_>, source: &str) -> Option<String> {
find_graphql_name_node(node)
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
}
fn find_graphql_name_node(node: Node<'_>) -> Option<Node<'_>> {
match node.kind() {
"name" | "fragment_name" => Some(node),
_ => named_children(node)
.into_iter()
.find_map(find_graphql_name_node),
}
}
fn extract_graphql_operation_type(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["operation_type"])
.and_then(|kind| sanitize_identifier(node_text(source, kind.start_byte(), kind.end_byte())))
}
fn extract_graphql_operation_chunk_name(node: Node<'_>, source: &str) -> String {
let operation =
extract_graphql_operation_type(node, source).unwrap_or_else(|| "operation".to_string());
match extract_graphql_name(node, source) {
Some(name) => format!("{operation}_{name}"),
None => operation,
}
}
fn first_named_child(node: Node<'_>) -> Option<Node<'_>> {
(0..node.named_child_count()).find_map(|index| node.named_child(index))
}
@@ -1,195 +0,0 @@
//! Language-specific chunk classifiers for Haskell and Scala.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct HaskellScalaClassifier;
const HASKELL_SCALA_ROOT_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"import_declaration",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"package_declaration",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"module",
ChunkKind::Module,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"function_declaration",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"function_definition",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"class_definition",
ChunkKind::Class,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"object_definition",
ChunkKind::Module,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"trait_definition",
ChunkKind::Iface,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"type_alias_declaration",
ChunkKind::Type,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"type_item",
ChunkKind::Type,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"variable_declaration",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"assignment",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"expression_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const HASKELL_SCALA_FUNCTION_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"if_statement",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"match_expression",
ChunkKind::Match,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"for_expression",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"while_expression",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"block_expression",
ChunkKind::Block,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
];
const HASKELL_SCALA_TABLES: ClassifierTables = ClassifierTables {
root: HASKELL_SCALA_ROOT_RULES,
class: &[],
function: HASKELL_SCALA_FUNCTION_RULES,
structural_overrides: super::classify::StructuralOverrides::EMPTY,
};
impl LangClassifier for HaskellScalaClassifier {
fn tables(&self) -> &'static ClassifierTables {
&HASKELL_SCALA_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
if context != ChunkContext::ClassBody {
return None;
}
match node.kind() {
"function_declaration" | "function_definition" | "method_definition" => {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
let kind = if name == "constructor" {
ChunkKind::Constructor
} else {
ChunkKind::Function
};
let identifier = (kind != ChunkKind::Constructor).then_some(name);
Some(make_kind_chunk(
node,
kind,
identifier,
source,
resolve_recurse(node, ChunkContext::FunctionBody),
))
},
"variable_declaration" | "property_declaration" => {
Some(extract_identifier(node, source).map_or_else(
|| group_candidate(node, ChunkKind::Fields, source),
|name| make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
))
},
_ => None,
}
}
}
-163
View File
@@ -1,163 +0,0 @@
//! Language-specific chunk classifiers for HTML and XML.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule,
},
common::*,
kind::ChunkKind,
};
use crate::language::SupportLang;
pub struct HtmlXmlClassifier;
const HTML_XML_SHARED_RULES: &[super::classify::SemanticRule] = &[semantic_rule(
"text_node",
ChunkKind::Text,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
)];
const HTML_XML_TABLES: ClassifierTables = ClassifierTables {
root: HTML_XML_SHARED_RULES,
class: HTML_XML_SHARED_RULES,
function: &[],
structural_overrides: super::classify::StructuralOverrides::EMPTY,
};
/// Classify an element-like node as a container with tag semantics.
///
/// Uses `extract_markup_tag_name` directly because the shared
/// `extract_identifier` does not handle HTML/XML start-tag structures.
fn classify_element<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"script_element" => Some(classify_script_element(node, source)),
"style_element" => Some(classify_style_element(node, source)),
"element" => {
let tag_name =
extract_markup_tag_name(node, source).unwrap_or_else(|| "anonymous".to_string());
// HTML: child elements are direct children of `element`.
// XML: child elements are inside a `content` wrapper node.
let recurse_target = child_by_kind(node, &["content"]).unwrap_or(node);
Some(force_container(make_container_chunk(
node,
ChunkKind::Tag,
Some(tag_name),
source,
Some(recurse_self(recurse_target, ChunkContext::ClassBody)),
)))
},
"text_node" => Some(group_candidate(node, ChunkKind::Text, source)),
_ => None,
}
}
fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
classify_injected_raw_text_block(node, ChunkKind::Script, source, SupportLang::JavaScript)
}
fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
classify_injected_raw_text_block(node, ChunkKind::Style, source, SupportLang::Css)
}
fn classify_injected_raw_text_block<'t>(
node: Node<'t>,
kind: ChunkKind,
source: &str,
default_language: SupportLang,
) -> RawChunkCandidate<'t> {
let Some(content_node) = child_by_kind(node, &["raw_text"]) else {
return positional_candidate(node, kind, source);
};
let candidate = with_region_node(positional_candidate(node, kind, source), Some(content_node));
match resolve_embedded_language(node, source, default_language) {
Some(language) => with_injected_subtree(candidate, language, content_node),
None => candidate,
}
}
/// Extract the tag name from an HTML/XML element node.
///
/// HTML: `element` → `start_tag`/`self_closing_tag` → `tag_name`
/// XML (tree-sitter-xml): `element` → `STag`/`EmptyElemTag` → `Name`
fn extract_markup_tag_name(node: Node<'_>, source: &str) -> Option<String> {
named_children(node).into_iter().find_map(|child| {
let tag_name_kinds: &[&str] = match child.kind() {
// HTML
"start_tag" | "self_closing_tag" => &["tag_name"],
// XML (tree-sitter-xml grammar)
"STag" | "EmptyElemTag" => &["Name"],
_ => return None,
};
child_by_kind(child, tag_name_kinds)
.and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte())))
})
}
fn resolve_embedded_language(
node: Node<'_>,
source: &str,
default_language: SupportLang,
) -> Option<SupportLang> {
if let Some(language) = attribute_value(node, "lang", source) {
return SupportLang::from_alias(language.as_str());
}
Some(default_language)
}
fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option<String> {
let start = start_like(node)?;
for child in named_children(start) {
if child.kind() != "attribute" {
continue;
}
if extract_attribute_name(child, source).as_deref() != Some(name) {
continue;
}
if let Some(value) = child_by_kind(child, &["attribute_value", "quoted_attribute_value"]) {
return sanitize_identifier(&unquote_text(node_text(
source,
value.start_byte(),
value.end_byte(),
)));
}
return Some(name.to_string());
}
None
}
fn extract_attribute_name(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["attribute_name"])
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
}
fn start_like(node: Node<'_>) -> Option<Node<'_>> {
child_by_kind(node, &["start_tag", "self_closing_tag"])
}
const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> {
candidate.force_recurse = true;
candidate
}
impl LangClassifier for HtmlXmlClassifier {
fn tables(&self) -> &'static ClassifierTables {
&HTML_XML_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
if matches!(context, ChunkContext::Root | ChunkContext::ClassBody) {
return classify_element(node, source);
}
None
}
}
-88
View File
@@ -1,88 +0,0 @@
//! Chunk classifier for INI.
//!
//! The tree-sitter INI grammar is intentionally flat: a document contains
//! root-level `setting` nodes and `section` containers, and a section contains
//! only its own `setting` children. Mirror that structure directly instead of
//! inventing deeper hierarchy.
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier},
common::*,
kind::ChunkKind,
};
pub struct IniClassifier;
impl LangClassifier for IniClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: super::classify::StructuralOverrides::EMPTY,
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_ini_root(node, source),
ChunkContext::ClassBody => classify_ini_class(node, source),
ChunkContext::FunctionBody => None,
}
}
}
fn classify_ini_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"section" => make_container_chunk(
node,
ChunkKind::Section,
Some(ini_name(node, source)?),
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
),
// INI permits settings before any section header; keep them as first-class
// chunks instead of forcing them under a synthetic container.
"setting" => {
make_kind_chunk(node, ChunkKind::Key, Some(ini_name(node, source)?), source, None)
},
_ => return None,
})
}
fn classify_ini_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"setting" => {
make_kind_chunk(node, ChunkKind::Key, Some(ini_name(node, source)?), source, None)
},
_ => return None,
})
}
fn ini_name(node: Node<'_>, source: &str) -> Option<String> {
find_named_text(node, source, &["section_name", "setting_name", "text"]).and_then(|text| {
sanitize_identifier(text.trim().trim_start_matches('[').trim_end_matches(']'))
})
}
fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> {
if kinds.iter().any(|kind| node.kind() == *kind) {
return Some(node_text(source, node.start_byte(), node.end_byte()));
}
for child in named_children(node) {
if let Some(text) = find_named_text(child, source, kinds) {
return Some(text);
}
}
None
}
-918
View File
@@ -1,918 +0,0 @@
//! Jupyter notebook (`.ipynb`) chunker.
//!
//! Notebooks are JSON documents whose `cells` array carries the code/markdown
//! that users actually edit. This module parses the JSON, extracts each cell
//! as its own source fragment, and assembles a *virtual source* — the
//! concatenation of all cell bodies with one-line marker headers — that the
//! rest of the chunk pipeline can treat like any other text file.
//!
//! The per-cell sub-chunks come from recursively running [`build_chunk_tree`]
//! on the individual cell sources with their appropriate language (code cells
//! use the notebook's kernel language, defaulting to Python; markdown cells
//! use `markdown`; raw cells fall through to the blank-line fallback). Their
//! byte and line offsets are shifted into the virtual source and then
//! rewrapped under `cell_<n>` parent chunks so the resulting tree looks to
//! edit.rs exactly like a normal multi-symbol file.
//!
//! On write-back, [`notebook_to_json`] walks the (possibly edited) virtual
//! source, splits it at the cell markers, updates each cell's `source` field
//! in a preserved [`NotebookContext`], and serializes the whole notebook back
//! to JSON. Cell metadata (`metadata`, `outputs`, `execution_count`, `id`,
//! attachments, etc.) is preserved verbatim.
use std::sync::Arc;
use serde::Serialize;
use serde_json::{Map, Value};
use crate::chunk::{
build_chunk_tree, chunk_checksum,
kind::ChunkKind,
line_start_offsets,
types::{ChunkNode, ChunkTree},
};
/// Marker line prefix placed before every cell body in the virtual source.
///
/// Format: `# %%% oh-my-pi cell_<N> [<type>]`
///
/// The leading `#` makes the marker a valid comment in Python and most other
/// code languages, and the `oh-my-pi` tag makes accidental collision with
/// user content vanishingly unlikely. Markdown cells get the same marker —
/// `#` in markdown is a heading, but the marker line itself is stripped
/// before the markdown chunker parses the cell body (see
/// [`build_cell_sub_tree`]).
const MARKER_PREFIX: &str = "# %%% oh-my-pi cell_";
/// Returns true if `line` is a cell marker; parses the cell index and type.
fn parse_marker_line(line: &str) -> Option<(usize, &str)> {
let rest = line.strip_prefix(MARKER_PREFIX)?;
let (num_str, after_num) = rest.split_once(' ')?;
let index: usize = num_str.parse().ok()?;
let cell_type = after_num.strip_prefix('[')?.strip_suffix(']')?;
Some((index, cell_type))
}
fn format_marker(index: usize, cell_type: &str) -> String {
format!("{MARKER_PREFIX}{index} [{cell_type}]")
}
/// Metadata for a single notebook cell. Everything except the joined `source`
/// is preserved verbatim for JSON round-tripping.
#[derive(Clone)]
pub struct NotebookCell {
pub cell_type: String,
pub source: String,
pub metadata: Value,
pub outputs: Option<Value>,
pub execution_count: Option<Value>,
/// Additional fields on the cell object (`id`, `attachments`, …) so we
/// emit the same keys we consumed.
pub other: Map<String, Value>,
/// Whether the original cell used `"source"` as a string (`true`) or an
/// array of lines (`false`). Preserved so round-trips stay byte-identical
/// when only metadata or unrelated cells change.
pub source_was_string: bool,
}
/// Preserved notebook-level state used for rebuilding the JSON after edits.
#[derive(Clone)]
pub struct NotebookContext {
/// Cells in document order.
pub cells: Vec<NotebookCell>,
/// Top-level notebook fields other than `cells`, in original key order.
pub top_fields: Map<String, Value>,
/// Indent string detected from the original JSON (typically `" "`).
pub indent: String,
/// Whether the original file ended with a trailing `\n`.
pub trailing_newline: bool,
/// Normalized kernel language used for code cells (e.g. `python`).
pub kernel_language: String,
}
/// Result of a notebook parse: the virtual source ready for chunking plus
/// the context needed to rebuild the JSON later.
pub struct NotebookParse {
pub virtual_source: String,
pub context: NotebookContext,
}
// ────────────────────────────────────────────────────────────────────
// JSON parsing
// ────────────────────────────────────────────────────────────────────
/// Parse the raw ipynb JSON into a [`NotebookContext`] and the derived
/// virtual source text. Returns a descriptive error if the JSON is invalid
/// or not a notebook document.
pub fn parse_notebook(source: &str) -> Result<NotebookParse, String> {
let normalized = strip_bom(source);
let value: Value = serde_json::from_str(normalized)
.map_err(|err| format!("Invalid Jupyter notebook JSON: {err}"))?;
let Value::Object(obj) = value else {
return Err("Invalid Jupyter notebook: top-level value must be an object.".to_string());
};
let cells_val = obj
.get("cells")
.ok_or_else(|| "Invalid Jupyter notebook: missing `cells` array.".to_string())?;
let cells_arr = cells_val
.as_array()
.ok_or_else(|| "Invalid Jupyter notebook: `cells` is not an array.".to_string())?;
let mut cells = Vec::with_capacity(cells_arr.len());
for (i, raw) in cells_arr.iter().enumerate() {
cells.push(parse_cell(raw, i)?);
}
// Top-level fields minus `cells`, preserving insertion order.
let mut top_fields = Map::new();
for (k, v) in &obj {
if k != "cells" {
top_fields.insert(k.clone(), v.clone());
}
}
let kernel_language =
extract_kernel_language(&top_fields).unwrap_or_else(|| "python".to_string());
let indent = detect_json_indent(normalized);
let trailing_newline = normalized.ends_with('\n');
let ctx = NotebookContext { cells, top_fields, indent, trailing_newline, kernel_language };
let virtual_source = build_virtual_source(&ctx);
Ok(NotebookParse { virtual_source, context: ctx })
}
fn parse_cell(raw: &Value, index: usize) -> Result<NotebookCell, String> {
let obj = raw
.as_object()
.ok_or_else(|| format!("Invalid Jupyter notebook: cell {} is not an object.", index + 1))?;
let cell_type = obj
.get("cell_type")
.and_then(Value::as_str)
.unwrap_or("code")
.to_string();
let (source, source_was_string) = match obj.get("source") {
Some(Value::String(s)) => (s.clone(), true),
Some(Value::Array(arr)) => {
let mut joined = String::new();
for item in arr {
match item {
Value::String(s) => joined.push_str(s),
_ => {
return Err(format!(
"Invalid Jupyter notebook: cell {} source array contains non-string element.",
index + 1
));
},
}
}
(joined, false)
},
Some(Value::Null) | None => (String::new(), false),
Some(_) => {
return Err(format!(
"Invalid Jupyter notebook: cell {} has a non-string `source` field.",
index + 1
));
},
};
let metadata = obj
.get("metadata")
.cloned()
.unwrap_or_else(|| Value::Object(Map::new()));
let outputs = obj.get("outputs").cloned();
let execution_count = obj.get("execution_count").cloned();
// Additional fields (id, attachments, …) that we pass through untouched.
let mut other = Map::new();
for (k, v) in obj {
if !matches!(k.as_str(), "cell_type" | "source" | "metadata" | "outputs" | "execution_count")
{
other.insert(k.clone(), v.clone());
}
}
Ok(NotebookCell {
cell_type,
source,
metadata,
outputs,
execution_count,
other,
source_was_string,
})
}
fn strip_bom(source: &str) -> &str {
source.strip_prefix('\u{feff}').unwrap_or(source)
}
fn extract_kernel_language(top_fields: &Map<String, Value>) -> Option<String> {
if let Some(meta) = top_fields.get("metadata").and_then(Value::as_object) {
if let Some(lang) = meta
.get("kernelspec")
.and_then(Value::as_object)
.and_then(|k| k.get("language"))
.and_then(Value::as_str)
&& !lang.is_empty()
{
return Some(lang.to_ascii_lowercase());
}
if let Some(lang) = meta
.get("language_info")
.and_then(Value::as_object)
.and_then(|k| k.get("name"))
.and_then(Value::as_str)
&& !lang.is_empty()
{
return Some(lang.to_ascii_lowercase());
}
}
None
}
fn detect_json_indent(source: &str) -> String {
// Look for the first `\n` followed by whitespace inside the top-level object
// (i.e. after the opening `{`). This is a heuristic; Jupyter canonically
// uses a single space per level.
let Some(brace) = source.find('{') else {
return " ".to_string();
};
let rest = &source[brace + 1..];
let Some(nl) = rest.find('\n') else {
return " ".to_string();
};
let after_nl = &rest[nl + 1..];
let mut end = 0usize;
for ch in after_nl.chars() {
if ch == ' ' || ch == '\t' {
end += ch.len_utf8();
} else {
break;
}
}
if end == 0 {
" ".to_string()
} else {
after_nl[..end].to_string()
}
}
// ────────────────────────────────────────────────────────────────────
// Virtual source assembly
// ────────────────────────────────────────────────────────────────────
/// Build the virtual source text from the current cell list.
///
/// Each cell is preceded by a marker line and its body. Cells do not include
/// trailing blank separators — we rely purely on the marker line to delimit
/// adjacent cells so the reconstructed sources stay byte-identical to the
/// originals after whole-cell edits.
pub fn build_virtual_source(ctx: &NotebookContext) -> String {
let mut out = String::new();
for (i, cell) in ctx.cells.iter().enumerate() {
out.push_str(&format_marker(i + 1, &cell.cell_type));
out.push('\n');
out.push_str(&cell.source);
if !cell.source.is_empty() && !cell.source.ends_with('\n') {
out.push('\n');
}
}
out
}
// ────────────────────────────────────────────────────────────────────
// Chunk tree construction
// ────────────────────────────────────────────────────────────────────
/// Locate every cell marker in `source`, returning tuples of:
/// (`cell_number`, `marker_line_byte_start`, `content_byte_start`,
/// `content_byte_end`, `marker_line_number_1based`,
/// `content_start_line_1based`,
/// `content_end_line_1based_inclusive_or_zero_if_empty`)
struct CellRegion {
cell_num: usize,
cell_type: String,
marker_start: usize, // byte offset of the `#` starting the marker line
content_start: usize, // byte offset of the first byte of the cell body
content_end: usize, // byte offset one past the last byte of the cell body
marker_line: u32, // 1-based line number of the marker
content_line: u32, // 1-based line number of the first body line (or marker_line + 1)
content_end_line: u32, /* 1-based line number of the last body line (== content_line - 1
* for empty bodies) */
}
/// Scan a virtual source text for cell markers and return the list of
/// regions. Assumes markers occur at the very start of their line.
fn scan_cells(virtual_source: &str) -> Vec<CellRegion> {
let line_starts = line_start_offsets(virtual_source);
let mut regions: Vec<CellRegion> = Vec::new();
for (line_idx, &line_start) in line_starts.iter().enumerate() {
let line_end = if line_idx + 1 < line_starts.len() {
// Exclude the trailing newline
line_starts[line_idx + 1] - 1
} else {
virtual_source.len()
};
let line = &virtual_source[line_start..line_end];
if let Some((cell_num, cell_type)) = parse_marker_line(line) {
// Close the previous region if any.
if let Some(prev) = regions.last_mut() {
prev.content_end = line_start;
// Trim trailing newline from content_end if present (i.e. the body ended with
// \n). Actually we keep the newline: cell bodies end with \n except
// possibly the last. content_end_line = line of the last body byte.
if prev.content_end > prev.content_start {
let body_last_char_line = line_idx; // line_idx is 0-based, so this is the previous line
prev.content_end_line = body_last_char_line as u32;
} else {
prev.content_end_line = prev.content_line.saturating_sub(1);
}
}
let content_start = if line_idx + 1 < line_starts.len() {
line_starts[line_idx + 1]
} else {
virtual_source.len()
};
regions.push(CellRegion {
cell_num,
cell_type: cell_type.to_string(),
marker_start: line_start,
content_start,
content_end: virtual_source.len(), // provisional, closed by the next marker
marker_line: (line_idx as u32) + 1,
content_line: (line_idx as u32) + 2,
content_end_line: 0,
});
}
}
// Close the last region.
if let Some(last) = regions.last_mut() {
last.content_end = virtual_source.len();
if last.content_end > last.content_start {
// Count lines inside the body.
let body = &virtual_source[last.content_start..last.content_end];
let body_lines = body.matches('\n').count();
// If body doesn't end with '\n', the final partial line still counts.
let has_trailing_nl = body.ends_with('\n');
let content_lines = if has_trailing_nl {
body_lines
} else {
body_lines + 1
};
if content_lines > 0 {
last.content_end_line = last.content_line + content_lines as u32 - 1;
} else {
last.content_end_line = last.content_line.saturating_sub(1);
}
} else {
last.content_end_line = last.content_line.saturating_sub(1);
}
}
regions
}
/// Build a chunk tree from a virtual source text.
///
/// Re-scans the virtual source for cell markers, parses each cell body with
/// its language, and wraps the results in `cell_<n>` parent chunks. This is
/// the entry point used by both the initial JSON-based parse (via
/// [`parse_notebook`] → `build_virtual_source` → this function) and the
/// post-edit rebuilds that operate directly on the mutated virtual source.
pub fn build_notebook_tree_from_virtual(
virtual_source: &str,
kernel_language: &str,
) -> Result<ChunkTree, String> {
let total_lines = total_line_count(virtual_source);
let root_checksum = chunk_checksum(virtual_source.as_bytes());
let regions = scan_cells(virtual_source);
// Accumulated chunk nodes. Index 0 is reserved for the synthetic root.
let mut chunks: Vec<ChunkNode> = Vec::with_capacity(1 + regions.len() * 2);
chunks.push(ChunkNode {
path: String::new(),
identifier: None,
kind: ChunkKind::Root,
leaf: false,
virtual_content: None,
parent_path: None,
children: Vec::new(),
signature: None,
start_line: u32::from(total_lines != 0),
end_line: total_lines as u32,
line_count: total_lines as u32,
start_byte: 0,
end_byte: virtual_source.len() as u32,
checksum_start_byte: 0,
prologue_end_byte: Some(0),
epilogue_start_byte: Some(virtual_source.len() as u32),
checksum: root_checksum.clone(),
error: false,
indent: 0,
indent_char: String::new(),
group: false,
});
let mut root_children: Vec<String> = Vec::with_capacity(regions.len());
for region in &regions {
let cell_path = format!("cell_{}", region.cell_num);
let cell_language_str = match region.cell_type.as_str() {
"code" => kernel_language.to_string(),
"markdown" => "markdown".to_string(),
_ => String::new(),
};
let body = &virtual_source[region.content_start..region.content_end];
let body_has_content = !body.is_empty();
let cell_checksum = chunk_checksum(body.as_bytes());
// Build sub-chunks by parsing the cell body in isolation. Offsets in
// the returned tree are relative to `body`; we translate them into
// virtual-source coordinates by adding `region.content_start` bytes
// and `region.content_line - 1` lines.
let mut cell_children_paths: Vec<String> = Vec::new();
if body_has_content {
let sub_tree = build_chunk_tree(body, cell_language_str.as_str())
.map_err(|err| format!("Failed to parse cell_{} body: {err}", region.cell_num))?;
for sub_chunk in sub_tree.chunks.into_iter().skip(1) {
let translated_path = format!("{}.{}", cell_path, sub_chunk.path);
let translated_parent = match sub_chunk.parent_path.as_deref() {
Some("") | None => Some(cell_path.clone()),
Some(other) => Some(format!("{cell_path}.{other}")),
};
let translated_children: Vec<String> = sub_chunk
.children
.iter()
.map(|c| format!("{cell_path}.{c}"))
.collect();
let shifted_start_byte = sub_chunk
.start_byte
.saturating_add(region.content_start as u32);
let shifted_end_byte = sub_chunk
.end_byte
.saturating_add(region.content_start as u32);
let line_shift = region.content_line.saturating_sub(1);
chunks.push(ChunkNode {
path: translated_path.clone(),
identifier: sub_chunk.identifier,
kind: sub_chunk.kind,
leaf: sub_chunk.leaf,
virtual_content: sub_chunk.virtual_content,
parent_path: translated_parent,
children: translated_children,
signature: sub_chunk.signature,
start_line: sub_chunk.start_line.saturating_add(line_shift),
end_line: sub_chunk.end_line.saturating_add(line_shift),
line_count: sub_chunk.line_count,
start_byte: shifted_start_byte,
end_byte: shifted_end_byte,
checksum_start_byte: sub_chunk
.checksum_start_byte
.saturating_add(region.content_start as u32),
prologue_end_byte: sub_chunk
.prologue_end_byte
.map(|b| b.saturating_add(region.content_start as u32)),
epilogue_start_byte: sub_chunk
.epilogue_start_byte
.map(|b| b.saturating_add(region.content_start as u32)),
checksum: sub_chunk.checksum,
error: sub_chunk.error,
indent: sub_chunk.indent,
indent_char: sub_chunk.indent_char,
group: false,
});
}
for sub_path in sub_tree.root_children {
cell_children_paths.push(format!("{cell_path}.{sub_path}"));
}
}
let cell_line_count = {
let body_lines = if body_has_content {
if body.ends_with('\n') {
body.matches('\n').count()
} else {
body.matches('\n').count() + 1
}
} else {
0
};
1 + body_lines as u32
};
let cell_end_line = region.marker_line + cell_line_count.saturating_sub(1);
let cell_leaf = cell_children_paths.is_empty();
chunks.push(ChunkNode {
path: cell_path.clone(),
identifier: Some(cell_path.clone()),
kind: ChunkKind::Cell,
leaf: cell_leaf,
virtual_content: None,
parent_path: Some(String::new()),
children: cell_children_paths,
signature: Some(format!("cell_{} ({})", region.cell_num, region.cell_type)),
start_line: region.marker_line,
end_line: cell_end_line,
line_count: cell_line_count,
start_byte: region.marker_start as u32,
end_byte: region.content_end as u32,
checksum_start_byte: region.content_start as u32,
prologue_end_byte: Some(region.content_start as u32),
epilogue_start_byte: Some(region.content_end as u32),
checksum: cell_checksum,
error: false,
indent: 0,
indent_char: String::new(),
group: false,
});
root_children.push(cell_path);
}
// Populate root children now that every cell is known.
if let Some(root) = chunks.get_mut(0) {
root.children.clone_from(&root_children);
}
// Sort chunks so the cell parent always comes before its sub-chunks,
// matching the invariant that other paths rely on (render, edit
// scheduling, line-to-chunk lookup). Keep the root at index 0.
// The insertion order above places sub-chunks before the cell parent, so
// we need to reorder: for each cell region, move the cell parent ahead of
// its sub-chunks.
//
// Simpler: rebuild the chunks list by iterating cells, emitting the cell
// parent followed by its sub-chunks in path order.
let mut reordered: Vec<ChunkNode> = Vec::with_capacity(chunks.len());
reordered.push(chunks.remove(0)); // root
let mut remaining: Vec<ChunkNode> = chunks;
for cell_path in &root_children {
// Extract the cell parent first.
if let Some(pos) = remaining.iter().position(|c| &c.path == cell_path) {
reordered.push(remaining.remove(pos));
}
// Then any descendants of this cell.
let prefix = format!("{cell_path}.");
let mut i = 0;
while i < remaining.len() {
if remaining[i].path.starts_with(&prefix) {
reordered.push(remaining.remove(i));
} else {
i += 1;
}
}
}
// Anything left over (shouldn't happen, but be defensive).
reordered.extend(remaining);
Ok(ChunkTree {
language: "ipynb".to_string(),
checksum: root_checksum,
line_count: total_lines as u32,
parse_errors: 0,
parse_error_lines: Vec::new(),
fallback: false,
root_path: String::new(),
root_children,
chunks: reordered,
})
}
fn total_line_count(source: &str) -> usize {
if source.is_empty() {
0
} else {
source.bytes().filter(|b| *b == b'\n').count() + 1
}
}
// ────────────────────────────────────────────────────────────────────
// Virtual → JSON round-trip
// ────────────────────────────────────────────────────────────────────
/// Update a [`NotebookContext`] from a (possibly edited) virtual source,
/// then serialize it back to JSON. Cells that no longer appear in the
/// virtual source are dropped; cells whose markers survive get their
/// `source` field replaced with the current body.
///
/// Returns the serialized JSON text ready to be written to disk.
pub fn notebook_to_json(
virtual_source: &str,
base_ctx: &NotebookContext,
) -> Result<String, String> {
let mut ctx = base_ctx.clone();
let regions = scan_cells(virtual_source);
// Rebuild the cells array in the order markers appear in the virtual
// source. Look up each marker's original cell by 1-based cell_num so
// edits that reorder cells via sibling insertion continue to track the
// right metadata.
let mut new_cells: Vec<NotebookCell> = Vec::with_capacity(regions.len());
for region in &regions {
let body_slice = &virtual_source[region.content_start..region.content_end];
// Trim the single trailing newline that the virtual source format
// adds so edits that replace an entire cell body don't grow by one
// line every round-trip.
let body = trim_virtual_body(body_slice);
let original = ctx.cells.get(region.cell_num.saturating_sub(1)).cloned();
let cell = match original {
Some(mut cell) => {
cell.source = body.to_string();
cell.cell_type.clone_from(&region.cell_type);
cell
},
None => NotebookCell {
cell_type: region.cell_type.clone(),
source: body.to_string(),
metadata: Value::Object(Map::new()),
outputs: match region.cell_type.as_str() {
"code" => Some(Value::Array(Vec::new())),
_ => None,
},
execution_count: match region.cell_type.as_str() {
"code" => Some(Value::Null),
_ => None,
},
other: Map::new(),
source_was_string: false,
},
};
new_cells.push(cell);
}
ctx.cells = new_cells;
let json = serialize_notebook(&ctx)?;
Ok(json)
}
/// Strip a single trailing newline from `body`, if present. The virtual
/// source always terminates each cell body with `\n` to make the markers
/// start on a fresh line; we remove that byte so the cell's stored source
/// matches the semantic content.
fn trim_virtual_body(body: &str) -> &str {
body.strip_suffix('\n').unwrap_or(body)
}
fn serialize_notebook(ctx: &NotebookContext) -> Result<String, String> {
// Build the cells array first.
let mut cells_arr: Vec<Value> = Vec::with_capacity(ctx.cells.len());
for cell in &ctx.cells {
cells_arr.push(cell_to_value(cell));
}
// Rebuild the top-level object preserving the original key order with
// `cells` injected at the position it originally occupied. If the input
// had no `cells` key (we wouldn't be here), we append.
let mut top = Map::new();
let mut cells_inserted = false;
for (k, v) in &ctx.top_fields {
top.insert(k.clone(), v.clone());
if k == "metadata" && !cells_inserted {
// Jupyter's canonical order is cells, metadata, nbformat,
// nbformat_minor. We preserve whatever we found.
}
}
// If the original document had `cells` somewhere, we want to re-insert
// it at roughly the same slot. Jupyter always writes `cells` first, so
// build a fresh Map in canonical order: cells then the preserved
// top_fields.
let mut final_top = Map::new();
final_top.insert("cells".to_string(), Value::Array(cells_arr));
cells_inserted = true;
for (k, v) in top {
if k != "cells" {
final_top.insert(k, v);
}
}
let _ = cells_inserted;
let indent_bytes = ctx.indent.as_bytes().to_vec();
let formatter = serde_json::ser::PrettyFormatter::with_indent(&indent_bytes);
let mut buf: Vec<u8> = Vec::with_capacity(1024);
{
let mut ser = serde_json::Serializer::with_formatter(&mut buf, formatter);
Value::Object(final_top)
.serialize(&mut ser)
.map_err(|err| format!("Failed to serialize notebook JSON: {err}"))?;
}
let mut text = String::from_utf8(buf)
.map_err(|err| format!("Serialized notebook is not valid UTF-8: {err}"))?;
if ctx.trailing_newline && !text.ends_with('\n') {
text.push('\n');
}
Ok(text)
}
fn cell_to_value(cell: &NotebookCell) -> Value {
let mut obj = Map::new();
obj.insert("cell_type".to_string(), Value::String(cell.cell_type.clone()));
// Preserve `id` and similar fields that idiomatically appear before
// `metadata` in nbformat 4+.
for (k, v) in &cell.other {
if !matches!(k.as_str(), "metadata" | "outputs" | "execution_count" | "source") {
obj.insert(k.clone(), v.clone());
}
}
obj.insert("metadata".to_string(), cell.metadata.clone());
if cell.cell_type == "code" {
obj.insert(
"execution_count".to_string(),
cell.execution_count.clone().unwrap_or(Value::Null),
);
obj.insert(
"outputs".to_string(),
cell
.outputs
.clone()
.unwrap_or_else(|| Value::Array(Vec::new())),
);
} else {
if let Some(outputs) = &cell.outputs {
obj.insert("outputs".to_string(), outputs.clone());
}
if let Some(ec) = &cell.execution_count {
obj.insert("execution_count".to_string(), ec.clone());
}
}
obj.insert("source".to_string(), source_to_value(&cell.source, cell.source_was_string));
Value::Object(obj)
}
/// Convert a flat source string to the Jupyter `source` field representation.
///
/// If the cell originally used a string (or a new cell was inserted), we keep
/// it as a string. Otherwise we split into the canonical `Vec<String>` with
/// each element preserving its trailing `\n`.
fn source_to_value(source: &str, was_string: bool) -> Value {
if was_string {
return Value::String(source.to_string());
}
if source.is_empty() {
return Value::Array(Vec::new());
}
let mut parts: Vec<Value> = Vec::new();
for line in source.split_inclusive('\n') {
parts.push(Value::String(line.to_string()));
}
Value::Array(parts)
}
// ────────────────────────────────────────────────────────────────────
// Shared helpers
// ────────────────────────────────────────────────────────────────────
/// Atomically-shareable notebook context; used by [`ChunkStateInner`] to
/// carry the notebook metadata through edit cycles.
pub type SharedNotebookContext = Arc<NotebookContext>;
#[cfg(test)]
mod tests {
use serde_json::json;
use super::*;
use crate::chunk::state::ChunkStateInner;
fn sample_notebook() -> String {
let value = json!({
"cells": [
{
"cell_type": "code",
"source": ["def foo():\n", " return 1\n"],
"metadata": {},
"outputs": [],
"execution_count": null
},
{
"cell_type": "markdown",
"source": ["# Hello\n", "World\n"],
"metadata": {}
},
{
"cell_type": "code",
"source": ["class Bar:\n", " def baz(self):\n", " pass\n"],
"metadata": {},
"outputs": [],
"execution_count": 3
}
],
"metadata": {
"kernelspec": {
"language": "python"
}
},
"nbformat": 4,
"nbformat_minor": 5
});
serde_json::to_string_pretty(&value).expect("static json")
}
#[test]
fn parses_notebook_into_cells() {
let nb = parse_notebook(&sample_notebook()).expect("valid notebook");
assert_eq!(nb.context.cells.len(), 3);
assert_eq!(nb.context.cells[0].cell_type, "code");
assert_eq!(nb.context.cells[1].cell_type, "markdown");
assert_eq!(nb.context.cells[2].cell_type, "code");
assert_eq!(nb.context.kernel_language, "python");
}
#[test]
fn virtual_source_contains_all_cells() {
let nb = parse_notebook(&sample_notebook()).expect("valid notebook");
let vs = &nb.virtual_source;
assert!(vs.contains("def foo():"), "cell 1 body missing");
assert!(vs.contains("# Hello"), "cell 2 body missing");
assert!(vs.contains("class Bar:"), "cell 3 body missing");
assert!(vs.contains("# %%% oh-my-pi cell_1 [code]"), "cell_1 marker missing");
assert!(vs.contains("# %%% oh-my-pi cell_2 [markdown]"), "cell_2 marker missing");
assert!(vs.contains("# %%% oh-my-pi cell_3 [code]"), "cell_3 marker missing");
}
#[test]
fn builds_cell_level_chunks() {
let nb = parse_notebook(&sample_notebook()).expect("valid notebook");
let tree =
build_notebook_tree_from_virtual(&nb.virtual_source, "python").expect("tree should build");
assert_eq!(
tree.root_children,
vec!["cell_1", "cell_2", "cell_3"],
"root children should be the three cells"
);
let cell1 = tree
.chunks
.iter()
.find(|c| c.path == "cell_1")
.expect("cell_1 chunk");
assert!(!cell1.leaf, "code cell with a function should not be a leaf");
assert!(
cell1
.children
.iter()
.any(|p| p.starts_with("cell_1.fn_foo")),
"cell_1 should contain fn_foo, got {:?}",
cell1.children
);
}
#[test]
fn sub_chunk_paths_are_prefixed_with_cell() {
let nb = parse_notebook(&sample_notebook()).expect("valid notebook");
let tree =
build_notebook_tree_from_virtual(&nb.virtual_source, "python").expect("tree should build");
let cell3 = tree
.chunks
.iter()
.find(|c| c.path == "cell_3")
.expect("cell_3 chunk");
assert!(
cell3
.children
.iter()
.any(|p| p.starts_with("cell_3.cls_Bar")),
"cell_3 should contain cls_Bar, got {:?}",
cell3.children
);
let bar_method = tree
.chunks
.iter()
.find(|c| c.path == "cell_3.cls_Bar.fn_baz");
assert!(bar_method.is_some(), "cell_3.cls_Bar.fn_baz should exist");
}
#[test]
fn chunk_state_parse_ipynb_carries_notebook_context() {
let json = sample_notebook();
let state =
ChunkStateInner::parse(json, "ipynb".to_string()).expect("ChunkState should parse ipynb");
assert_eq!(state.language(), "ipynb");
// Source is the virtual source, not the JSON
assert!(state.source().contains("# %%% oh-my-pi cell_1"));
// The notebook context is preserved for JSON round-trip
let ctx = state
.notebook
.as_ref()
.expect("notebook context should be set");
let json_out = notebook_to_json(state.source(), ctx).expect("should serialize back to JSON");
let reparsed: serde_json::Value =
serde_json::from_str(&json_out).expect("output should be valid JSON");
let cells = reparsed["cells"].as_array().expect("cells array");
assert_eq!(cells.len(), 3, "should still have 3 cells");
assert_eq!(cells[0]["cell_type"], "code");
assert_eq!(cells[2]["execution_count"], 3);
}
#[test]
fn empty_notebook_produces_empty_tree() {
let json = r#"{"cells": [], "metadata": {}, "nbformat": 4, "nbformat_minor": 5}"#;
let state = ChunkStateInner::parse(json.to_string(), "ipynb".to_string())
.expect("empty notebook should parse");
assert!(state.tree().root_children.is_empty());
}
}
-564
View File
@@ -1,564 +0,0 @@
//! JavaScript / TypeScript / TSX chunk classifier.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, WrapperSignature,
WrapperTransform, classify_with_defaults, first_wrapper_content_child,
promote_wrapper_candidate, semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct JsTsClassifier;
fn recurse_internal_module(node: Node<'_>) -> Option<RecurseSpec<'_>> {
recurse_into(node, ChunkContext::ClassBody, &["body"], &["statement_block"])
}
static JSTS_TABLES: ClassifierTables = ClassifierTables {
root: &[
semantic_rule(
"import_statement",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"import_declaration",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"function_declaration",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"function",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"function_expression",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"arrow_function",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"generator_function",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"generator_function_declaration",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"class_declaration",
ChunkKind::Class,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"class",
ChunkKind::Class,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"class_expression",
ChunkKind::Class,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"interface_declaration",
ChunkKind::Interface,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"enum_declaration",
ChunkKind::Enum,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"type_alias_declaration",
ChunkKind::Type,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::None,
),
],
class: &[
semantic_rule(
"constructor",
ChunkKind::Constructor,
RuleStyle::Named,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"class_static_block",
ChunkKind::StaticInit,
RuleStyle::Named,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"type_alias_declaration",
ChunkKind::Type,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::None,
),
],
function: &[],
structural_overrides: super::classify::StructuralOverrides::EMPTY,
};
impl LangClassifier for JsTsClassifier {
fn tables(&self) -> &'static ClassifierTables {
&JSTS_TABLES
}
fn is_trivia(&self, kind: &str) -> bool {
// Whitespace/text runs between JSX elements carry no structure and
// should be absorbed as leading trivia of the next element (matching
// the existing comment-absorption semantics).
kind == "jsx_text"
}
fn should_skip_child(&self, kind: &str) -> bool {
// JSX opening and closing elements are part of the enclosing
// `jsx_element` chunk's framing, not children in their own right.
// Skip them entirely when enumerating children so they don't pollute
// the chunk tree with noisy 1‑line entries and, crucially, so they
// don't get absorbed backward into the next real child.
matches!(kind, "jsx_opening_element" | "jsx_closing_element")
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_root_custom(node, source),
ChunkContext::ClassBody => classify_class_custom(node, source),
ChunkContext::FunctionBody => Some(classify_function_js(node, source)),
}
}
}
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Exports / decorators ──
"export_statement" => Some(classify_export_statement(ChunkContext::Root, node, source)),
"decorated_definition" => promote_wrapper_candidate(
&JsTsClassifier,
ChunkContext::Root,
node,
source,
WrapperTransform { signature: WrapperSignature::Wrapper, ..WrapperTransform::default() },
)
.or_else(|| Some(positional_candidate(node, ChunkKind::Block, source))),
// ── Variables ──
"lexical_declaration" | "variable_declaration" => Some(classify_var_decl_js(node, source)),
// ── Containers with custom recursion ──
"internal_module" => {
Some(container_candidate(node, ChunkKind::Module, source, recurse_internal_module(node)))
},
// ── Control flow at top level ──
"if_statement" | "switch_statement" | "switch_expression" | "try_statement"
| "for_statement" | "for_in_statement" | "for_of_statement" | "while_statement"
| "do_statement" | "with_statement" => Some(classify_function_js(node, source)),
// ── Statements ──
"expression_statement" => {
// Unwrap `expression_statement` wrapping an `internal_module` (namespace).
let inner = named_children(node)
.into_iter()
.find(|c| c.kind() == "internal_module");
if let Some(ns) = inner {
Some(container_candidate(ns, ChunkKind::Module, source, recurse_internal_module(ns)))
} else {
Some(group_candidate(node, ChunkKind::Statements, source))
}
},
_ => None,
}
}
fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Exports / decorators (re-exported members) ──
"export_statement" => Some(classify_export_statement(ChunkContext::ClassBody, node, source)),
"decorated_definition" => promote_wrapper_candidate(
&JsTsClassifier,
ChunkContext::ClassBody,
node,
source,
WrapperTransform { signature: WrapperSignature::Wrapper, ..WrapperTransform::default() },
)
.or_else(|| Some(positional_candidate(node, ChunkKind::Block, source))),
// ── Variables ──
"lexical_declaration" | "variable_declaration" => Some(classify_var_decl_js(node, source)),
// ── Methods ──
"method_definition" | "method_signature" | "abstract_method_signature" => {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
if name == "constructor" {
Some(make_kind_chunk(
node,
ChunkKind::Constructor,
None,
source,
recurse_body(node, ChunkContext::FunctionBody),
))
} else {
Some(make_kind_chunk(
node,
ChunkKind::Function,
Some(name),
source,
recurse_body(node, ChunkContext::FunctionBody),
))
}
},
// ── Fields ──
"public_field_definition"
| "field_definition"
| "property_definition"
| "property_signature"
| "property_declaration"
| "abstract_class_field" => match extract_identifier(node, source) {
Some(name) => Some(make_kind_chunk(node, ChunkKind::Field, Some(name), source, None)),
None => Some(group_candidate(node, ChunkKind::Fields, source)),
},
// ── Enum members ──
"enum_assignment" | "enum_member_declaration" => match extract_identifier(node, source) {
Some(name) => Some(make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None)),
None => Some(group_candidate(node, ChunkKind::Variants, source)),
},
_ => None,
}
}
/// Classify nodes inside a function body for JS/TS.
fn classify_function_js<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
match node.kind() {
// ── Control flow ──
"if_statement" => {
make_candidate(node, ChunkKind::If, None, NameStyle::Named, None, fn_recurse(), source)
},
"switch_statement" | "switch_expression" => {
make_candidate(node, ChunkKind::Switch, None, NameStyle::Named, None, fn_recurse(), source)
},
"try_statement" => {
make_candidate(node, ChunkKind::Try, None, NameStyle::Named, None, fn_recurse(), source)
},
// ── Loops ──
"for_statement" => {
make_candidate(node, ChunkKind::For, None, NameStyle::Named, None, fn_recurse(), source)
},
"for_in_statement" => {
make_candidate(node, ChunkKind::ForIn, None, NameStyle::Named, None, fn_recurse(), source)
},
"for_of_statement" => {
make_candidate(node, ChunkKind::ForOf, None, NameStyle::Named, None, fn_recurse(), source)
},
"while_statement" => {
make_candidate(node, ChunkKind::While, None, NameStyle::Named, None, fn_recurse(), source)
},
"do_statement" => {
make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source)
},
// ── Blocks ──
"with_statement" => {
make_candidate(node, ChunkKind::Block, None, NameStyle::Named, None, fn_recurse(), source)
},
// ── Variables ──
"lexical_declaration" | "variable_declaration" => {
if let Some(name) = extract_single_declarator_name(node, source) {
make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None)
} else {
group_from_sanitized(node, source)
}
},
// ── Return statements ──
// A bare `return <Link>…</Link>` or `return (<Link>…</Link>)` creates a
// huge monolithic leaf chunk in React components. Recurse into the JSX
// so each child element inside the returned tree stays individually
// addressable. Callback-with-trailing-block patterns such as
// `return items.map(item => { … })` are handled by the shared
// call-with-callback promotion in `classify_with_defaults` via the
// `return_statement` arm below.
"return_statement" => classify_return_statement_js(node, source),
// ── JSX elements ──
// Inside function bodies, JSX elements become container chunks with
// their tag name so React component trees are navigable instead of
// opaque walls of markup.
"jsx_element" => classify_jsx_element(node, source),
"jsx_self_closing_element" => classify_jsx_self_closing_element(node, source),
"jsx_fragment" => make_candidate(
node,
ChunkKind::Tag,
Some("fragment".to_string()),
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::FunctionBody)),
source,
),
// ── Expression statements (enable call-with-callback promotion) ──
"expression_statement" => group_candidate(node, ChunkKind::Statements, source),
// ── Fallback ──
_ => group_from_sanitized(node, source),
}
}
/// Classify a `jsx_element` as a container chunk named after its tag.
///
/// The chunk recurses into itself so that nested JSX children are emitted as
/// sub-chunks. Structural JSX nodes (opening/closing elements, text,
/// attributes) are filtered out as trivia by the classifier's `is_trivia`
/// override, so only meaningful children (child elements, expression
/// containers) become chunks.
fn classify_jsx_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let tag_name = extract_jsx_tag_name(node, source);
let mut candidate = make_candidate(
node,
ChunkKind::Tag,
tag_name,
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::FunctionBody)),
source,
);
// Force recursion for jsx_elements that span more than a single
// source line. Without this, a `<div>` wrapping a single
// near-equal-sized child fails `recursion_narrows_scope` and the
// whole subtree collapses into one opaque chunk. One-line elements
// keep natural collapse behavior so short inline JSX stays a leaf.
if candidate
.range_end_line
.saturating_sub(candidate.range_start_line)
> 0
{
candidate.force_recurse = true;
}
candidate
}
/// Classify a `return_statement`.
///
/// Two patterns matter:
///
/// 1. `return <Link>…</Link>` / `return (<Link>…</Link>)` — unwrap any
/// parentheses and recurse directly into the JSX tree so each nested JSX
/// element is individually addressable.
/// 2. `return items.map(item => { … })` — the shared call-with-trailing-
/// callback promoter turns this into a named expression container that
/// recurses into the callback body. We invoke it explicitly here because the
/// shared promotion in `classify_with_defaults` only runs on groupable
/// leaves, and `Return` is not groupable.
fn classify_return_statement_js<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
if let Some(expr) = named_children(node).into_iter().next() {
let target = unwrap_parenthesized(expr);
if matches!(target.kind(), "jsx_element" | "jsx_fragment" | "jsx_self_closing_element") {
let mut candidate = make_candidate(
node,
ChunkKind::Return,
None::<String>,
NameStyle::Named,
signature_for_node(node, source),
Some(RecurseSpec { node: target, context: ChunkContext::FunctionBody }),
source,
);
// The JSX tree may span nearly the entire return statement, which
// would fail the `recursion_narrows_scope` check. Force recursion
// so the JSX children are always individually addressable.
candidate.force_recurse = true;
return candidate;
}
}
if let Some(mut promoted) = try_promote_call_with_callback(node, source) {
// The callback body is the sole child of the return value, so it
// spans nearly the entire return statement. Force recursion to
// guarantee the callback internals are addressable.
promoted.force_recurse = true;
return promoted;
}
group_from_sanitized(node, source)
}
/// Classify a `jsx_self_closing_element` as a leaf chunk named after its tag.
fn classify_jsx_self_closing_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let tag_name = extract_jsx_tag_name(node, source);
make_kind_chunk(node, ChunkKind::Tag, tag_name, source, None)
}
/// Unwrap nested `parenthesized_expression` wrappers to reach the meaningful
/// inner expression.
fn unwrap_parenthesized(mut node: Node<'_>) -> Node<'_> {
while node.kind() == "parenthesized_expression" {
let Some(inner) = named_children(node).into_iter().next() else {
break;
};
node = inner;
}
node
}
/// Extract the tag name from a `jsx_element` or `jsx_self_closing_element`.
///
/// The tag may be an identifier (`div`, `Link`), a member expression
/// (`Foo.Bar`), or a nested identifier. We sanitize the full text so path
/// segments remain valid identifiers.
fn extract_jsx_tag_name(node: Node<'_>, source: &str) -> Option<String> {
let name_holder = match node.kind() {
"jsx_element" => child_by_kind(node, &["jsx_opening_element"])?,
"jsx_self_closing_element" => node,
_ => return None,
};
let name_node = named_children(name_holder).into_iter().find(|child| {
matches!(
child.kind(),
"identifier" | "member_expression" | "nested_identifier" | "jsx_namespace_name"
)
})?;
sanitize_identifier(node_text(source, name_node.start_byte(), name_node.end_byte()))
}
/// Classify `const`/`let`/`var` declarations, promoting arrow functions
/// and class expressions to fn_/class_ chunks.
fn classify_var_decl_js<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
// Inline promotion logic — look for single variable_declarator with fn/class
// value.
let declarators: Vec<Node<'t>> = named_children(node)
.into_iter()
.filter(|c| c.kind() == "variable_declarator")
.collect();
if declarators.len() == 1 {
let decl = declarators[0];
if let Some(value) = decl.child_by_field_name("value") {
let name = extract_identifier(decl, source).unwrap_or_else(|| "anonymous".to_string());
match value.kind() {
"arrow_function" | "function_expression" | "function" => {
let recurse = recurse_body(value, ChunkContext::FunctionBody);
return make_kind_chunk(node, ChunkKind::Function, Some(name), source, recurse);
},
"class" | "class_expression" => {
let recurse = recurse_class(value);
return make_container_chunk(node, ChunkKind::Class, Some(name), source, recurse);
},
_ => {},
}
}
}
// Not promoted — fall back to var_NAME or group.
if let Some(name) = extract_single_declarator_name(node, source) {
return make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None);
}
group_candidate(node, ChunkKind::Declarations, source)
}
fn group_from_sanitized<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let sanitized = sanitize_node_kind(node.kind());
let kind = ChunkKind::from_sanitized_kind(sanitized);
let identifier = if kind == ChunkKind::Chunk {
Some(sanitized.to_string())
} else {
None
};
make_candidate(node, kind, identifier, NameStyle::Group, None, None, source)
}
/// Unwrap `export` / `export default` to classify the inner declaration.
///
/// Wrapper promotion handles declaration-like exports automatically.
/// `export default …` remaps the promoted child to `default_export`, while
/// re-exports and bare expression exports still fall through to `stmts`.
fn classify_export_statement<'t>(
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> RawChunkCandidate<'t> {
let header = normalized_header(source, node.start_byte(), node.end_byte());
let is_default = header.starts_with("export default");
if let Some(candidate) =
promote_wrapper_candidate(&JsTsClassifier, context, node, source, WrapperTransform {
kind: is_default.then_some(ChunkKind::DefaultExport),
name_style: is_default.then_some(NameStyle::Named),
clear_identifier: is_default,
..WrapperTransform::default()
}) {
return candidate;
}
let Some(child) = first_wrapper_content_child(&JsTsClassifier, node) else {
return if is_default {
make_kind_chunk(node, ChunkKind::DefaultExport, None, source, None)
} else {
group_candidate(node, ChunkKind::Statements, source)
};
};
if is_default {
return make_kind_chunk(node, ChunkKind::DefaultExport, None, source, None);
}
match child.kind() {
"lexical_declaration" | "variable_declaration" => {
classify_with_defaults(&JsTsClassifier, context, child, source)
},
_ => group_candidate(child, ChunkKind::Statements, source),
}
}
-129
View File
@@ -1,129 +0,0 @@
//! Language-specific chunk classifier for Just.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct JustClassifier;
const JUST_FUNCTION_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"recipe_line",
ChunkKind::Cmd,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"shebang",
ChunkKind::Shebang,
RuleStyle::Named,
NamingMode::None,
RecurseMode::None,
),
];
const JUST_TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: JUST_FUNCTION_RULES,
structural_overrides: StructuralOverrides {
extra_trivia: &[],
preserved_trivia: &[],
extra_root_wrappers: &["source_file"],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
fn first_named_child(node: Node<'_>) -> Option<Node<'_>> {
named_children(node).into_iter().next()
}
fn first_named_child_of_kind<'t>(node: Node<'t>, kind: &str) -> Option<Node<'t>> {
named_children(node)
.into_iter()
.find(|child| child.kind() == kind)
}
fn child_text<'a>(source: &'a str, node: Node<'_>) -> &'a str {
node_text(source, node.start_byte(), node.end_byte())
}
/// `set shell := ...` uses a dedicated `shell` token instead of a named
/// identifier, so parse the assignment head text instead of relying on fields.
fn extract_setting_name(node: Node<'_>, source: &str) -> Option<String> {
let header = child_text(source, node).lines().next()?.trim();
let rest = header.strip_prefix("set ")?;
let name = rest.split_once(":=")?.0.trim();
sanitize_identifier(name)
}
fn extract_alias_name(node: Node<'_>, source: &str) -> Option<String> {
first_named_child(node).and_then(|child| sanitize_identifier(child_text(source, child)))
}
fn extract_recipe_name(node: Node<'_>, source: &str) -> Option<String> {
let header = first_named_child_of_kind(node, "recipe_header")?;
first_named_child(header).and_then(|child| sanitize_identifier(child_text(source, child)))
}
fn classify_just_root_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"setting" => {
let name = extract_setting_name(node, source).unwrap_or_else(|| "anonymous".to_string());
make_kind_chunk(node, ChunkKind::Setting, Some(name), source, None)
},
"alias" => {
let name = extract_alias_name(node, source).unwrap_or_else(|| "anonymous".to_string());
make_kind_chunk(node, ChunkKind::Alias, Some(name), source, None)
},
"recipe" => {
let name = extract_recipe_name(node, source).unwrap_or_else(|| "anonymous".to_string());
make_container_chunk(
node,
ChunkKind::Recipe,
Some(name),
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["recipe_body"]),
)
},
_ => return None,
})
}
fn classify_just_body_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
// Just recipe bodies are line-oriented; tree-sitter exposes shell lines as
// `recipe_line` leaves rather than a nested shell AST.
"recipe_line" => group_candidate(node, ChunkKind::Cmd, source),
"shebang" => make_kind_chunk(node, ChunkKind::Shebang, None, source, None),
_ => return None,
})
}
impl LangClassifier for JustClassifier {
fn tables(&self) -> &'static ClassifierTables {
&JUST_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_just_root_node(node, source),
ChunkContext::FunctionBody => classify_just_body_node(node, source),
ChunkContext::ClassBody => None,
}
}
}
-341
View File
@@ -1,341 +0,0 @@
//! Language-specific chunk classifiers for Markdown and Handlebars.
use tree_sitter::Node;
use super::{
chunk_checksum,
classify::{ClassifierTables, LangClassifier},
common::*,
kind::ChunkKind,
types::ChunkNode,
};
use crate::language::SupportLang;
pub struct MarkupClassifier;
impl MarkupClassifier {
fn classify_section<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let name = extract_markdown_heading(node, source).unwrap_or_else(|| "anonymous".to_string());
force_container(make_container_chunk(
node,
ChunkKind::Section,
Some(name),
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
))
}
fn classify_block_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let name =
extract_glimmer_block_name(node, source).unwrap_or_else(|| "anonymous".to_string());
force_container(make_container_chunk(
node,
ChunkKind::Block,
Some(name),
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
))
}
fn classify_mustache_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let name =
extract_glimmer_mustache_name(node, source).unwrap_or_else(|| "anonymous".to_string());
make_kind_chunk(node, ChunkKind::Mustache, Some(name), source, None)
}
/// Classify HTML-like element nodes that appear inside handlebars blocks.
fn classify_element<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"element" | "script_element" | "style_element" | "element_node" => {
let name =
extract_element_tag_name(node, source).unwrap_or_else(|| "anonymous".to_string());
Some(force_container(make_container_chunk(
node,
ChunkKind::Tag,
Some(name),
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
)))
},
"text_node" => Some(group_candidate(node, ChunkKind::Text, source)),
_ => None,
}
}
}
impl LangClassifier for MarkupClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: super::classify::StructuralOverrides::EMPTY,
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
if !matches!(context, ChunkContext::Root | ChunkContext::ClassBody) {
return None;
}
match node.kind() {
"section" => Some(Self::classify_section(node, source)),
"fenced_code_block" => Some(classify_fenced_code_block(node, source)),
"html_block" => Some(classify_html_block(node, source)),
"block_statement" => Some(Self::classify_block_statement(node, source)),
"mustache_statement" => Some(Self::classify_mustache_statement(node, source)),
"element" | "script_element" | "style_element" | "element_node" | "text_node" => {
Self::classify_element(node, source)
},
_ => None,
}
}
fn post_process(
&self,
chunks: &mut Vec<ChunkNode>,
_root_children: &mut Vec<String>,
source: &str,
) {
add_markdown_table_row_chunks(chunks, source);
}
}
const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> {
candidate.force_recurse = true;
candidate
}
fn classify_fenced_code_block<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let embedded_language = fenced_code_language(node, source);
let identifier = embedded_language
.map(embedded_selector_token)
.map(str::to_string);
let candidate =
with_region_node(make_kind_chunk(node, ChunkKind::Code, identifier, source, None), None);
match (child_by_kind(node, &["code_fence_content"]), embedded_language) {
(Some(content_node), Some(language)) => {
with_injected_subtree(candidate, language, content_node)
},
_ => candidate,
}
}
fn classify_html_block<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let candidate =
with_region_node(make_kind_chunk(node, ChunkKind::Html, None, source, None), None);
with_injected_subtree(candidate, SupportLang::Html, node)
}
fn add_markdown_table_row_chunks(chunks: &mut Vec<ChunkNode>, source: &str) {
let original_len = chunks.len();
let mut additions = Vec::<(usize, Vec<ChunkNode>)>::new();
for (index, chunk) in chunks.iter().enumerate().take(original_len) {
if !chunk.children.is_empty() || matches!(chunk.kind, ChunkKind::Code | ChunkKind::Html) {
continue;
}
let rows = markdown_table_rows_for_chunk(source, chunk);
if rows.len() < 2
|| !rows
.iter()
.any(|row| markdown_table_separator_row(row.text))
{
continue;
}
let nodes = rows
.into_iter()
.enumerate()
.map(|(row_index, row)| {
let identifier = (row_index + 1).to_string();
let path = format!("{}.row_{}", chunk.path, identifier);
let (indent, indent_char) = detect_indent(source, row.start_byte);
let row_source = source.get(row.start_byte..row.end_byte).unwrap_or_default();
ChunkNode {
path,
identifier: Some(identifier),
kind: ChunkKind::Row,
leaf: true,
virtual_content: None,
parent_path: Some(chunk.path.clone()),
children: Vec::new(),
signature: Some(row.text.trim().to_owned()),
start_line: row.line,
end_line: row.line,
line_count: 1,
start_byte: row.start_byte as u32,
end_byte: row.end_byte as u32,
checksum_start_byte: row.start_byte as u32,
prologue_end_byte: None,
epilogue_start_byte: None,
checksum: chunk_checksum(row_source.as_bytes()),
error: false,
indent,
indent_char,
group: false,
}
})
.collect();
additions.push((index, nodes));
}
for (index, nodes) in additions {
let child_paths = nodes.iter().map(|node| node.path.clone()).collect();
chunks[index].leaf = false;
chunks[index].children = child_paths;
chunks.extend(nodes);
}
}
struct MarkdownTableRow<'a> {
line: u32,
start_byte: usize,
end_byte: usize,
text: &'a str,
}
fn markdown_table_rows_for_chunk<'a>(
source: &'a str,
chunk: &ChunkNode,
) -> Vec<MarkdownTableRow<'a>> {
let line_offsets = source_line_offsets(source);
let mut rows = Vec::new();
let mut saw_non_empty = false;
for line in chunk.start_line..=chunk.end_line {
let Some((start_byte, end_byte)) = line_bounds(source, &line_offsets, line) else {
continue;
};
let text = source
.get(start_byte..end_byte)
.unwrap_or_default()
.trim_end_matches('\n');
if text.trim().is_empty() {
continue;
}
saw_non_empty = true;
if !markdown_table_row(text) {
return Vec::new();
}
rows.push(MarkdownTableRow { line, start_byte, end_byte, text });
}
if saw_non_empty { rows } else { Vec::new() }
}
fn source_line_offsets(source: &str) -> Vec<usize> {
let mut offsets = vec![0usize];
for (index, ch) in source.char_indices() {
if ch == '\n' {
offsets.push(index + 1);
}
}
offsets
}
fn line_bounds(source: &str, offsets: &[usize], line: u32) -> Option<(usize, usize)> {
if line == 0 {
return None;
}
let start = *offsets.get((line - 1) as usize)?;
let end = offsets.get(line as usize).copied().unwrap_or(source.len());
Some((start, end))
}
fn markdown_table_row(line: &str) -> bool {
let trimmed = line.trim();
trimmed.starts_with('|') && trimmed.ends_with('|') && trimmed.matches('|').count() >= 2
}
fn markdown_table_separator_row(line: &str) -> bool {
let trimmed = line.trim();
trimmed.contains('-')
&& trimmed
.chars()
.all(|ch| matches!(ch, '|' | '-' | ':' | ' ' | '\t'))
}
/// Extract heading text from a Markdown `section` node's `atx_heading` or
/// `setext_heading` child.
fn extract_markdown_heading(node: Node<'_>, source: &str) -> Option<String> {
named_children(node)
.into_iter()
.find(|child| child.kind() == "atx_heading" || child.kind() == "setext_heading")
.and_then(|heading| {
sanitize_identifier(node_text(source, heading.start_byte(), heading.end_byte()))
})
}
/// Extract name from a Handlebars `block_statement` via its
/// `block_statement_start` child.
fn extract_glimmer_block_name(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["block_statement_start"]).and_then(|start| {
start
.child_by_field_name("path")
.or_else(|| child_by_kind(start, &["identifier"]))
.and_then(|name| {
sanitize_identifier(node_text(source, name.start_byte(), name.end_byte()))
})
})
}
fn fenced_code_language(node: Node<'_>, source: &str) -> Option<SupportLang> {
child_by_kind(node, &["info_string"])
.and_then(|info| child_by_kind(info, &["language"]))
.and_then(|lang| {
SupportLang::from_alias(node_text(source, lang.start_byte(), lang.end_byte()))
})
}
/// Extract name from a Handlebars `mustache_statement`:
/// tries `helper_invocation`'s helper field first, then direct
/// `identifier`/`path_expression`.
fn extract_glimmer_mustache_name(node: Node<'_>, source: &str) -> Option<String> {
let children = named_children(node);
for child in children {
if child.kind() == "helper_invocation"
&& let Some(helper) = child
.child_by_field_name("helper")
.or_else(|| child_by_kind(child, &["identifier", "path_expression"]))
{
return sanitize_identifier(node_text(source, helper.start_byte(), helper.end_byte()));
}
if matches!(child.kind(), "identifier" | "path_expression") {
return sanitize_identifier(node_text(source, child.start_byte(), child.end_byte()));
}
}
None
}
/// Extract tag name from an HTML-like element node.
///
/// Handles both standard HTML (`element` → `start_tag`/`self_closing_tag` →
/// `tag_name`) and Handlebars element nodes (`element_node` →
/// `element_node_start`/`element_node_void` → `tag_name`).
fn extract_element_tag_name(node: Node<'_>, source: &str) -> Option<String> {
// Handlebars element_node uses element_node_start / element_node_void
if node.kind() == "element_node" {
return named_children(node).into_iter().find_map(|child| {
if child.kind() == "element_node_start" || child.kind() == "element_node_void" {
child_by_kind(child, &["tag_name"]).and_then(|tag| {
sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte()))
})
} else {
None
}
});
}
// Standard HTML: element → start_tag / self_closing_tag → tag_name
named_children(node).into_iter().find_map(|child| {
if child.kind() == "start_tag" || child.kind() == "self_closing_tag" {
child_by_kind(child, &["tag_name"]).and_then(|tag| {
sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte()))
})
} else {
None
}
})
}
File diff suppressed because it is too large Load Diff
-228
View File
@@ -1,228 +0,0 @@
//! Language-specific chunk classifiers for Nix and HCL (Terraform).
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
common::*,
kind::ChunkKind,
};
pub struct NixHclClassifier;
const NIX_HCL_TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: StructuralOverrides {
extra_trivia: &[],
preserved_trivia: &[],
extra_root_wrappers: &["body"],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
/// Extract a structured name from an HCL `block` node.
///
/// Shape: `block_type label1 label2 … { body }` where labels are `string_lit`.
/// Returns e.g. `resource_aws_instance_web` for `resource "aws_instance" "web"
/// { … }`.
fn extract_hcl_block_name(node: Node<'_>, source: &str) -> Option<String> {
let mut children = named_children(node).into_iter();
let block_type = children.next()?;
let mut parts =
vec![node_text(source, block_type.start_byte(), block_type.end_byte()).to_string()];
for child in children {
if child.kind() == "string_lit" {
let text = unquote_text(node_text(source, child.start_byte(), child.end_byte()));
if !text.is_empty() {
parts.push(text);
}
continue;
}
if child.kind() == "body" || child.kind() == "block_end" || child.kind() == "block_start" {
continue;
}
let text = node_text(source, child.start_byte(), child.end_byte());
if !text.is_empty() {
parts.push(text.to_string());
}
}
sanitize_identifier(parts.join("_").as_str())
}
/// Extract the attrpath name from a Nix `binding` node.
fn extract_nix_binding_name(node: Node<'_>, source: &str) -> Option<String> {
node.child_by_field_name("attrpath").and_then(|attrpath| {
sanitize_identifier(node_text(source, attrpath.start_byte(), attrpath.end_byte()))
})
}
fn recurse_nix_attrset(node: Node<'_>) -> Option<RecurseSpec<'_>> {
recurse_into(node, ChunkContext::ClassBody, &[], &["binding_set"])
}
fn recurse_nix_binding_value(node: Node<'_>) -> Option<RecurseSpec<'_>> {
let expression = node.child_by_field_name("expression")?;
if matches!(
expression.kind(),
"attrset_expression" | "let_attrset_expression" | "rec_attrset_expression"
) {
return recurse_nix_attrset(expression);
}
recurse_value_container(node)
}
fn classify_nix_binding<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let name = extract_nix_binding_name(node, source).unwrap_or_else(|| "anonymous".to_string());
let expression = node.child_by_field_name("expression");
if let Some(expression) = expression
&& matches!(
expression.kind(),
"attrset_expression" | "let_attrset_expression" | "rec_attrset_expression"
) {
return make_container_chunk(
node,
ChunkKind::Attr,
Some(name),
source,
recurse_nix_attrset(expression),
);
}
make_kind_chunk(node, ChunkKind::Attr, Some(name), source, recurse_nix_binding_value(node))
}
impl LangClassifier for NixHclClassifier {
fn tables(&self) -> &'static ClassifierTables {
&NIX_HCL_TABLES
}
fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// Nix top-level attrsets should recurse into their binding_set so the file exposes
// structural attr chunks instead of a single opaque attrset_expr leaf.
"attrset_expression" | "let_attrset_expression" | "rec_attrset_expression" => {
Some(make_candidate(
node,
ChunkKind::Attrs,
None,
NameStyle::Named,
signature_for_node(node, source),
recurse_nix_attrset(node),
source,
))
},
// Older tree-sitter-nix revisions used `attribute`; current grammars expose `binding`.
"attribute" | "binding" => Some(classify_nix_binding(node, source)),
// HCL top-level block, or diff hunk fallback
"block" => {
if let Some(name) = extract_hcl_block_name(node, source) {
Some(make_container_chunk(
node,
ChunkKind::Block,
Some(name),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["body"]),
))
} else {
Some(group_candidate(node, ChunkKind::Hunks, source))
}
},
// Nix expressions
"function_expression" | "let_expression" => Some(named_candidate(
node,
ChunkKind::Expression,
source,
recurse_value_container(node),
)),
// Nix inherit
"inherit" => Some(group_candidate(node, ChunkKind::Imports, source)),
// Variable/assignment declarations
"variable_declaration" | "assignment" => {
Some(group_candidate(node, ChunkKind::Declarations, source))
},
// HCL top-level block types
"provider" | "resource" | "data" | "locals" | "variable" | "output" | "module" => {
let kind = match node.kind() {
"locals" => ChunkKind::BlockLocals,
"variable" => ChunkKind::Variable,
"module" => ChunkKind::Module,
_ => ChunkKind::Block,
};
Some(make_candidate(
node,
kind,
prefixed_name(sanitize_node_kind(node.kind()), node, source),
NameStyle::Named,
signature_for_node(node, source),
recurse_into(node, ChunkContext::ClassBody, &[], &["body"]),
source,
))
},
_ => None,
}
}
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// Nested HCL block — only promote if it has an identifiable block name
"block" => extract_hcl_block_name(node, source).map(|name| {
make_container_chunk(
node,
ChunkKind::Block,
Some(name),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["body"]),
)
}),
// Nested Nix attrset values recurse into their binding_set just like top-level ones.
"attrset_expression" | "let_attrset_expression" | "rec_attrset_expression" => {
Some(make_candidate(
node,
ChunkKind::Attrs,
None,
NameStyle::Named,
signature_for_node(node, source),
recurse_nix_attrset(node),
source,
))
},
// Nix binding_set is a transparent wrapper around individual bindings.
// Without this, the binding_set becomes an opaque leaf chunk hiding
// all bindings inside a single massive block.
"binding_set" => Some(make_candidate(
node,
ChunkKind::Attrs,
None,
NameStyle::Named,
None,
Some(RecurseSpec { node, context: ChunkContext::ClassBody }),
source,
)),
// Nix let_expression inside a class-body context — recurse into its
// binding_set so individual bindings are addressable.
"let_expression" => Some(make_candidate(
node,
ChunkKind::Expression,
None,
NameStyle::Named,
signature_for_node(node, source),
recurse_value_container(node),
source,
)),
// Nested Nix binding
"binding" => Some(classify_nix_binding(node, source)),
_ => None,
}
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// Nix control flow
"if_expression" => Some(positional_candidate(node, ChunkKind::If, source)),
"let_expression" => Some(positional_candidate(node, ChunkKind::Block, source)),
_ => None,
}
}
}
-236
View File
@@ -1,236 +0,0 @@
//! OCaml-specific chunk classifier.
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
common::*,
kind::ChunkKind,
};
pub struct OcamlClassifier;
impl LangClassifier for OcamlClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: StructuralOverrides::EMPTY,
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_ocaml_item(node, source),
ChunkContext::ClassBody => classify_class(node, source),
ChunkContext::FunctionBody => classify_function(node, source),
}
}
}
fn classify_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"method_definition" => Some(make_kind_chunk(
node,
ChunkKind::Function,
ocaml_named_text(node, source, &["method_name"]),
source,
ocaml_method_recurse(node),
)),
"method_specification" => Some(make_kind_chunk(
node,
ChunkKind::Function,
ocaml_named_text(node, source, &["method_name"]),
source,
None,
)),
"instance_variable_definition" => {
Some(match ocaml_named_text(node, source, &["instance_variable_name"]) {
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
None => group_candidate(node, ChunkKind::Fields, source),
})
},
_ => classify_ocaml_item(node, source),
}
}
fn classify_function<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"function_expression" | "match_expression" => Some(make_candidate(
node,
ChunkKind::Match,
None,
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::FunctionBody)),
source,
)),
"match_case" => Some(make_candidate(
node,
ChunkKind::Case,
None,
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::FunctionBody)),
source,
)),
"let_expression" => Some(make_candidate(
node,
ChunkKind::Let,
None,
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::FunctionBody)),
source,
)),
_ => classify_ocaml_item(node, source),
}
}
fn classify_ocaml_item<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"open_module" => group_candidate(node, ChunkKind::Imports, source),
"module_definition" => make_container_chunk(
node,
ChunkKind::Module,
ocaml_named_text(node, source, &["module_name"]),
source,
ocaml_module_recurse(node),
),
"module_type_definition" => make_candidate(
node,
ChunkKind::Interface,
format!("modtype_{}", ocaml_named_text(node, source, &["module_type_name"])?),
NameStyle::Named,
signature_for_node(node, source),
ocaml_module_type_recurse(node),
source,
),
"class_definition" => make_container_chunk(
node,
ChunkKind::Class,
ocaml_named_text(node, source, &["class_name"]),
source,
ocaml_class_recurse(node),
),
"class_type_definition" => make_candidate(
node,
ChunkKind::Iface,
format!("classtype_{}", ocaml_named_text(node, source, &["class_type_name"])?),
NameStyle::Named,
signature_for_node(node, source),
ocaml_class_type_recurse(node),
source,
),
"type_definition" => make_kind_chunk(
node,
ChunkKind::Type,
ocaml_named_text(node, source, &["type_constructor"]),
source,
None,
),
"exception_definition" => make_candidate(
node,
ChunkKind::Constructor,
format!("exception_{}", ocaml_named_text(node, source, &["constructor_name"])?),
NameStyle::Named,
signature_for_node(node, source),
None,
source,
),
"value_definition" => classify_ocaml_value_definition(node, source)?,
"value_specification" => make_kind_chunk(
node,
ChunkKind::Val,
ocaml_named_text(node, source, &["value_name"]),
source,
None,
),
_ => return None,
})
}
fn classify_ocaml_value_definition<'t>(
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
let name = ocaml_named_text(node, source, &["value_name"])?;
let recurse = ocaml_value_recurse(node);
if ocaml_value_definition_is_function(node) {
Some(make_kind_chunk(node, ChunkKind::Function, Some(name), source, recurse))
} else {
Some(make_kind_chunk(node, ChunkKind::Val, Some(name), source, recurse))
}
}
fn ocaml_named_text(node: Node<'_>, source: &str, kinds: &[&str]) -> Option<String> {
find_named_text(node, source, kinds).and_then(sanitize_identifier)
}
fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> {
if kinds.iter().any(|kind| node.kind() == *kind) {
return Some(node_text(source, node.start_byte(), node.end_byte()));
}
for child in named_children(node) {
if let Some(text) = find_named_text(child, source, kinds) {
return Some(text);
}
}
None
}
fn ocaml_module_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
recurse_into(node, ChunkContext::ClassBody, &[], &["module_binding"]).and_then(|binding| {
recurse_into(binding.node, ChunkContext::ClassBody, &["body"], &["structure", "signature"])
})
}
fn ocaml_module_type_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
recurse_into(node, ChunkContext::ClassBody, &["body"], &["signature"])
}
fn ocaml_class_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
recurse_into(node, ChunkContext::ClassBody, &[], &["class_binding"]).and_then(|binding| {
recurse_into(binding.node, ChunkContext::ClassBody, &["body"], &["object_expression"])
})
}
fn ocaml_class_type_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
recurse_into(node, ChunkContext::ClassBody, &[], &["class_type_binding"]).and_then(|binding| {
recurse_into(binding.node, ChunkContext::ClassBody, &["body"], &["class_body_type"])
})
}
fn ocaml_method_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
recurse_into(node, ChunkContext::FunctionBody, &["body"], &[
"function_expression",
"match_expression",
"let_expression",
])
}
fn ocaml_value_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
recurse_into(node, ChunkContext::FunctionBody, &[], &["let_binding"]).and_then(|binding| {
named_children(binding.node)
.into_iter()
.find(|child| {
matches!(child.kind(), "function_expression" | "match_expression" | "let_expression")
})
.map(|child| RecurseSpec { node: child, context: ChunkContext::FunctionBody })
})
}
fn ocaml_value_definition_is_function(node: Node<'_>) -> bool {
recurse_into(node, ChunkContext::FunctionBody, &[], &["let_binding"]).is_some_and(|binding| {
named_children(binding.node)
.into_iter()
.any(|child| matches!(child.kind(), "parameter" | "function_expression"))
})
}
-137
View File
@@ -1,137 +0,0 @@
//! Language-specific chunk classifier for Perl.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct PerlClassifier;
const PERL_SHARED_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"use_statement",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"conditional_statement",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"for_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"loop_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
];
const PERL_TABLES: ClassifierTables = ClassifierTables {
root: PERL_SHARED_RULES,
class: &[],
function: PERL_SHARED_RULES,
structural_overrides: StructuralOverrides {
extra_trivia: &[],
preserved_trivia: &[],
extra_root_wrappers: &["statement_list"],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
impl LangClassifier for PerlClassifier {
fn tables(&self) -> &'static ClassifierTables {
&PERL_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root | ChunkContext::FunctionBody => classify_perl_node(node, source),
ChunkContext::ClassBody => None,
}
}
}
fn classify_perl_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let body_recurse = || recurse_into(node, ChunkContext::FunctionBody, &["body"], &["block"]);
Some(match node.kind() {
"package_statement" => {
make_kind_chunk(node, ChunkKind::Module, Some(perl_name(node, source)?), source, None)
},
"subroutine_declaration_statement" => make_kind_chunk(
node,
ChunkKind::Function,
Some(perl_name(node, source)?),
source,
body_recurse(),
),
"expression_statement" => classify_perl_statement(node, source),
_ => return None,
})
}
fn classify_perl_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
if perl_declares_variable(node) {
group_candidate(node, ChunkKind::Declarations, source)
} else {
group_candidate(node, ChunkKind::Statements, source)
}
}
fn perl_declares_variable(node: Node<'_>) -> bool {
if node.kind() == "variable_declaration" {
return true;
}
if node.kind() == "assignment_expression"
&& named_children(node)
.into_iter()
.any(|child| child.kind() == "variable_declaration")
{
return true;
}
named_children(node).into_iter().any(perl_declares_variable)
}
fn perl_name(node: Node<'_>, source: &str) -> Option<String> {
find_named_text(node, source, &["bareword", "package", "varname"]).and_then(sanitize_identifier)
}
fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> {
if kinds.iter().any(|kind| node.kind() == *kind) {
return Some(node_text(source, node.start_byte(), node.end_byte()));
}
for child in named_children(node) {
if let Some(text) = find_named_text(child, source, kinds) {
return Some(text);
}
}
None
}
@@ -1,290 +0,0 @@
//! PowerShell-specific chunk classifier.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct PowershellClassifier;
static POWERSHELL_TABLES: ClassifierTables = ClassifierTables {
root: &[
semantic_rule(
"param_block",
ChunkKind::Parameters,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"flow_control_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
],
class: &[],
function: &[
semantic_rule(
"class_method_parameter_list",
ChunkKind::Parameters,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"param_block",
ChunkKind::Parameters,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"flow_control_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
],
structural_overrides: StructuralOverrides {
extra_trivia: &[
"function_name",
"simple_name",
"type_literal",
"switch_condition",
],
preserved_trivia: &[],
extra_root_wrappers: &[],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
impl LangClassifier for PowershellClassifier {
fn tables(&self) -> &'static ClassifierTables {
&POWERSHELL_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_root_custom(node, source),
ChunkContext::ClassBody => classify_class_custom(node, source),
ChunkContext::FunctionBody => classify_function_custom(node, source),
}
}
}
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"statement_list" => make_container_chunk(
node,
ChunkKind::Body,
None,
source,
Some(recurse_self(node, ChunkContext::Root)),
),
"class_statement" => make_container_chunk(
node,
ChunkKind::Class,
Some(powershell_name(node, source)?),
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
),
"function_statement" => make_container_chunk(
node,
ChunkKind::Function,
Some(powershell_name(node, source)?),
source,
Some(recurse_self(node, ChunkContext::FunctionBody)),
),
"pipeline" => classify_powershell_pipeline(node, source),
"switch_statement" | "if_statement" | "foreach_statement" => {
return classify_function_custom(node, source);
},
_ => return None,
})
}
fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"class_property_definition" => match powershell_name(node, source) {
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
None => group_candidate(node, ChunkKind::Fields, source),
},
"class_method_definition" => classify_class_method(node, source)?,
_ => return None,
})
}
fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"script_block" => make_container_chunk(
node,
block_kind_for_parent(node),
None,
source,
Some(recurse_self(node, ChunkContext::FunctionBody)),
),
"script_block_body" | "statement_block" => make_container_chunk(
node,
ChunkKind::Block,
None,
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_list"]),
),
"pipeline" => classify_powershell_pipeline(node, source),
"if_statement" => make_container_chunk(
node,
ChunkKind::If,
None,
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_block"]),
),
"foreach_statement" => make_container_chunk(
node,
ChunkKind::Loop,
None,
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_block"]),
),
"switch_statement" => make_container_chunk(
node,
ChunkKind::Switch,
None,
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["switch_body"]),
),
"switch_clauses" => make_container_chunk(
node,
ChunkKind::Cases,
None,
source,
Some(recurse_self(node, ChunkContext::FunctionBody)),
),
"switch_clause" => make_container_chunk(
node,
ChunkKind::Case,
None,
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["statement_block"]),
),
_ => return None,
})
}
fn classify_class_method<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let name = powershell_name(node, source)?;
let class_name = powershell_name(node.parent()?, source)?;
let (kind, identifier) = if name == "new" || name == class_name {
(ChunkKind::Constructor, None)
} else {
(ChunkKind::Function, Some(name))
};
Some(make_container_chunk(
node,
kind,
identifier,
source,
Some(recurse_self(node, ChunkContext::FunctionBody)),
))
}
fn classify_powershell_pipeline<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
if let Some(command_name) = powershell_command_name(node, source)
&& matches!(command_name.as_str(), "using" | "using-module" | "Import-Module")
{
return group_candidate(node, ChunkKind::Imports, source);
}
if let Some((name, script_block)) = assigned_script_block(node, source) {
return make_container_chunk_from(
node,
node,
ChunkKind::Block,
Some(name),
source,
Some(recurse_self(script_block, ChunkContext::FunctionBody)),
);
}
if child_by_kind(node, &["assignment_expression"]).is_some() {
group_candidate(node, ChunkKind::Declarations, source)
} else {
group_candidate(node, ChunkKind::Statements, source)
}
}
fn assigned_script_block<'t>(node: Node<'t>, source: &str) -> Option<(String, Node<'t>)> {
let assignment = child_by_kind(node, &["assignment_expression"])?;
let lhs = child_by_kind(assignment, &["left_assignment_expression"])?;
let name = sanitize_identifier(
node_text(source, lhs.start_byte(), lhs.end_byte()).trim_start_matches('$'),
)?;
let script_block = named_children(assignment)
.into_iter()
.filter(|child| child.kind() != "left_assignment_expression")
.find_map(find_script_block)?;
Some((name, script_block))
}
fn find_script_block(node: Node<'_>) -> Option<Node<'_>> {
if node.kind() == "script_block" {
return Some(node);
}
for child in named_children(node) {
if let Some(script_block) = find_script_block(child) {
return Some(script_block);
}
}
None
}
fn block_kind_for_parent(node: Node<'_>) -> ChunkKind {
match node.parent().map(|parent| parent.kind()) {
Some("function_statement" | "class_method_definition") => ChunkKind::Body,
_ => ChunkKind::Block,
}
}
fn powershell_name(node: Node<'_>, source: &str) -> Option<String> {
find_named_text(node, source, &[
"function_name",
"simple_name",
"member_name",
"type_identifier",
"variable",
])
.and_then(|text| sanitize_identifier(text.trim_start_matches('$')))
}
fn powershell_command_name(node: Node<'_>, source: &str) -> Option<String> {
find_named_text(node, source, &["command_name"]).and_then(sanitize_identifier)
}
fn find_named_text<'a>(node: Node<'_>, source: &'a str, kinds: &[&str]) -> Option<&'a str> {
if kinds.iter().any(|kind| node.kind() == *kind) {
return Some(node_text(source, node.start_byte(), node.end_byte()));
}
for child in named_children(node) {
if let Some(text) = find_named_text(child, source, kinds) {
return Some(text);
}
}
None
}
-193
View File
@@ -1,193 +0,0 @@
//! Chunk classifier for Protocol Buffers.
//!
//! Mirror the grammar's declaration structure directly: the root owns headers,
//! imports, options, messages, enums, and services; message bodies own fields,
//! oneofs, and nested messages/enums; services own rpc declarations and service
//! options; rpc blocks may contain rpc-scoped options.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct ProtoClassifier;
const PROTO_ROOT_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"syntax",
ChunkKind::Headers,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"package",
ChunkKind::Headers,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"import",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"option",
ChunkKind::Options,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const PROTO_CLASS_RULES: &[super::classify::SemanticRule] = &[semantic_rule(
"option",
ChunkKind::Options,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
)];
const PROTO_TABLES: ClassifierTables = ClassifierTables {
root: PROTO_ROOT_RULES,
class: PROTO_CLASS_RULES,
function: &[],
structural_overrides: super::classify::StructuralOverrides::EMPTY,
};
impl LangClassifier for ProtoClassifier {
fn tables(&self) -> &'static ClassifierTables {
&PROTO_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_proto_root(node, source),
ChunkContext::ClassBody => classify_proto_class(node, source),
ChunkContext::FunctionBody => None,
}
}
}
fn classify_proto_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"message" => make_named_proto_chunk(
node,
ChunkKind::Type,
format!("msg_{}", proto_name(node, source)?),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["message_body"]),
),
"enum" => make_container_chunk(
node,
ChunkKind::Enum,
proto_name(node, source),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["enum_body"]),
),
"service" => make_named_proto_chunk(
node,
ChunkKind::Interface,
format!("service_{}", proto_name(node, source)?),
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
),
_ => return None,
})
}
fn classify_proto_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"field" if is_proto_message_field(node) => {
make_kind_chunk(node, ChunkKind::Field, proto_name(node, source), source, None)
},
"oneof" => make_named_proto_chunk(
node,
ChunkKind::Either,
format!("oneof_{}", proto_name(node, source)?),
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
),
"oneof_field" => {
make_kind_chunk(node, ChunkKind::Field, proto_name(node, source), source, None)
},
"message" => make_named_proto_chunk(
node,
ChunkKind::Type,
format!("msg_{}", proto_name(node, source)?),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["message_body"]),
),
"enum" => make_container_chunk(
node,
ChunkKind::Enum,
proto_name(node, source),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["enum_body"]),
),
"enum_field" => {
make_kind_chunk(node, ChunkKind::Variant, proto_name(node, source), source, None)
},
"rpc" => make_named_proto_chunk(
node,
ChunkKind::Proc,
format!("rpc_{}", proto_name(node, source)?),
source,
proto_rpc_recurse(node),
),
_ => return None,
})
}
fn make_named_proto_chunk<'t>(
node: Node<'t>,
kind: ChunkKind,
identifier: impl Into<Option<String>>,
source: &str,
recurse: Option<RecurseSpec<'t>>,
) -> RawChunkCandidate<'t> {
make_candidate(
node,
kind,
identifier,
NameStyle::Named,
signature_for_node(node, source),
recurse,
source,
)
}
fn is_proto_message_field(node: Node<'_>) -> bool {
node
.parent()
.is_some_and(|parent| parent.kind() == "message_body")
}
fn proto_rpc_recurse(node: Node<'_>) -> Option<RecurseSpec<'_>> {
let has_nested_option = named_children(node)
.into_iter()
.any(|child| child.kind() == "option");
if has_nested_option {
Some(recurse_self(node, ChunkContext::ClassBody))
} else {
None
}
}
fn proto_name(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["message_name", "enum_name", "service_name", "rpc_name", "identifier"])
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
}
-244
View File
@@ -1,244 +0,0 @@
//! Language-specific chunk classifiers for Python and Starlark.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, WrapperSignature,
WrapperTransform, promote_wrapper_candidate, semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct PythonClassifier;
const ROOT_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"import_statement",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"import_from_statement",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"assignment",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"function_definition",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"class_definition",
ChunkKind::Class,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"if_statement",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"for_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"while_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"try_statement",
ChunkKind::Try,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"with_statement",
ChunkKind::Block,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"expression_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"global_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const CLASS_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"expression_statement",
ChunkKind::Fields,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"assignment",
ChunkKind::Fields,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"type_alias_statement",
ChunkKind::Type,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::None,
),
];
const FUNCTION_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"if_statement",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"for_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"while_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"try_statement",
ChunkKind::Try,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"with_statement",
ChunkKind::Block,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"elif_clause",
ChunkKind::Elif,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"except_clause",
ChunkKind::Except,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"match_statement",
ChunkKind::Match,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
];
const PYTHON_TABLES: ClassifierTables = ClassifierTables {
root: ROOT_RULES,
class: CLASS_RULES,
function: FUNCTION_RULES,
structural_overrides: super::classify::StructuralOverrides::EMPTY,
};
impl LangClassifier for PythonClassifier {
fn tables(&self) -> &'static ClassifierTables {
&PYTHON_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root | ChunkContext::ClassBody if node.kind() == "decorated_definition" => {
promote_wrapper_candidate(self, context, node, source, WrapperTransform {
signature: WrapperSignature::Wrapper,
..WrapperTransform::default()
})
.or_else(|| Some(positional_candidate(node, ChunkKind::Block, source)))
},
ChunkContext::ClassBody if node.kind() == "function_definition" => {
Some(classify_class_method(node, source))
},
_ => None,
}
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let _ = source;
Some(group_candidate(node, ChunkKind::Statements, source))
}
}
fn classify_class_method<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
let kind = if name == "__init__" || name == "__new__" {
ChunkKind::Constructor
} else {
ChunkKind::Function
};
let identifier = if kind == ChunkKind::Constructor {
None
} else {
Some(name)
};
make_kind_chunk(
node,
kind,
identifier,
source,
resolve_recurse(node, ChunkContext::FunctionBody),
)
}
-173
View File
@@ -1,173 +0,0 @@
//! R-specific chunk classifier.
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier},
common::*,
kind::ChunkKind,
};
pub struct RClassifier;
static R_TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: super::classify::StructuralOverrides::EMPTY,
};
impl LangClassifier for RClassifier {
fn tables(&self) -> &'static ClassifierTables {
&R_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_root_custom(node, source),
ChunkContext::FunctionBody => classify_function_custom(node, source),
_ => None,
}
}
}
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
// ── Imports ──
"call" if is_import_call(node, source) => group_candidate(node, ChunkKind::Imports, source),
"call" => group_candidate(node, ChunkKind::Statements, source),
// ── Function / value assignments ──
"binary_operator" => classify_assignment(node, source, ChunkScope::Root)?,
// ── Control flow at script scope ──
"if_statement" => control_candidate(node, ChunkKind::If, source, recurse_if(node)),
"for_statement" | "while_statement" | "repeat_statement" => {
control_candidate(node, ChunkKind::Loop, source, recurse_loop(node))
},
// ── Bare expressions ──
"identifier" | "subset" | "subset2" | "extract_operator" => {
group_candidate(node, ChunkKind::Statements, source)
},
_ => return None,
})
}
fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
// ── Local assignments ──
"binary_operator" => classify_assignment(node, source, ChunkScope::Function)?,
// ── Control flow ──
"if_statement" => control_candidate(node, ChunkKind::If, source, recurse_if(node)),
"for_statement" | "while_statement" | "repeat_statement" => {
control_candidate(node, ChunkKind::Loop, source, recurse_loop(node))
},
// ── Calls / bare expressions ──
"call" | "identifier" | "subset" | "subset2" | "extract_operator" | "break" | "next"
| "return" => group_candidate(node, ChunkKind::Statements, source),
_ => return None,
})
}
#[derive(Clone, Copy)]
enum ChunkScope {
Root,
Function,
}
fn classify_assignment<'t>(
node: Node<'t>,
source: &str,
scope: ChunkScope,
) -> Option<RawChunkCandidate<'t>> {
let (lhs, rhs) = assignment_sides(node, source)?;
if rhs.kind() == "function_definition" {
let name = simple_lhs_name(lhs, source).unwrap_or_else(|| "anonymous".to_string());
return Some(make_kind_chunk_from(
node,
rhs,
ChunkKind::Function,
Some(name),
source,
recurse_body(rhs, ChunkContext::FunctionBody),
));
}
match (scope, simple_lhs_name(lhs, source)) {
(ChunkScope::Root, Some(name)) => {
Some(make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None))
},
(ChunkScope::Function, Some(name)) if spans_multiple_lines(node) => {
Some(make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None))
},
_ => Some(group_candidate(
node,
match scope {
ChunkScope::Root => ChunkKind::Declarations,
ChunkScope::Function => ChunkKind::Statements,
},
source,
)),
}
}
fn assignment_sides<'t>(node: Node<'t>, source: &str) -> Option<(Node<'t>, Node<'t>)> {
if node.kind() != "binary_operator" {
return None;
}
let operator = node.child_by_field_name("operator")?;
let operator_text = node_text(source, operator.start_byte(), operator.end_byte());
if !matches!(operator_text, "<-" | "<<-" | "=") {
return None;
}
Some((node.child_by_field_name("lhs")?, node.child_by_field_name("rhs")?))
}
fn simple_lhs_name(lhs: Node<'_>, source: &str) -> Option<String> {
(lhs.kind() == "identifier")
.then(|| extract_identifier(lhs, source))
.flatten()
}
fn is_import_call(node: Node<'_>, source: &str) -> bool {
matches!(
extract_identifier(node, source).as_deref(),
Some("library" | "require" | "requireNamespace" | "source")
)
}
fn recurse_if(node: Node<'_>) -> Option<RecurseSpec<'_>> {
recurse_into(node, ChunkContext::FunctionBody, &["consequence", "alternative"], &[
"braced_expression",
])
}
fn recurse_loop(node: Node<'_>) -> Option<RecurseSpec<'_>> {
recurse_into(node, ChunkContext::FunctionBody, &["body"], &["braced_expression"])
}
fn control_candidate<'t>(
node: Node<'t>,
kind: ChunkKind,
source: &str,
recurse: Option<RecurseSpec<'t>>,
) -> RawChunkCandidate<'t> {
make_candidate(node, kind, None, NameStyle::Named, None, recurse, source)
}
fn spans_multiple_lines(node: Node<'_>) -> bool {
node.start_position().row != node.end_position().row
}
-253
View File
@@ -1,253 +0,0 @@
//! Language-specific chunk classifiers for Ruby and Lua.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct RubyLuaClassifier;
const RUBY_LUA_ROOT_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"function_definition",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"method",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"singleton_method",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"class",
ChunkKind::Class,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"module",
ChunkKind::Module,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"if_statement",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"unless",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"while_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"for_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"assignment",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"expression_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"function_call",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const RUBY_LUA_CLASS_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"class",
ChunkKind::Class,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"module",
ChunkKind::Module,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"assignment",
ChunkKind::Fields,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"call",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"command",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"identifier",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const RUBY_LUA_FUNCTION_RULES: &[super::classify::SemanticRule] = &[
semantic_rule(
"if_statement",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"unless",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"case_statement",
ChunkKind::Switch,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"case_match",
ChunkKind::Switch,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"while_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"for_statement",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"assignment",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const RUBY_LUA_TABLES: ClassifierTables = ClassifierTables {
root: RUBY_LUA_ROOT_RULES,
class: RUBY_LUA_CLASS_RULES,
function: RUBY_LUA_FUNCTION_RULES,
structural_overrides: StructuralOverrides {
extra_trivia: &[],
preserved_trivia: &[],
extra_root_wrappers: &[],
preserved_root_wrappers: &["module"],
absorbable_attrs: &[],
},
};
impl LangClassifier for RubyLuaClassifier {
fn tables(&self) -> &'static ClassifierTables {
&RUBY_LUA_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match (context, node.kind()) {
(ChunkContext::Root, "command" | "call") => {
let target = extract_identifier(node, source);
Some(match target.as_deref() {
Some("require" | "require_relative" | "load" | "autoload") => {
group_candidate(node, ChunkKind::Imports, source)
},
_ => group_candidate(node, ChunkKind::Statements, source),
})
},
(ChunkContext::ClassBody, "method" | "singleton_method") => {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
let kind = if name == "initialize" {
ChunkKind::Constructor
} else {
ChunkKind::Function
};
let identifier = (kind != ChunkKind::Constructor).then_some(name);
Some(make_kind_chunk(
node,
kind,
identifier,
source,
resolve_recurse(node, ChunkContext::FunctionBody),
))
},
_ => None,
}
}
}
-404
View File
@@ -1,404 +0,0 @@
//! Rust-specific chunk classifier.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::*,
kind::ChunkKind,
};
pub struct RustClassifier;
const ROOT_RULES: &[super::classify::SemanticRule] = &[
// ── Imports ──
semantic_rule(
"use_declaration",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"extern_crate_declaration",
ChunkKind::Imports,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
// ── Functions ──
semantic_rule(
"function_item",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"function_definition",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
// ── Containers ──
semantic_rule(
"struct_item",
ChunkKind::Struct,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"enum_item",
ChunkKind::Enum,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"trait_item",
ChunkKind::Trait,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"mod_item",
ChunkKind::Module,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
semantic_rule(
"foreign_block",
ChunkKind::Module,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
// ── Types ──
semantic_rule(
"type_item",
ChunkKind::Type,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::ClassBody),
),
// ── Macros ──
semantic_rule(
"macro_definition",
ChunkKind::Macro,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"macro_rule",
ChunkKind::Macro,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
// ── Statics / consts ──
semantic_rule(
"static_item",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"const_item",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
// ── Attributes ──
semantic_rule(
"inner_attribute_item",
ChunkKind::Attrs,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
// ── Expression statements ──
semantic_rule(
"expression_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const CLASS_RULES: &[super::classify::SemanticRule] = &[
// ── Functions (methods in impl/trait) ──
semantic_rule(
"function_item",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
semantic_rule(
"function_definition",
ChunkKind::Function,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::Auto(ChunkContext::FunctionBody),
),
// ── Types ──
semantic_rule(
"type_item",
ChunkKind::Type,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::None,
),
semantic_rule(
"type_alias",
ChunkKind::Type,
RuleStyle::Named,
NamingMode::AutoIdentifier,
RecurseMode::None,
),
// ── Consts / macros in class body ──
semantic_rule(
"const_item",
ChunkKind::Fields,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"macro_invocation",
ChunkKind::Fields,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const FUNCTION_RULES: &[super::classify::SemanticRule] = &[
// ── Control flow ──
semantic_rule(
"match_expression",
ChunkKind::Match,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"loop_expression",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"while_expression",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"for_expression",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
// ── Expression statements ──
semantic_rule(
"expression_statement",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
];
const RUST_TABLES: ClassifierTables = ClassifierTables {
root: ROOT_RULES,
class: CLASS_RULES,
function: FUNCTION_RULES,
structural_overrides: StructuralOverrides::EMPTY,
};
impl LangClassifier for RustClassifier {
fn tables(&self) -> &'static ClassifierTables {
&RUST_TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_root_custom(node, source),
ChunkContext::ClassBody => classify_class_custom(node, source),
ChunkContext::FunctionBody => classify_function_custom(node, source),
}
}
}
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Impl blocks (custom name extraction) ──
"impl_item" => {
let name = extract_impl_name(node, source).unwrap_or_else(|| "anonymous".to_string());
Some(make_container_chunk(
node,
ChunkKind::Impl,
Some(name),
source,
recurse_into(node, ChunkContext::ClassBody, &["body"], &["declaration_list"]),
))
},
// ── Variables (conditional auto-id vs group) ──
"let_declaration" => Some(match extract_identifier(node, source) {
Some(name) => make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None),
None => group_candidate(node, ChunkKind::Declarations, source),
}),
_ => None,
}
}
fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Fields (conditional auto-id vs group) ──
"field_declaration" => Some(match extract_identifier(node, source) {
Some(name) => make_kind_chunk(node, ChunkKind::Field, Some(name), source, None),
None => group_candidate(node, ChunkKind::Fields, source),
}),
// ── Enum variants (conditional auto-id vs group) ──
"enum_variant" => Some(match extract_identifier(node, source) {
Some(name) => make_kind_chunk(node, ChunkKind::Variant, Some(name), source, None),
None => group_candidate(node, ChunkKind::Variants, source),
}),
// ── Attributes (explicitly return None — absorbed by framework) ──
"attribute_item" => None,
_ => None,
}
}
fn classify_function_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
match node.kind() {
// ── Control flow with recurse ──
"if_expression" => Some(make_candidate(
node,
ChunkKind::If,
None,
NameStyle::Named,
None,
fn_recurse(),
source,
)),
// ── Blocks ──
"unsafe_block" | "async_block" | "const_block" | "block_expression" => Some(make_candidate(
node,
ChunkKind::Block,
None,
NameStyle::Named,
None,
fn_recurse(),
source,
)),
// ── Variables (conditional line span) ──
"let_declaration" => {
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
Some(if span > 1 {
match extract_identifier(node, source) {
Some(name) => make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None),
None => group_candidate(node, ChunkKind::Let, source),
}
} else {
group_candidate(node, ChunkKind::Let, source)
})
},
_ => None,
}
}
/// Extract the name for an `impl` block.
///
/// - Plain impl: `impl Foo` → `"Foo"`
/// - Trait impl: `impl Trait for Foo` → `"Trait_for_Foo"`
/// - Scoped trait: `impl fmt::Display for Foo` → `"Display_for_Foo"`
fn extract_impl_name(node: Node<'_>, source: &str) -> Option<String> {
// Collect ALL children (including anonymous keywords like `for`).
let all_children: Vec<Node<'_>> = (0..node.child_count())
.filter_map(|i| node.child(i))
.collect();
// Find the `for` keyword position.
let for_index = all_children
.iter()
.position(|c| node_text(source, c.start_byte(), c.end_byte()) == "for");
if let Some(fi) = for_index {
// Trait impl: trait name before `for`, type name after `for`.
let trait_node = all_children[..fi].iter().rev().find(|c| {
matches!(c.kind(), "type_identifier" | "scoped_type_identifier" | "generic_type")
});
let type_node = all_children[fi + 1..].iter().find(|c| {
matches!(c.kind(), "type_identifier" | "scoped_type_identifier" | "generic_type")
});
if let (Some(tn), Some(ty)) = (trait_node, type_node) {
let trait_name = extract_last_type_identifier(*tn, source)
.or_else(|| sanitize_identifier(node_text(source, tn.start_byte(), tn.end_byte())))?;
let type_name = extract_last_type_identifier(*ty, source)
.or_else(|| sanitize_identifier(node_text(source, ty.start_byte(), ty.end_byte())))?;
return Some(format!("{trait_name}_for_{type_name}"));
}
}
// Plain impl: take the last type_identifier.
let type_ids: Vec<Node<'_>> = named_children(node)
.into_iter()
.filter(|c| c.kind() == "type_identifier")
.collect();
type_ids
.last()
.and_then(|n| sanitize_identifier(node_text(source, n.start_byte(), n.end_byte())))
}
/// Recursively find the innermost `type_identifier` from a type node.
///
/// Handles scoped types like `fmt::Display` by traversing into
/// `scoped_type_identifier` and `generic_type` children.
fn extract_last_type_identifier(node: Node<'_>, source: &str) -> Option<String> {
if node.kind() == "type_identifier" {
return sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()));
}
let mut result = None;
for child in named_children(node) {
if child.kind() == "type_identifier" {
result = sanitize_identifier(node_text(source, child.start_byte(), child.end_byte()));
} else if matches!(child.kind(), "scoped_type_identifier" | "generic_type")
&& let Some(inner) = extract_last_type_identifier(child, source)
{
result = Some(inner);
}
}
result
}
-282
View File
@@ -1,282 +0,0 @@
//! SQL-specific chunk classifier.
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
common::*,
kind::ChunkKind,
};
pub struct SqlClassifier;
impl LangClassifier for SqlClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: StructuralOverrides {
extra_trivia: &["empty_statement", "dollar_quote", "keyword_from"],
preserved_trivia: &[],
extra_root_wrappers: &[],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_sql_root(node, source),
ChunkContext::ClassBody => classify_sql_class(node, source),
ChunkContext::FunctionBody => classify_sql_function(node, source),
}
}
}
fn classify_sql_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
if node.kind() == "statement" {
return classify_sql_statement_root(node, source);
}
classify_sql_root_node(node, node, source).or_else(|| classify_sql_query_node(node, source))
}
fn classify_sql_statement_root<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let children = named_children(node);
if children.len() == 1 {
return classify_sql_root_node(node, children[0], source)
.or_else(|| classify_sql_query_node(children[0], source));
}
if children.iter().any(|child| is_sql_query_kind(child.kind())) {
return Some(make_named_sql_chunk(
node,
ChunkKind::Query,
None,
source,
Some(recurse_self(node, ChunkContext::FunctionBody)),
));
}
None
}
fn classify_sql_root_node<'t>(
range_node: Node<'t>,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"create_schema" => make_kind_chunk_from(
range_node,
node,
ChunkKind::Schema,
extract_sql_identifier(node, source),
source,
None,
),
"create_table" => make_container_chunk_from(
range_node,
node,
ChunkKind::Table,
extract_sql_object_name(node, source),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["column_definitions"]),
),
"create_view" => make_named_sql_chunk_from(
range_node,
node,
ChunkKind::Query,
format!(
"view_{}",
extract_sql_object_name(node, source).unwrap_or_else(|| "anonymous".to_string())
),
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["create_query"]),
),
"create_materialized_view" => make_named_sql_chunk_from(
range_node,
node,
ChunkKind::Query,
format!(
"matview_{}",
extract_sql_object_name(node, source).unwrap_or_else(|| "anonymous".to_string())
),
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["create_query"]),
),
"create_function" => make_container_chunk_from(
range_node,
node,
ChunkKind::Function,
extract_sql_object_name(node, source),
source,
recurse_sql_function_query(node),
),
"create_trigger" => make_named_sql_chunk_from(
range_node,
node,
ChunkKind::Function,
format!(
"trigger_{}",
extract_sql_object_name(node, source).unwrap_or_else(|| "anonymous".to_string())
),
source,
None,
),
"create_index" => make_named_sql_chunk_from(
range_node,
node,
ChunkKind::Key,
format!(
"index_{}",
extract_sql_identifier(node, source).unwrap_or_else(|| "anonymous".to_string())
),
source,
None,
),
_ => return None,
})
}
fn classify_sql_class<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"column_definition" => {
make_kind_chunk(node, ChunkKind::Field, extract_sql_identifier(node, source), source, None)
},
_ => return None,
})
}
fn classify_sql_function<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
if node.kind() == "statement" {
return Some(make_named_sql_chunk(
node,
ChunkKind::Query,
None,
source,
Some(recurse_self(node, ChunkContext::FunctionBody)),
));
}
classify_sql_query_node(node, source)
}
fn classify_sql_query_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
"insert" => group_candidate(node, ChunkKind::Statements, source),
"keyword_with" => group_candidate(node, ChunkKind::With, source),
"cte" => make_named_sql_chunk(
node,
ChunkKind::With,
format!(
"cte_{}",
extract_sql_identifier(node, source).unwrap_or_else(|| "anonymous".to_string())
),
source,
recurse_into(node, ChunkContext::FunctionBody, &[], &["statement"]),
),
"select" => positional_candidate(node, ChunkKind::Select, source),
"from" => make_named_sql_chunk(
node,
ChunkKind::Query,
"from".to_string(),
source,
Some(recurse_self(node, ChunkContext::FunctionBody)),
),
"relation" => group_candidate(node, ChunkKind::Relations, source),
"join" => positional_candidate(node, ChunkKind::Join, source),
"where" => positional_candidate(node, ChunkKind::Where, source),
"group_by" => positional_candidate(node, ChunkKind::GroupBy, source),
"order_by" => positional_candidate(node, ChunkKind::OrderBy, source),
_ => return None,
})
}
fn make_named_sql_chunk<'t>(
node: Node<'t>,
kind: ChunkKind,
identifier: impl Into<Option<String>>,
source: &str,
recurse: Option<RecurseSpec<'t>>,
) -> RawChunkCandidate<'t> {
make_candidate(
node,
kind,
identifier,
NameStyle::Named,
signature_for_node(node, source),
recurse,
source,
)
}
fn make_named_sql_chunk_from<'t>(
range_node: Node<'t>,
signature_node: Node<'t>,
kind: ChunkKind,
identifier: impl Into<Option<String>>,
source: &str,
recurse: Option<RecurseSpec<'t>>,
) -> RawChunkCandidate<'t> {
make_candidate(
range_node,
kind,
identifier,
NameStyle::Named,
signature_for_node(signature_node, source),
recurse,
source,
)
}
fn recurse_sql_function_query(node: Node<'_>) -> Option<RecurseSpec<'_>> {
let body = child_by_kind(node, &["function_body"])?;
recurse_into(body, ChunkContext::FunctionBody, &[], &["statement"])
}
fn is_sql_query_kind(kind: &str) -> bool {
matches!(
kind,
"insert"
| "keyword_with"
| "cte"
| "select"
| "from"
| "where"
| "group_by"
| "order_by"
| "join"
)
}
fn extract_sql_identifier(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["identifier"])
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
}
fn extract_sql_object_name(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["object_reference"]).and_then(|name| last_identifier(name, source))
}
fn last_identifier(node: Node<'_>, source: &str) -> Option<String> {
if node.kind() == "identifier" {
return sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()));
}
for child in named_children(node).into_iter().rev() {
if let Some(identifier) = last_identifier(child, source) {
return Some(identifier);
}
}
None
}
-299
View File
@@ -1,299 +0,0 @@
//! Language-specific chunk classifier for Svelte.
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
common::*,
kind::ChunkKind,
};
use crate::language::SupportLang;
pub struct SvelteClassifier;
impl LangClassifier for SvelteClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: StructuralOverrides {
extra_trivia: &[],
preserved_trivia: &[],
extra_root_wrappers: &["document"],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
let include_plain_elements = matches!(context, ChunkContext::Root);
classify_svelte_node(node, source, include_plain_elements)
}
}
fn classify_svelte_node<'t>(
node: Node<'t>,
source: &str,
include_plain_elements: bool,
) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"script_element" => Some(classify_script_element(node, source)),
"style_element" => Some(classify_style_element(node, source)),
"snippet_statement" => Some(classify_snippet_statement(node, source)),
"if_statement" => Some(classify_if_statement(node, source)),
"else_if_statement" => Some(classify_else_if_statement(node, source)),
"else_statement" => Some(classify_else_statement(node, source)),
"each_statement" => Some(classify_each_statement(node, source)),
"await_statement" => Some(classify_await_statement(node, source)),
"then_statement" => Some(classify_then_statement(node, source)),
"catch_statement" => Some(classify_catch_statement(node, source)),
"render_expr" => Some(classify_render_expr(node, source)),
"html_interpolation" => Some(group_candidate(node, ChunkKind::Html, source)),
"interpolation" => Some(group_candidate(node, ChunkKind::Interpolation, source)),
"expression" => Some(group_candidate(node, ChunkKind::Expression, source)),
"element" if include_plain_elements || element_has_structure(node) => {
classify_element(node, source)
},
_ => None,
}
}
fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let kind = if has_attribute(node, "module", source)
|| attribute_value(node, "context", source).as_deref() == Some("module")
{
ChunkKind::ScriptModule
} else {
ChunkKind::Script
};
classify_raw_text_block(node, kind, source, SupportLang::TypeScript)
}
fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let kind = if has_attribute(node, "scoped", source) {
ChunkKind::StyleScoped
} else {
ChunkKind::Style
};
classify_raw_text_block(node, kind, source, SupportLang::Css)
}
fn classify_snippet_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let identifier = child_by_kind(node, &["snippet_start_expr"])
.and_then(|start| child_by_kind(start, &["snippet_name"]))
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())));
force_container(make_container_chunk(
node,
ChunkKind::Snippet,
identifier,
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
))
}
fn classify_if_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let identifier = block_expr_identifier(node, source, "if_start_expr", &["raw_text_expr"]);
force_container(make_container_chunk(
node,
ChunkKind::If,
identifier,
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
))
}
fn classify_else_if_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let identifier = block_expr_identifier(node, source, "else_if_expr", &["raw_text_expr"])
.map_or_else(|| "if".to_string(), |expr| format!("if_{expr}"));
make_named_container_chunk(node, ChunkKind::Else, identifier, source)
}
fn classify_else_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
force_container(make_container_chunk(
node,
ChunkKind::Else,
None,
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
))
}
fn classify_each_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let expr = block_expr_identifier(node, source, "each_start_expr", &["raw_text_each"]);
let id = expr.map_or_else(|| "each".to_string(), |expr| format!("each_{expr}"));
make_named_container_chunk(node, ChunkKind::Loop, id, source)
}
fn classify_await_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let expr = block_expr_identifier(node, source, "await_start_expr", &["raw_text_expr"]);
let id = expr.map_or_else(|| "await".to_string(), |expr| format!("await_{expr}"));
make_named_container_chunk(node, ChunkKind::With, id, source)
}
fn classify_then_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let expr = block_expr_identifier(node, source, "then_expr", &["raw_text_expr"]);
let id = expr.map_or_else(|| "then".to_string(), |expr| format!("then_{expr}"));
make_named_container_chunk(node, ChunkKind::After, id, source)
}
fn classify_catch_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let identifier = block_expr_identifier(node, source, "catch_expr", &["raw_text_expr"]);
force_container(make_container_chunk(
node,
ChunkKind::Catch,
identifier,
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
))
}
fn classify_render_expr<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let identifier = child_by_kind(node, &["snippet_name"])
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())));
make_kind_chunk(node, ChunkKind::Render, identifier, source, None)
}
fn classify_element<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let tag_name = extract_markup_tag_name(node, source)?;
Some(force_container(make_container_chunk(
node,
ChunkKind::Tag,
Some(tag_name),
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
)))
}
fn make_named_container_chunk<'t>(
node: Node<'t>,
kind: ChunkKind,
identifier: impl Into<Option<String>>,
source: &str,
) -> RawChunkCandidate<'t> {
force_container(make_candidate(
node,
kind,
identifier,
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::ClassBody)),
source,
))
}
const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> {
candidate.force_recurse = true;
candidate
}
fn classify_raw_text_block<'t>(
node: Node<'t>,
kind: ChunkKind,
source: &str,
default_language: SupportLang,
) -> RawChunkCandidate<'t> {
let Some(content_node) = child_by_kind(node, &["raw_text"]) else {
return positional_candidate(node, kind, source);
};
let candidate = with_region_node(positional_candidate(node, kind, source), Some(content_node));
match resolve_embedded_language(node, source, default_language) {
Some(language) => with_injected_subtree(candidate, language, content_node),
None => candidate,
}
}
fn block_expr_identifier(
node: Node<'_>,
source: &str,
header_kind: &str,
expr_kinds: &[&str],
) -> Option<String> {
child_by_kind(node, &[header_kind])
.and_then(|header| child_by_kind(header, expr_kinds))
.and_then(|expr| sanitize_identifier(node_text(source, expr.start_byte(), expr.end_byte())))
}
fn element_has_structure(node: Node<'_>) -> bool {
named_children(node).into_iter().any(|child| {
matches!(
child.kind(),
"snippet_statement"
| "if_statement"
| "else_if_statement"
| "else_statement"
| "each_statement"
| "await_statement"
| "then_statement"
| "catch_statement"
| "render_expr"
| "html_interpolation"
| "interpolation"
| "expression"
| "element"
)
})
}
fn extract_markup_tag_name(node: Node<'_>, source: &str) -> Option<String> {
start_like(node)
.and_then(|start| child_by_kind(start, &["tag_name"]))
.and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte())))
}
fn has_attribute(node: Node<'_>, name: &str, source: &str) -> bool {
start_like(node)
.into_iter()
.flat_map(named_children)
.filter(|child| child.kind() == "attribute")
.filter_map(|attr| extract_attribute_name(attr, source))
.any(|attr_name| attr_name == name)
}
fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option<String> {
let start = start_like(node)?;
for child in named_children(start) {
if child.kind() != "attribute" {
continue;
}
if extract_attribute_name(child, source).as_deref() != Some(name) {
continue;
}
if let Some(value) = child_by_kind(child, &["attribute_value", "quoted_attribute_value"]) {
return sanitize_identifier(&unquote_text(node_text(
source,
value.start_byte(),
value.end_byte(),
)));
}
return Some(name.to_string());
}
None
}
fn resolve_embedded_language(
node: Node<'_>,
source: &str,
default_language: SupportLang,
) -> Option<SupportLang> {
if let Some(language) = attribute_value(node, "lang", source) {
return SupportLang::from_alias(language.as_str());
}
Some(default_language)
}
fn extract_attribute_name(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["attribute_name"])
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
}
fn start_like(node: Node<'_>) -> Option<Node<'_>> {
child_by_kind(node, &["start_tag", "self_closing_tag"])
}
-387
View File
@@ -1,387 +0,0 @@
//! Language-specific chunk classification for TLA+ / `PlusCal`.
//!
//! The shared defaults are too noisy for TLA+: the parser exposes a top-level
//! `module` wrapper, `PlusCal` algorithms live inside block comments, and the
//! generated translation section introduces operator definitions we do not want
//! to surface in chunked read/edit views.
use tree_sitter::Node;
use super::{
classify::{
ClassifierTables, LangClassifier, NamingMode, RecurseMode, RuleStyle, StructuralOverrides,
semantic_rule,
},
common::{
ChunkContext, RawChunkCandidate, RecurseSpec, child_by_kind, extract_identifier,
make_container_chunk, make_container_chunk_from, make_kind_chunk, recurse_self,
sanitize_identifier,
},
kind::ChunkKind,
types::ChunkNode,
};
pub struct TlaplusClassifier;
impl LangClassifier for TlaplusClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[
semantic_rule(
"variable_declaration",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"constant_declaration",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"recursive_declaration",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
],
class: &[semantic_rule(
"pcal_var_decls",
ChunkKind::Declarations,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
)],
function: &[
semantic_rule(
"pcal_if",
ChunkKind::If,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"pcal_while",
ChunkKind::Loop,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"pcal_either",
ChunkKind::Either,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"pcal_with",
ChunkKind::With,
RuleStyle::Positional,
NamingMode::None,
RecurseMode::None,
),
semantic_rule(
"pcal_assign",
ChunkKind::Statements,
RuleStyle::Group,
NamingMode::None,
RecurseMode::None,
),
],
structural_overrides: StructuralOverrides {
extra_trivia: &[
"header_line",
"double_line",
"extends",
"pcal_algorithm_start",
],
preserved_trivia: &["block_comment"],
extra_root_wrappers: &[],
preserved_root_wrappers: &["module"],
absorbable_attrs: &[],
},
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_root_custom(node, source),
ChunkContext::ClassBody => classify_class_custom(node, source),
_ => None,
}
}
fn preserve_children(
&self,
parent: &RawChunkCandidate<'_>,
_children: &[RawChunkCandidate<'_>],
) -> bool {
matches!(
parent.kind,
ChunkKind::Module | ChunkKind::Algo | ChunkKind::Proc | ChunkKind::Process
)
}
fn post_process(
&self,
chunks: &mut Vec<ChunkNode>,
root_children: &mut Vec<String>,
source: &str,
) {
let ranges = translation_ranges(source);
if ranges.is_empty() {
return;
}
let removed_by_range = ranges
.iter()
.map(|range| {
chunks
.iter()
.filter(|chunk| {
!chunk.path.is_empty() && chunk_wholly_inside_translation_fence(chunk, *range)
})
.cloned()
.collect::<Vec<_>>()
})
.collect::<Vec<_>>();
let removed_paths = removed_by_range
.iter()
.flatten()
.map(|chunk| chunk.path.clone())
.collect::<Vec<_>>();
if removed_paths.is_empty() {
return;
}
chunks.retain(|chunk| !removed_paths.iter().any(|removed| removed == &chunk.path));
for chunk in chunks.iter_mut() {
chunk
.children
.retain(|child| !removed_paths.iter().any(|removed| removed == child));
}
root_children.retain(|child| !removed_paths.iter().any(|removed| removed == child));
for (index, range) in ranges.iter().copied().enumerate() {
let removed_chunks = &removed_by_range[index];
if removed_chunks.is_empty() {
continue;
}
let synthetic = translation_chunk(range, removed_chunks[0].parent_path.clone(), source);
let synthetic_path = synthetic.path.clone();
if let Some(parent_path) = synthetic.parent_path.as_ref() {
if let Some(parent) = chunks.iter_mut().find(|chunk| chunk.path == *parent_path) {
parent.children.push(synthetic_path);
}
} else {
root_children.push(synthetic_path);
}
chunks.push(synthetic);
}
}
}
fn classify_root_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"module" => Some(make_container_chunk(
node,
ChunkKind::Module,
tla_identifier(node, source),
source,
Some(recurse_self(node, ChunkContext::Root)),
)),
"operator_definition" => Some(make_kind_chunk(
node,
ChunkKind::Operator,
tla_identifier(node, source),
source,
None,
)),
"module_definition" => Some(make_container_chunk(
node,
ChunkKind::Module,
tla_identifier(node, source),
source,
Some(recurse_self(node, ChunkContext::Root)),
)),
"pcal_algorithm" => Some(make_container_chunk(
node,
ChunkKind::Algo,
tla_identifier(node, source),
source,
recurse_child(node, "pcal_algorithm_body", ChunkContext::ClassBody),
)),
"block_comment" => child_by_kind(node, &["pcal_algorithm"]).map(|algorithm| {
make_container_chunk_from(
node,
algorithm,
ChunkKind::Algo,
tla_identifier(algorithm, source),
source,
recurse_child(algorithm, "pcal_algorithm_body", ChunkContext::ClassBody),
)
}),
_ => None,
}
}
fn classify_class_custom<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"pcal_procedure" => Some(make_container_chunk(
node,
ChunkKind::Proc,
tla_identifier(node, source),
source,
recurse_child(node, "pcal_algorithm_body", ChunkContext::ClassBody),
)),
"pcal_process" => Some(make_container_chunk(
node,
ChunkKind::Process,
tla_identifier(node, source),
source,
recurse_child(node, "pcal_algorithm_body", ChunkContext::ClassBody),
)),
_ => None,
}
}
fn tla_identifier(node: Node<'_>, source: &str) -> Option<String> {
extract_identifier(node, source).or_else(|| {
child_by_kind(node, &["identifier"])
.and_then(|child| sanitize_identifier(child.utf8_text(source.as_bytes()).ok()?))
})
}
fn recurse_child<'tree>(
node: Node<'tree>,
kind: &'static str,
context: ChunkContext,
) -> Option<RecurseSpec<'tree>> {
child_by_kind(node, &[kind]).map(|child| RecurseSpec { node: child, context })
}
#[derive(Clone, Copy)]
struct TranslationRange {
start_line: u32,
end_line: u32,
}
fn translation_ranges(source: &str) -> Vec<TranslationRange> {
let lines = source.split('\n').collect::<Vec<_>>();
let mut ranges = Vec::new();
let mut current_start: Option<u32> = None;
for (index, line) in lines.iter().enumerate() {
let line_no = index as u32 + 1;
let trimmed = line.trim();
if trimmed == r"\* BEGIN TRANSLATION" {
current_start = Some(line_no);
continue;
}
if trimmed == r"\* END TRANSLATION"
&& let Some(start_line) = current_start.take()
{
push_translation_range(&mut ranges, start_line, line_no, lines.len() as u32);
}
}
if let Some(start_line) = current_start {
push_translation_range(&mut ranges, start_line, lines.len() as u32, lines.len() as u32);
}
ranges
}
fn push_translation_range(
ranges: &mut Vec<TranslationRange>,
start_line: u32,
end_line: u32,
total_lines: u32,
) {
let clamped_end = end_line.min(total_lines.max(start_line));
ranges.push(TranslationRange { start_line, end_line: clamped_end });
}
/// True when the chunk's span lies entirely inside the `\* BEGIN` … `\* END`
/// translation fence. We must not treat broad containers (e.g. the `module`
/// chunk spanning the whole file) as translation-only, or the module node is
/// removed and children become orphaned.
const fn chunk_wholly_inside_translation_fence(chunk: &ChunkNode, range: TranslationRange) -> bool {
chunk.start_line >= range.start_line && chunk.end_line <= range.end_line
}
fn translation_chunk(
range: TranslationRange,
parent_path: Option<String>,
source: &str,
) -> ChunkNode {
let path = match &parent_path {
Some(parent) => format!("{parent}.translation_{}", range.start_line),
None => format!("translation_{}", range.start_line),
};
let (start_byte, end_byte) = byte_range_for_lines(source, range.start_line, range.end_line);
let checksum = super::chunk_checksum(&source.as_bytes()[start_byte as usize..end_byte as usize]);
ChunkNode {
path,
identifier: Some(range.start_line.to_string()),
kind: ChunkKind::Translation,
leaf: true,
virtual_content: None,
parent_path,
children: Vec::new(),
signature: Some("translation block".to_string()),
start_line: range.start_line,
end_line: range.end_line,
line_count: range.end_line.saturating_sub(range.start_line) + 1,
start_byte,
end_byte,
checksum_start_byte: start_byte,
prologue_end_byte: None,
epilogue_start_byte: None,
checksum,
error: false,
indent: 0,
indent_char: String::new(),
group: false,
}
}
fn byte_range_for_lines(source: &str, start_line: u32, end_line: u32) -> (u32, u32) {
let mut start_byte = 0usize;
let mut current_line = 1u32;
for (byte_index, byte) in source.bytes().enumerate() {
if current_line == start_line {
start_byte = byte_index;
break;
}
if byte == b'\n' {
current_line += 1;
start_byte = byte_index + 1;
}
}
let mut end_byte = source.len();
current_line = 1;
for (byte_index, byte) in source.bytes().enumerate() {
if current_line > end_line {
end_byte = byte_index;
break;
}
if byte == b'\n' {
current_line += 1;
}
}
(start_byte as u32, end_byte as u32)
}
-318
View File
@@ -1,318 +0,0 @@
//! Language-specific chunk classifier for Vue single-file components.
use tree_sitter::Node;
use super::{
classify::{ClassifierTables, LangClassifier, StructuralOverrides},
common::*,
kind::ChunkKind,
};
use crate::language::SupportLang;
pub struct VueClassifier;
impl LangClassifier for VueClassifier {
fn tables(&self) -> &'static ClassifierTables {
static TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: StructuralOverrides {
extra_trivia: &[],
preserved_trivia: &[],
extra_root_wrappers: &["document"],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
},
};
&TABLES
}
fn classify_override<'t>(
&self,
context: ChunkContext,
node: Node<'t>,
source: &str,
) -> Option<RawChunkCandidate<'t>> {
match context {
ChunkContext::Root => classify_root_node(node, source),
ChunkContext::ClassBody | ChunkContext::FunctionBody => classify_nested_node(node, source),
}
}
}
fn classify_root_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"template_element" => Some(classify_template_element(node, source)),
"script_element" => Some(classify_script_element(node, source)),
"style_element" => Some(classify_style_element(node, source)),
// Vue custom blocks (for example <i18n>) currently parse as plain `element`
// nodes at the document root, so infer custom-block semantics from position.
"element" => Some(classify_custom_block(node, source)),
_ => None,
}
}
fn classify_nested_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"template_element" => Some(classify_template_element(node, source)),
"element" => classify_element(node, source),
"start_tag" => classify_start_tag(node, source),
"directive_attribute" => Some(classify_directive_attribute(node, source)),
"attribute" => Some(classify_attribute(node, source)),
"interpolation" => Some(make_kind_chunk(node, ChunkKind::Expression, None, source, None)),
"text" => Some(group_candidate(node, ChunkKind::Text, source)),
"raw_text" => Some(group_candidate(node, ChunkKind::Text, source)),
_ => None,
}
}
fn classify_template_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let recurse = Some(recurse_self(node, ChunkContext::ClassBody));
if let Some(slot_name) = extract_slot_name(node, source) {
force_container(make_container_chunk(node, ChunkKind::Slot, Some(slot_name), source, recurse))
} else {
force_container(make_container_chunk(node, ChunkKind::Template, None, source, recurse))
}
}
fn classify_script_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let kind = if has_attribute(node, "setup", source) {
ChunkKind::ScriptSetup
} else if attribute_value(node, "context", source).as_deref() == Some("module") {
ChunkKind::ScriptModule
} else {
ChunkKind::Script
};
classify_raw_text_block(node, kind, source, SupportLang::JavaScript)
}
fn classify_style_element<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let kind = if has_attribute(node, "scoped", source) {
ChunkKind::StyleScoped
} else {
ChunkKind::Style
};
classify_raw_text_block(node, kind, source, SupportLang::Css)
}
fn classify_custom_block<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let tag_name = extract_markup_tag_name(node, source).unwrap_or_else(|| "anonymous".to_string());
make_named_container_chunk(node, ChunkKind::Custom, tag_name, source)
}
fn classify_element<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let tag_name = extract_markup_tag_name(node, source)?;
Some(force_container(make_container_chunk(
node,
ChunkKind::Tag,
Some(tag_name),
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
)))
}
fn classify_start_tag<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
if !named_children(node)
.into_iter()
.any(|child| matches!(child.kind(), "attribute" | "directive_attribute"))
{
return None;
}
let tag_name = child_by_kind(node, &["tag_name"])
.and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte())))
.unwrap_or_else(|| "anonymous".to_string());
Some(make_named_container_chunk(node, ChunkKind::Attrs, tag_name, source))
}
fn classify_attribute<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let name = child_by_kind(node, &["attribute_name"])
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
.unwrap_or_else(|| "attr".to_string());
make_kind_chunk(node, ChunkKind::Attr, Some(name), source, None)
}
fn classify_directive_attribute<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let raw = node_text(source, node.start_byte(), node.end_byte()).trim();
let directive_name =
extract_directive_name(node, source).unwrap_or_else(|| "directive".to_string());
let modifier_suffix = extract_directive_modifiers(node, source)
.filter(|mods| !mods.is_empty())
.map(|mods| format!("_{mods}"))
.unwrap_or_default();
if raw.starts_with('@') {
make_named_leaf_chunk(
node,
ChunkKind::Directive,
format!("on_{directive_name}{modifier_suffix}"),
source,
)
} else if raw.starts_with(':') {
make_named_leaf_chunk(
node,
ChunkKind::Directive,
format!("bind_{directive_name}{modifier_suffix}"),
source,
)
} else if raw.starts_with('#') {
make_kind_chunk(
node,
ChunkKind::Slot,
Some(format!("{directive_name}{modifier_suffix}")),
source,
None,
)
} else {
make_named_leaf_chunk(
node,
ChunkKind::Directive,
format!("dir_{directive_name}{modifier_suffix}"),
source,
)
}
}
fn make_named_leaf_chunk<'t>(
node: Node<'t>,
kind: ChunkKind,
identifier: impl Into<Option<String>>,
source: &str,
) -> RawChunkCandidate<'t> {
make_candidate(
node,
kind,
identifier,
NameStyle::Named,
signature_for_node(node, source),
None,
source,
)
}
fn make_named_container_chunk<'t>(
node: Node<'t>,
kind: ChunkKind,
identifier: impl Into<Option<String>>,
source: &str,
) -> RawChunkCandidate<'t> {
force_container(make_candidate(
node,
kind,
identifier,
NameStyle::Named,
signature_for_node(node, source),
Some(recurse_self(node, ChunkContext::ClassBody)),
source,
))
}
const fn force_container(mut candidate: RawChunkCandidate<'_>) -> RawChunkCandidate<'_> {
candidate.force_recurse = true;
candidate
}
fn classify_raw_text_block<'t>(
node: Node<'t>,
kind: ChunkKind,
source: &str,
default_language: SupportLang,
) -> RawChunkCandidate<'t> {
let Some(content_node) = child_by_kind(node, &["raw_text"]) else {
return positional_candidate(node, kind, source);
};
let candidate = with_region_node(positional_candidate(node, kind, source), Some(content_node));
match resolve_embedded_language(node, source, default_language) {
Some(language) => with_injected_subtree(candidate, language, content_node),
None => candidate,
}
}
fn extract_markup_tag_name(node: Node<'_>, source: &str) -> Option<String> {
start_like(node)
.and_then(|start| child_by_kind(start, &["tag_name"]))
.and_then(|tag| sanitize_identifier(node_text(source, tag.start_byte(), tag.end_byte())))
}
fn extract_slot_name(node: Node<'_>, source: &str) -> Option<String> {
let start = start_like(node)?;
named_children(start)
.into_iter()
.find(|child| {
node_text(source, child.start_byte(), child.end_byte())
.trim()
.starts_with('#')
})
.and_then(|child| extract_directive_name(child, source))
}
fn extract_directive_name(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["directive_name", "directive_value"])
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
}
fn extract_directive_modifiers(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["directive_modifiers"])
.and_then(|mods| sanitize_identifier(node_text(source, mods.start_byte(), mods.end_byte())))
}
fn has_attribute(node: Node<'_>, name: &str, source: &str) -> bool {
start_like(node)
.into_iter()
.flat_map(named_children)
.filter(|child| matches!(child.kind(), "attribute" | "directive_attribute"))
.filter_map(|attr| extract_attribute_name(attr, source))
.any(|attr_name| attr_name == name)
}
fn attribute_value(node: Node<'_>, name: &str, source: &str) -> Option<String> {
let start = start_like(node)?;
for child in named_children(start) {
if !matches!(child.kind(), "attribute" | "directive_attribute") {
continue;
}
if extract_attribute_name(child, source).as_deref() != Some(name) {
continue;
}
if let Some(value) =
child_by_kind(child, &["attribute_value", "quoted_attribute_value", "directive_value"])
{
return sanitize_identifier(&unquote_text(node_text(
source,
value.start_byte(),
value.end_byte(),
)));
}
return Some(name.to_string());
}
None
}
fn resolve_embedded_language(
node: Node<'_>,
source: &str,
default_language: SupportLang,
) -> Option<SupportLang> {
if let Some(language) = attribute_value(node, "lang", source) {
return SupportLang::from_alias(language.as_str());
}
Some(default_language)
}
fn extract_attribute_name(node: Node<'_>, source: &str) -> Option<String> {
child_by_kind(node, &["attribute_name", "directive_name"])
.and_then(|name| sanitize_identifier(node_text(source, name.start_byte(), name.end_byte())))
.or_else(|| {
if node_text(source, node.start_byte(), node.end_byte())
.trim()
.starts_with('#')
{
extract_directive_name(node, source)
} else {
None
}
})
}
fn start_like(node: Node<'_>) -> Option<Node<'_>> {
child_by_kind(node, &["start_tag", "self_closing_tag"])
}
-145
View File
@@ -1,145 +0,0 @@
use std::collections::{HashMap, HashSet};
use super::schema;
type AtomSet = HashSet<&'static str>;
static ATOM_NODES: std::sync::LazyLock<HashMap<&'static str, AtomSet>> =
std::sync::LazyLock::new(|| {
HashMap::from([
("astro", HashSet::from(["frontmatter"])),
("bash", HashSet::from(["string", "raw_string", "heredoc_body", "simple_expansion"])),
("c", HashSet::from(["string_literal", "char_literal"])),
("clojure", HashSet::from(["kwd_lit", "regex_lit"])),
("cmake", HashSet::from(["argument"])),
("cpp", HashSet::from(["string_literal", "char_literal"])),
(
"csharp",
HashSet::from([
"string_literal",
"verbatim_string_literal",
"character_literal",
"modifier",
]),
),
("css", HashSet::from(["integer_value", "float_value", "color_value", "string_value"])),
("elixir", HashSet::from(["string_constant_expr"])),
("go", HashSet::from(["interpreted_string_literal", "raw_string_literal"])),
(
"haskell",
HashSet::from([
"qualified_variable",
"qualified_module",
"qualified_constructor",
"strict_type",
]),
),
("hcl", HashSet::from(["string_lit", "heredoc_template"])),
(
"html",
HashSet::from(["doctype", "quoted_attribute_value", "raw_text", "tag_name", "text"]),
),
(
"java",
HashSet::from([
"string_literal",
"boolean_type",
"integral_type",
"floating_point_type",
"void_type",
]),
),
("json", HashSet::from(["string"])),
(
"julia",
HashSet::from([
"string_literal",
"prefixed_string_literal",
"command_literal",
"character_literal",
]),
),
(
"kotlin",
HashSet::from([
"nullable_type",
"string_literal",
"line_string_literal",
"character_literal",
]),
),
("lua", HashSet::from(["string"])),
("make", HashSet::from(["shell_text", "text"])),
("nix", HashSet::from(["string_expression", "indented_string_expression"])),
("objc", HashSet::from(["string_literal"])),
(
"perl",
HashSet::from([
"string_single_quoted",
"string_double_quoted",
"comments",
"command_qx_quoted",
"pattern_matcher_m",
"regex_pattern_qr",
"transliteration_tr_or_y",
"substitution_pattern_s",
"scalar_variable",
"array_variable",
"hash_variable",
"hash_access_variable",
]),
),
("php", HashSet::from(["string", "encapsed_string"])),
("protobuf", HashSet::from(["string"])),
("python", HashSet::from(["string"])),
("r", HashSet::from(["string", "special"])),
("ruby", HashSet::from(["string", "heredoc_body", "regex"])),
("rust", HashSet::from(["char_literal", "string_literal", "raw_string_literal"])),
("scala", HashSet::from(["string", "template_string", "interpolated_string_expression"])),
("solidity", HashSet::from(["string", "hex_string_literal", "unicode_string_literal"])),
("sql", HashSet::from(["string", "identifier"])),
("swift", HashSet::from(["line_string_literal"])),
("toml", HashSet::from(["string", "quoted_key"])),
("tsx", HashSet::from(["string", "template_string"])),
("typescript", HashSet::from(["string", "template_string", "regex", "predefined_type"])),
("xml", HashSet::from(["AttValue", "XMLDecl"])),
(
"yaml",
HashSet::from([
"string_scalar",
"double_quote_scalar",
"single_quote_scalar",
"block_scalar",
]),
),
("verilog", HashSet::from(["integral_number"])),
("zig", HashSet::from(["string"])),
])
});
pub fn is_atom_node(language: &str, kind: &str) -> bool {
ATOM_NODES
.get(language)
.is_some_and(|atom_nodes| atom_nodes.contains(kind))
}
pub fn is_atom_node_current(kind: &str) -> bool {
schema::current_language().is_some_and(|language| is_atom_node(language, kind))
}
#[cfg(test)]
mod tests {
use super::is_atom_node;
#[test]
fn nix_binding_set_is_not_an_atom() {
assert!(!is_atom_node("nix", "binding_set"));
assert!(is_atom_node("nix", "string_expression"));
}
#[test]
fn typescript_predefined_types_stay_atomic() {
assert!(is_atom_node("typescript", "predefined_type"));
assert!(!is_atom_node("typescript", "class_declaration"));
}
}
-509
View File
@@ -1,509 +0,0 @@
//! Per-language chunk classification trait.
//!
//! Languages now provide semantic tables plus a narrow override hook for
//! genuinely custom behavior.
use tree_sitter::Node;
use super::{
common::{
ChunkContext, NameStyle, RawChunkCandidate, extract_identifier, is_absorbable_attribute,
is_trivia_node, make_candidate, named_children, recurse_self, resolve_recurse,
resolve_value_container, sanitize_node_kind, signature_for_node,
try_promote_call_with_callback,
},
defaults,
kind::ChunkKind,
schema,
};
use crate::chunk::types::ChunkNode;
#[derive(Clone, Copy, Debug)]
pub enum RuleStyle {
Named,
Group,
Positional,
}
#[derive(Clone, Copy, Debug)]
pub enum NamingMode {
AutoIdentifier,
None,
SanitizedKind,
}
#[derive(Clone, Copy, Debug)]
pub enum RecurseMode {
None,
Auto(ChunkContext),
SelfNode(ChunkContext),
ValueContainer,
}
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
pub enum WrapperSignature {
#[default]
Child,
Wrapper,
}
#[derive(Clone, Copy, Debug, Default)]
pub struct WrapperTransform {
pub kind: Option<ChunkKind>,
pub name_style: Option<NameStyle>,
pub clear_identifier: bool,
pub signature: WrapperSignature,
}
#[derive(Clone, Copy, Debug)]
pub struct SemanticRule {
pub ts_kind: &'static str,
pub chunk_kind: ChunkKind,
pub style: RuleStyle,
pub naming: NamingMode,
pub recurse: RecurseMode,
}
pub const fn semantic_rule(
ts_kind: &'static str,
chunk_kind: ChunkKind,
style: RuleStyle,
naming: NamingMode,
recurse: RecurseMode,
) -> SemanticRule {
SemanticRule { ts_kind, chunk_kind, style, naming, recurse }
}
#[derive(Clone, Copy, Debug)]
pub struct StructuralOverrides {
pub extra_trivia: &'static [&'static str],
pub preserved_trivia: &'static [&'static str],
pub extra_root_wrappers: &'static [&'static str],
pub preserved_root_wrappers: &'static [&'static str],
pub absorbable_attrs: &'static [&'static str],
}
impl StructuralOverrides {
pub const EMPTY: Self = Self {
extra_trivia: &[],
preserved_trivia: &[],
extra_root_wrappers: &[],
preserved_root_wrappers: &[],
absorbable_attrs: &[],
};
pub fn is_extra_trivia(&self, kind: &str) -> bool {
self.extra_trivia.contains(&kind)
}
pub fn preserves_trivia(&self, kind: &str) -> bool {
self.preserved_trivia.contains(&kind)
}
pub fn is_extra_root_wrapper(&self, kind: &str) -> bool {
self.extra_root_wrappers.contains(&kind)
}
pub fn preserves_root_wrapper(&self, kind: &str) -> bool {
self.preserved_root_wrappers.contains(&kind)
}
pub fn is_absorbable_attr(&self, kind: &str) -> bool {
self.absorbable_attrs.contains(&kind)
}
}
#[derive(Clone, Copy, Debug)]
pub struct ClassifierTables {
pub root: &'static [SemanticRule],
pub class: &'static [SemanticRule],
pub function: &'static [SemanticRule],
pub structural_overrides: StructuralOverrides,
}
pub const EMPTY_CLASSIFIER_TABLES: ClassifierTables = ClassifierTables {
root: &[],
class: &[],
function: &[],
structural_overrides: StructuralOverrides::EMPTY,
};
pub trait LangClassifier {
fn tables(&self) -> &'static ClassifierTables {
&EMPTY_CLASSIFIER_TABLES
}
fn classify_root<'t>(&self, _node: Node<'t>, _source: &str) -> Option<RawChunkCandidate<'t>> {
None
}
fn classify_class<'t>(&self, _node: Node<'t>, _source: &str) -> Option<RawChunkCandidate<'t>> {
None
}
fn classify_function<'t>(
&self,
_node: Node<'t>,
_source: &str,
) -> Option<RawChunkCandidate<'t>> {
None
}
fn is_root_wrapper(&self, _kind: &str) -> bool {
false
}
fn preserve_root_wrapper(&self, _kind: &str) -> bool {
false
}
fn preserve_trivia(&self, _kind: &str) -> bool {
false
}
fn is_trivia(&self, _kind: &str) -> bool {
false
}
fn is_absorbable_attr(&self, _kind: &str) -> bool {
false
}
/// Return true to drop a named child from `collect_children_for_context`
/// without turning it into a chunk and without absorbing its byte range
/// into the next sibling via `attach_leading_trivia`.
///
/// Use this for structural framing nodes (JSX opening/closing elements,
/// framework fragment markers, etc.) that have no meaningful chunk of
/// their own and should NOT extend the following chunk's span backward.
/// Prefer `is_trivia` for comment-like nodes that should be absorbed as
/// leading context of the next chunk.
fn should_skip_child(&self, _kind: &str) -> bool {
false
}
fn classify_override<'t>(
&self,
_context: ChunkContext,
_node: Node<'t>,
_source: &str,
) -> Option<RawChunkCandidate<'t>> {
None
}
fn preserve_children(
&self,
_parent: &RawChunkCandidate<'_>,
_children: &[RawChunkCandidate<'_>],
) -> bool {
false
}
fn post_process(
&self,
_chunks: &mut Vec<ChunkNode>,
_root_children: &mut Vec<String>,
_source: &str,
) {
}
}
pub fn structural_overrides(classifier: &dyn LangClassifier) -> StructuralOverrides {
classifier.tables().structural_overrides
}
pub fn classify_with_tables<'tree>(
classifier: &dyn LangClassifier,
context: ChunkContext,
node: Node<'tree>,
source: &str,
) -> Option<RawChunkCandidate<'tree>> {
if let Some(candidate) = classifier.classify_override(context, node, source) {
return Some(candidate);
}
find_rule(classifier.tables(), context, node.kind())
.map(|rule| build_candidate_from_rule(node, source, *rule))
.or_else(|| match context {
ChunkContext::Root => classifier.classify_root(node, source),
ChunkContext::ClassBody => classifier.classify_class(node, source),
ChunkContext::FunctionBody => classifier.classify_function(node, source),
})
}
pub fn classify_with_defaults<'tree>(
classifier: &dyn LangClassifier,
context: ChunkContext,
node: Node<'tree>,
source: &str,
) -> RawChunkCandidate<'tree> {
if node.is_error() || node.kind() == "ERROR" {
return make_candidate(node, ChunkKind::Error, None, NameStyle::Error, None, None, source);
}
let candidate = match context {
ChunkContext::Root => classify_with_tables(classifier, context, node, source)
.unwrap_or_else(|| defaults::classify_root_default(node, source)),
ChunkContext::ClassBody => classify_with_tables(classifier, context, node, source)
.unwrap_or_else(|| defaults::classify_class_default(node, source)),
ChunkContext::FunctionBody => classify_with_tables(classifier, context, node, source)
.unwrap_or_else(|| defaults::classify_function_default(node, source)),
};
// If the classifier produced a groupable leaf (no recurse), try to
// promote call-with-trailing-callback patterns into named container
// chunks. This handles `describe(...)`, `t.Run(...)`, etc. across
// all languages without per-language opt-in.
if candidate.recurse.is_none()
&& candidate.groupable
&& let Some(promoted) = try_promote_call_with_callback(node, source)
{
return promoted;
}
candidate
}
pub fn first_wrapper_content_child<'tree>(
classifier: &dyn LangClassifier,
node: Node<'tree>,
) -> Option<Node<'tree>> {
if let Some(child) = schema_wrapper_child(node) {
return Some(child);
}
let overrides = structural_overrides(classifier);
named_children(node)
.into_iter()
.find(|child| !is_wrapper_metadata_child(*child, classifier, overrides))
}
pub fn promote_wrapper_candidate<'tree>(
classifier: &dyn LangClassifier,
context: ChunkContext,
node: Node<'tree>,
source: &str,
transform: WrapperTransform,
) -> Option<RawChunkCandidate<'tree>> {
let (child, candidate) = promotable_wrapper_child(classifier, context, node, source)?;
let signature_node = match transform.signature {
WrapperSignature::Child => child,
WrapperSignature::Wrapper => node,
};
let kind = transform.kind.unwrap_or(candidate.kind);
let name_style = transform.name_style.unwrap_or(candidate.name_style);
let identifier = if transform.clear_identifier {
None
} else {
candidate.identifier
};
Some(make_candidate(
node,
kind,
identifier,
name_style,
signature_for_node(signature_node, source),
candidate.recurse,
source,
))
}
pub fn build_candidate_from_rule<'tree>(
node: Node<'tree>,
source: &str,
rule: SemanticRule,
) -> RawChunkCandidate<'tree> {
let identifier = match rule.naming {
NamingMode::AutoIdentifier => extract_identifier(node, source),
NamingMode::None => None,
NamingMode::SanitizedKind => Some(sanitize_node_kind(node.kind()).to_string()),
};
let recurse = match rule.recurse {
RecurseMode::None => None,
RecurseMode::Auto(context) => resolve_recurse(node, context),
RecurseMode::SelfNode(context) => Some(recurse_self(node, context)),
RecurseMode::ValueContainer => resolve_value_container(node),
};
match rule.style {
RuleStyle::Named => make_candidate(
node,
rule.chunk_kind,
identifier,
NameStyle::Named,
signature_for_node(node, source),
recurse,
source,
),
RuleStyle::Group => {
make_candidate(node, rule.chunk_kind, identifier, NameStyle::Group, None, recurse, source)
},
RuleStyle::Positional => make_candidate(
node,
rule.chunk_kind,
None::<String>,
NameStyle::Named,
None,
recurse,
source,
),
}
}
fn find_rule(
tables: &ClassifierTables,
context: ChunkContext,
kind: &str,
) -> Option<&'static SemanticRule> {
let rules = match context {
ChunkContext::Root => tables.root,
ChunkContext::ClassBody => tables.class,
ChunkContext::FunctionBody => tables.function,
};
rules.iter().find(|rule| rule.ts_kind == kind)
}
fn promotable_wrapper_child<'tree>(
classifier: &dyn LangClassifier,
context: ChunkContext,
node: Node<'tree>,
source: &str,
) -> Option<(Node<'tree>, RawChunkCandidate<'tree>)> {
if let Some(child) = schema_wrapper_child(node) {
let candidate = classify_with_defaults(classifier, context, child, source);
if is_promotable_wrapper_candidate(child, &candidate) {
return Some((child, candidate));
}
}
let overrides = structural_overrides(classifier);
let mut promoted = named_children(node).into_iter().filter_map(|child| {
if is_wrapper_metadata_child(child, classifier, overrides) {
return None;
}
let candidate = classify_with_defaults(classifier, context, child, source);
is_promotable_wrapper_candidate(child, &candidate).then_some((child, candidate))
});
let promoted_child = promoted.next()?;
if promoted.next().is_some() {
return None;
}
Some(promoted_child)
}
fn is_wrapper_metadata_child(
node: Node<'_>,
classifier: &dyn LangClassifier,
overrides: StructuralOverrides,
) -> bool {
let kind = node.kind();
((is_trivia_node(node) || classifier.is_trivia(kind))
&& !overrides.preserves_trivia(kind)
&& !classifier.preserve_trivia(kind))
|| (overrides.is_extra_trivia(kind)
&& !overrides.preserves_trivia(kind)
&& !classifier.preserve_trivia(kind))
|| is_absorbable_attribute(kind)
|| overrides.is_absorbable_attr(kind)
|| classifier.is_absorbable_attr(kind)
}
fn is_promotable_wrapper_candidate(node: Node<'_>, candidate: &RawChunkCandidate<'_>) -> bool {
if matches!(candidate.kind, ChunkKind::Error | ChunkKind::Chunk | ChunkKind::Statements) {
return false;
}
candidate.identifier.is_some()
|| candidate.recurse.is_some()
|| candidate.kind.traits().container
|| node.kind().ends_with("_definition")
|| node.kind().ends_with("_declaration")
}
fn schema_wrapper_child(node: Node<'_>) -> Option<Node<'_>> {
let schema = schema::schema_for_current(node.kind())?;
for field in &schema.promotion_fields {
if let Some(child) = node.child_by_field_name(field) {
return Some(child);
}
}
None
}
/// Resolve a [`LangClassifier`] for the given language.
pub fn classifier_for(lang: &str) -> &'static dyn LangClassifier {
match lang {
"astro" => &super::ast_astro::AstroClassifier,
// JS / TS family
"javascript" | "js" | "jsx" | "typescript" | "ts" | "tsx" => {
&super::ast_js_ts::JsTsClassifier
},
// Python / Starlark
"python" | "starlark" => &super::ast_python::PythonClassifier,
// Rust
"rust" => &super::ast_rust::RustClassifier,
// Go
"go" | "golang" => &super::ast_go::GoClassifier,
// C / C++ / Objective-C
"c" | "cpp" | "c++" | "objc" | "objective-c" => &super::ast_c_cpp_objc::CCppClassifier,
// C# / Java
"csharp" | "java" => &super::ast_csharp_java::CSharpJavaClassifier,
// Clojure
"clojure" => &super::ast_clojure::ClojureClassifier,
// CMake
"cmake" => &super::ast_cmake::CMakeClassifier,
// CSS
"css" => &super::ast_css::CssClassifier,
// Data formats
"json" | "toml" | "yaml" => &super::ast_data_formats::DataFormatsClassifier,
// Dockerfile
"dockerfile" => &super::ast_dockerfile::DockerfileClassifier,
// Elixir
"elixir" => &super::ast_elixir::ElixirClassifier,
// Erlang
"erlang" => &super::ast_erlang::ErlangClassifier,
// GraphQL
"graphql" => &super::ast_graphql::GraphqlClassifier,
// Haskell / Scala
"haskell" | "scala" => &super::ast_haskell_scala::HaskellScalaClassifier,
// HTML / XML
"html" | "xml" => &super::ast_html_xml::HtmlXmlClassifier,
// INI
"ini" => &super::ast_ini::IniClassifier,
// Just
"just" => &super::ast_just::JustClassifier,
// Markdown / Handlebars
"markdown" | "handlebars" => &super::ast_markup::MarkupClassifier,
// Nix / HCL
"nix" | "hcl" => &super::ast_nix_hcl::NixHclClassifier,
// OCaml
"ocaml" => &super::ast_ocaml::OcamlClassifier,
// Perl
"perl" => &super::ast_perl::PerlClassifier,
// PowerShell
"powershell" => &super::ast_powershell::PowershellClassifier,
// Protobuf
"protobuf" | "proto" => &super::ast_proto::ProtoClassifier,
// R
"r" => &super::ast_r::RClassifier,
// Ruby / Lua
"ruby" | "lua" => &super::ast_ruby_lua::RubyLuaClassifier,
// SQL
"sql" => &super::ast_sql::SqlClassifier,
// Svelte
"svelte" => &super::ast_svelte::SvelteClassifier,
// TLA+ / PlusCal
"tlaplus" | "pluscal" | "pcal" | "tla" | "tla+" => &super::ast_tlaplus::TlaplusClassifier,
// Bash / Make / Diff
"bash" | "make" | "diff" => &super::ast_bash_make_diff::ShellBuildClassifier,
// Vue
"vue" => &super::ast_vue::VueClassifier,
// Everything else (Kotlin, Swift, PHP, Solidity, etc.)
_ => &super::ast_misc::MiscClassifier,
}
}
-903
View File
@@ -1,903 +0,0 @@
//! Shared helpers for chunk classification.
//!
//! These are the building blocks that per-language classifiers use to construct
//! [`RawChunkCandidate`] values. They are also used by the default (shared)
//! classification in [`super::defaults`].
use tree_sitter::Node;
use super::{
kind::{ChunkKind, SummaryStyle},
shape,
types::ChunkNode,
};
use crate::{env_uint, language::SupportLang};
// ── Configuration (environment overrides) ────────────────────────────────
env_uint! {
// Configured leaf threshold.
pub static LEAF_THRESHOLD: usize = "PI_CHUNK_LEAF_THRESHOLD" or 8 => [1, usize::MAX];
// Configured max chunk lines.
pub static MAX_CHUNK_LINES: usize = "PI_CHUNK_MAX_LINES" or 25 => [1, usize::MAX];
// Configured min recurse savings.
pub static MIN_RECURSE_SAVINGS: usize = "PI_CHUNK_MIN_SAVINGS" or 4 => [1, usize::MAX];
}
// ── Internal types ───────────────────────────────────────────────────────
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum ChunkContext {
Root,
ClassBody,
FunctionBody,
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum NameStyle {
Named,
Group,
Error,
}
#[derive(Clone, Copy, Debug)]
pub struct RecurseSpec<'tree> {
pub node: Node<'tree>,
pub context: ChunkContext,
}
#[derive(Clone, Copy, Debug)]
pub struct InjectedChunkSpec<'tree> {
pub language: SupportLang,
pub content_node: Node<'tree>,
}
#[derive(Clone, Debug)]
pub struct RawChunkCandidate<'tree> {
pub identifier: Option<String>,
pub kind: ChunkKind,
pub name_style: NameStyle,
pub range_start_byte: usize,
pub range_end_byte: usize,
/// Start byte for `chunk_checksum`; stays at the primary node's start while
/// `range_start_byte` may be extended backward to include leading
/// attributes/comments.
pub checksum_start_byte: usize,
pub range_start_line: usize,
pub range_end_line: usize,
pub signature: Option<String>,
pub error: bool,
pub groupable: bool,
pub has_leading_comment: bool,
pub force_recurse: bool,
pub region_node: Option<Node<'tree>>,
pub injected: Option<InjectedChunkSpec<'tree>>,
pub recurse: Option<RecurseSpec<'tree>>,
}
#[derive(Default)]
pub struct ChunkAccumulator {
pub chunks: Vec<ChunkNode>,
}
// ── Candidate constructors ───────────────────────────────────────────────
/// Convert a tree-sitter `end_position` into a 1-indexed line number.
///
/// Tree-sitter byte ranges are half-open, so `end_position` points to the byte
/// immediately *after* the last byte of the node. When that byte lands at
/// column 0 of a new row, the node's last byte actually sits on the previous
/// row and the 1-indexed last line is exactly `end.row` — not `end.row + 1`.
/// This matters for grammars whose container nodes terminate on the start of
/// the next sibling (tree-sitter-markdown sections, tree-sitter-toml tables,
/// etc.): the naive `end.row + 1` would claim the sibling's heading line and
/// make `replace_range_by_lines` clobber it.
const fn end_row_as_line(start: tree_sitter::Point, end: tree_sitter::Point) -> usize {
if end.column == 0 && end.row > start.row {
end.row
} else {
end.row + 1
}
}
pub fn make_candidate<'tree>(
node: Node<'tree>,
kind: ChunkKind,
identifier: impl Into<Option<String>>,
name_style: NameStyle,
signature: Option<String>,
recurse: Option<RecurseSpec<'tree>>,
source: &str,
) -> RawChunkCandidate<'tree> {
let identifier = identifier.into();
let start = node.start_position();
let end = node.end_position();
let summary = summary_for_node(node, kind, identifier.as_deref(), signature.as_deref(), source);
let start_byte = node.start_byte();
RawChunkCandidate {
identifier,
kind,
name_style,
range_start_byte: start_byte,
range_end_byte: node.end_byte(),
checksum_start_byte: start_byte,
range_start_line: start.row + 1,
range_end_line: end_row_as_line(start, end),
signature: summary,
error: kind == ChunkKind::Error,
groupable: kind.traits().groupable,
has_leading_comment: false,
force_recurse: kind.traits().container || (recurse.is_some() && node.has_error()),
region_node: recurse.map(|spec| spec.node),
injected: None,
recurse,
}
}
pub fn group_candidate<'tree>(
node: Node<'tree>,
kind: ChunkKind,
source: &str,
) -> RawChunkCandidate<'tree> {
make_candidate(node, kind, None, NameStyle::Group, None, None, source)
}
pub fn positional_candidate<'tree>(
node: Node<'tree>,
kind: ChunkKind,
source: &str,
) -> RawChunkCandidate<'tree> {
make_candidate(node, kind, None, NameStyle::Named, None, None, source)
}
pub fn named_candidate<'tree>(
node: Node<'tree>,
kind: ChunkKind,
source: &str,
recurse: Option<RecurseSpec<'tree>>,
) -> RawChunkCandidate<'tree> {
make_kind_chunk(node, kind, extract_identifier(node, source), source, recurse)
}
pub fn container_candidate<'tree>(
node: Node<'tree>,
kind: ChunkKind,
source: &str,
recurse: Option<RecurseSpec<'tree>>,
) -> RawChunkCandidate<'tree> {
make_kind_chunk(node, kind, extract_identifier(node, source), source, recurse)
}
pub fn make_kind_chunk<'tree>(
node: Node<'tree>,
kind: ChunkKind,
identifier: Option<String>,
source: &str,
recurse: Option<RecurseSpec<'tree>>,
) -> RawChunkCandidate<'tree> {
make_candidate(
node,
kind,
identifier,
NameStyle::Named,
signature_for_node(node, source),
recurse,
source,
)
}
pub fn make_kind_chunk_from<'tree>(
range_node: Node<'tree>,
signature_node: Node<'tree>,
kind: ChunkKind,
identifier: Option<String>,
source: &str,
recurse: Option<RecurseSpec<'tree>>,
) -> RawChunkCandidate<'tree> {
make_candidate(
range_node,
kind,
identifier,
NameStyle::Named,
signature_for_node(signature_node, source),
recurse,
source,
)
}
pub fn make_container_chunk<'tree>(
node: Node<'tree>,
kind: ChunkKind,
identifier: Option<String>,
source: &str,
recurse: Option<RecurseSpec<'tree>>,
) -> RawChunkCandidate<'tree> {
make_candidate(
node,
kind,
identifier,
NameStyle::Named,
signature_for_node(node, source),
recurse,
source,
)
}
pub fn make_container_chunk_from<'tree>(
range_node: Node<'tree>,
signature_node: Node<'tree>,
kind: ChunkKind,
identifier: Option<String>,
source: &str,
recurse: Option<RecurseSpec<'tree>>,
) -> RawChunkCandidate<'tree> {
make_candidate(
range_node,
kind,
identifier,
NameStyle::Named,
signature_for_node(signature_node, source),
recurse,
source,
)
}
/// Derive a "`prefix_identifier`" name from a node.
pub fn prefixed_name(prefix: &str, node: Node<'_>, source: &str) -> String {
let identifier = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
format!("{prefix}_{identifier}")
}
pub const fn with_region_node<'tree>(
mut candidate: RawChunkCandidate<'tree>,
region_node: Option<Node<'tree>>,
) -> RawChunkCandidate<'tree> {
candidate.region_node = region_node;
candidate
}
pub const fn with_injected_subtree<'tree>(
mut candidate: RawChunkCandidate<'tree>,
language: SupportLang,
content_node: Node<'tree>,
) -> RawChunkCandidate<'tree> {
candidate.injected = Some(InjectedChunkSpec { language, content_node });
candidate
}
pub const fn embedded_selector_token(language: SupportLang) -> &'static str {
match language {
SupportLang::Bash => "bash",
SupportLang::C => "c",
SupportLang::Cmake => "cmake",
SupportLang::Cpp => "cpp",
SupportLang::CSharp => "cs",
SupportLang::Css => "css",
SupportLang::Go => "go",
SupportLang::Html => "html",
SupportLang::Java => "java",
SupportLang::JavaScript => "js",
SupportLang::Json => "json",
SupportLang::Kotlin => "kt",
SupportLang::Lua => "lua",
SupportLang::Markdown => "md",
SupportLang::Php => "php",
SupportLang::Python => "py",
SupportLang::Ruby => "rb",
SupportLang::Rust => "rust",
SupportLang::Scala => "scala",
SupportLang::Sql => "sql",
SupportLang::Swift => "swift",
SupportLang::Toml => "toml",
SupportLang::Tsx => "tsx",
SupportLang::TypeScript => "ts",
SupportLang::Yaml => "yaml",
_ => language.canonical_name(),
}
}
// ── Inferred / catch-all candidates ──────────────────────────────────────
/// Derive a semantic name from a node's kind and/or identifier.
pub fn infer_named_candidate<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
let kind_name = sanitize_node_kind(node.kind());
let kind = ChunkKind::from_sanitized_kind(kind_name);
auto_classify(node, kind, source)
}
pub fn auto_classify<'tree>(
node: Node<'tree>,
kind: ChunkKind,
source: &str,
) -> RawChunkCandidate<'tree> {
make_kind_chunk(
node,
kind,
extract_identifier(node, source),
source,
auto_recurse_for_kind(node, kind),
)
}
// ── Tree navigation helpers ──────────────────────────────────────────────
pub fn named_children(node: Node<'_>) -> Vec<Node<'_>> {
let mut children = Vec::new();
for index in 0..node.child_count() {
if let Some(child) = node.child(index)
&& (child.is_named() || child.is_error() || child.kind() == "ERROR")
{
children.push(child);
}
}
children
}
pub fn child_by_kind<'tree>(node: Node<'tree>, kinds: &[&str]) -> Option<Node<'tree>> {
named_children(node)
.into_iter()
.find(|child| kinds.iter().any(|kind| child.kind() == *kind))
}
pub fn child_by_field_or_kind<'tree>(
node: Node<'tree>,
fields: &[&str],
kinds: &[&str],
) -> Option<Node<'tree>> {
for field in fields {
if let Some(child) = node.child_by_field_name(field) {
return Some(child);
}
}
child_by_kind(node, kinds)
}
pub fn resolve_recurse(node: Node<'_>, context: ChunkContext) -> Option<RecurseSpec<'_>> {
shape::recurse_target(node).map(|child| RecurseSpec { node: child, context })
}
pub fn resolve_value_container(node: Node<'_>) -> Option<RecurseSpec<'_>> {
shape::value_container_target(node)
.map(|child| RecurseSpec { node: child, context: ChunkContext::ClassBody })
}
fn auto_recurse_for_kind(node: Node<'_>, kind: ChunkKind) -> Option<RecurseSpec<'_>> {
let context = match kind {
ChunkKind::Constructor
| ChunkKind::Function
| ChunkKind::Macro
| ChunkKind::Method
| ChunkKind::Proc
| ChunkKind::Recipe => ChunkContext::FunctionBody,
_ if kind.traits().container => ChunkContext::ClassBody,
_ => return None,
};
resolve_recurse(node, context)
}
pub fn compute_body_inner_boundaries(
source: &str,
body_start: usize,
body_end: usize,
) -> (usize, usize) {
let bounded_start = body_start.min(source.len());
let bounded_end = body_end.min(source.len()).max(bounded_start);
let slice = &source[bounded_start..bounded_end];
let Some((first_non_ws_rel, first_non_ws)) = slice
.char_indices()
.find(|(_, ch)| !matches!(ch, ' ' | '\t' | '\n' | '\r'))
else {
return (bounded_start, bounded_end);
};
let Some((last_non_ws_rel, last_non_ws)) = slice
.char_indices()
.rev()
.find(|(_, ch)| !matches!(ch, ' ' | '\t' | '\n' | '\r'))
else {
return (bounded_start, bounded_end);
};
let has_delimiters = matches!((first_non_ws, last_non_ws), ('{', '}') | ('(', ')') | ('[', ']'));
if !has_delimiters {
let line_start = source[..bounded_start].rfind('\n').map_or(0, |pos| pos + 1);
let leading_indent = &source[line_start..bounded_start];
if !leading_indent.is_empty() && leading_indent.chars().all(|ch| matches!(ch, ' ' | '\t')) {
let mut inner_end = bounded_end;
let trailing = &source[bounded_end..];
if let Some(rel_newline) = trailing.find('\n') {
if trailing[..rel_newline]
.chars()
.all(|ch| matches!(ch, ' ' | '\t' | '\r'))
{
inner_end = bounded_end + rel_newline + 1;
}
} else if trailing.chars().all(|ch| matches!(ch, ' ' | '\t' | '\r')) {
inner_end = source.len();
}
return (line_start, inner_end);
}
return (bounded_start, bounded_end);
}
let mut inner_start = bounded_start + first_non_ws_rel + first_non_ws.len_utf8();
if source[inner_start..].starts_with("\r\n") {
inner_start += 2;
} else if source[inner_start..].starts_with('\n') {
inner_start += 1;
}
// Epilogue starts at the beginning of the line containing the closing
// delimiter so that the closing line's indentation is part of the epilogue,
// not the body.
let close_abs = bounded_start + last_non_ws_rel;
let inner_end = source[..close_abs]
.rfind('\n')
.map_or(close_abs, |nl| nl + 1);
(inner_start.min(bounded_end), inner_end.max(inner_start).min(bounded_end))
}
// ── Recurse helpers ──────────────────────────────────────────────────────
pub fn recurse_into<'tree>(
node: Node<'tree>,
context: ChunkContext,
fields: &[&str],
kinds: &[&str],
) -> Option<RecurseSpec<'tree>> {
child_by_field_or_kind(node, fields, kinds).map(|child| RecurseSpec { node: child, context })
}
pub const fn recurse_self(node: Node<'_>, context: ChunkContext) -> RecurseSpec<'_> {
RecurseSpec { node, context }
}
pub fn recurse_body(node: Node<'_>, context: ChunkContext) -> Option<RecurseSpec<'_>> {
resolve_recurse(node, context)
}
pub fn recurse_class(node: Node<'_>) -> Option<RecurseSpec<'_>> {
resolve_recurse(node, ChunkContext::ClassBody)
}
pub fn recurse_enum(node: Node<'_>) -> Option<RecurseSpec<'_>> {
resolve_recurse(node, ChunkContext::ClassBody)
}
pub fn recurse_value_container(node: Node<'_>) -> Option<RecurseSpec<'_>> {
resolve_value_container(node)
}
/// Try to promote a node that wraps a call expression with a trailing
/// callback/block argument. Returns a named chunk candidate with `recurse`
/// pointing into the callback body.
///
/// This is language-agnostic: it uses structural shape detection to find
/// call-with-callback patterns in any language (JS `describe(...)`, Go
/// `t.Run(...)`, Rust `tokio::spawn(async { ... })`, etc.).
pub fn try_promote_call_with_callback<'tree>(
node: Node<'tree>,
source: &str,
) -> Option<RawChunkCandidate<'tree>> {
let (func_node, body) = shape::trailing_callback_body(node)?;
// Extract a name from the call target (e.g. `describe`, `describe.serial`,
// `app.use`). Sanitize the raw source text (dots become underscores).
let name = sanitize_identifier(node_text(source, func_node.start_byte(), func_node.end_byte()));
let recurse = Some(RecurseSpec { node: body, context: ChunkContext::FunctionBody });
Some(make_kind_chunk(node, ChunkKind::Expression, name, source, recurse))
}
// ── Identifier extraction ────────────────────────────────────────────────
pub fn extract_identifier(node: Node<'_>, source: &str) -> Option<String> {
if node.kind() == "constructor" {
return Some("constructor".to_string());
}
if let Some(name_node) = shape::identifier_node(node) {
return sanitize_identifier(node_text(source, name_node.start_byte(), name_node.end_byte()));
}
None
}
/// Extract the name of a single-declarator binding like `const FOO = ...`.
pub fn extract_single_declarator_name(node: Node<'_>, source: &str) -> Option<String> {
let declarators: Vec<Node<'_>> = named_children(node)
.into_iter()
.filter(|c| c.kind() == "variable_declarator")
.collect();
if declarators.len() != 1 {
return None;
}
extract_identifier(declarators[0], source)
}
// ── Text helpers ─────────────────────────────────────────────────────────
pub fn node_text(source: &str, start_byte: usize, end_byte: usize) -> &str {
source.get(start_byte..end_byte).unwrap_or("")
}
pub fn sanitize_identifier(text: &str) -> Option<String> {
let mut out = String::new();
let mut previous_was_underscore = false;
for ch in text.chars() {
if ch.is_alphanumeric() || ch == '_' || ch == '$' {
out.push(ch);
previous_was_underscore = false;
continue;
}
if !previous_was_underscore {
out.push('_');
previous_was_underscore = true;
}
}
let sanitized = out.trim_matches('_').to_string();
if sanitized.is_empty() {
None
} else {
Some(sanitized)
}
}
pub fn unquote_text(text: &str) -> String {
text.trim().trim_matches('"').trim_matches('\'').to_string()
}
pub fn sanitize_node_kind(kind: &str) -> &str {
let mut kind_stripped = kind;
for suffix in [
"_instruction",
"_statement",
"_declaration",
"_definition",
"_item",
"ession", // _expression -> _expr
] {
if let Some(stripped) = kind_stripped.strip_suffix(suffix) {
kind_stripped = stripped;
}
}
if kind_stripped.is_empty() {
kind
} else {
kind_stripped
}
}
pub fn normalized_header(source: &str, start_byte: usize, end_byte: usize) -> String {
let slice = node_text(source, start_byte, end_byte);
let mut header = String::new();
for line in slice.lines().take(4) {
let trimmed = line.trim();
if trimmed.is_empty() {
continue;
}
if !header.is_empty() {
header.push(' ');
}
header.push_str(trimmed);
if trimmed.contains('{') || trimmed.ends_with(';') {
break;
}
}
collapse_whitespace(header.as_str())
}
pub fn collapse_whitespace(text: &str) -> String {
let mut out = String::new();
let mut pending_space = false;
for ch in text.chars() {
if ch.is_whitespace() {
pending_space = true;
continue;
}
if pending_space && !out.is_empty() {
out.push(' ');
}
out.push(ch);
pending_space = false;
}
out
}
// ── Signature helpers ────────────────────────────────────────────────────
pub fn signature_for_node(node: Node<'_>, source: &str) -> Option<String> {
let raw = if let Some(end_byte) = shape::signature_end_byte(node) {
node_text(source, node.start_byte(), end_byte)
} else {
node_text(source, node.start_byte(), node.end_byte())
};
let sig = collapse_whitespace(raw.trim());
let sig = sig
.trim_end_matches('{')
.trim_end_matches(':')
.trim_end_matches(';')
.trim();
if sig.is_empty() {
None
} else {
Some(sig.to_string())
}
}
// ── Summary / canonical naming ───────────────────────────────────────────
fn normalize_summary_text(summary: &str) -> Option<String> {
let summary = collapse_whitespace(summary.trim())
.trim_end_matches('{')
.trim_end_matches(':')
.trim_end_matches(';')
.trim()
.to_string();
if summary.is_empty() {
None
} else {
Some(summary)
}
}
fn summarize_function_node(
kind: ChunkKind,
identifier: Option<&str>,
raw_signature: &str,
) -> String {
let name = identifier.unwrap_or_else(|| kind.prefix());
let tail = go_method_signature(raw_signature, identifier)
.or_else(|| function_signature(raw_signature))
.or_else(|| python_function_signature(raw_signature))
.or_else(|| rust_function_signature(raw_signature))
.unwrap_or_else(|| raw_signature.to_string());
let tail = tail.replacen("): ", ") → ", 1);
format!("{} {name}{tail}", kind.prefix())
}
fn summarize_variable_node(
node: Node<'_>,
kind: ChunkKind,
identifier: Option<&str>,
source: &str,
) -> Option<String> {
let header = normalized_header(source, node.start_byte(), node.end_byte());
let keyword = header.split_whitespace().next()?;
let name = identifier.unwrap_or_else(|| kind.prefix());
Some(format!("{keyword} {name}"))
}
fn summarize_statement_node(node: Node<'_>, source: &str) -> Option<String> {
normalize_summary_text(normalized_header(source, node.start_byte(), node.end_byte()).as_str())
}
pub fn summary_for_node(
node: Node<'_>,
kind: ChunkKind,
identifier: Option<&str>,
raw_signature: Option<&str>,
source: &str,
) -> Option<String> {
match kind.traits().summary {
SummaryStyle::Imports => return Some("imports".to_string()),
SummaryStyle::Function => {
if let Some(signature) = raw_signature {
return Some(summarize_function_node(kind, identifier, signature));
}
},
SummaryStyle::Variable => {
return summarize_variable_node(node, kind, identifier, source);
},
SummaryStyle::Default => {},
}
if matches!(
node.kind(),
"for_statement"
| "for_in_statement"
| "for_of_statement"
| "if_statement"
| "return_statement"
| "expression_statement"
| "call_expression"
| "call"
| "function_call"
) {
return summarize_statement_node(node, source);
}
raw_signature
.and_then(normalize_summary_text)
.or_else(|| summarize_statement_node(node, source))
}
fn function_signature(header: &str) -> Option<String> {
let start = header.find('(')?;
let end = header.rfind('{').unwrap_or(header.len());
let signature = header.get(start..end)?.trim().trim_end_matches(';').trim();
if signature.is_empty() {
None
} else {
Some(signature.to_string())
}
}
fn go_method_signature(header: &str, identifier: Option<&str>) -> Option<String> {
let name = identifier?;
let declaration = header
.trim()
.trim_end_matches('{')
.trim_end_matches(';')
.trim();
let rest = declaration.strip_prefix("func")?.trim_start();
if !rest.starts_with('(') {
return None;
}
let receiver_end = find_matching_paren(rest, 0)?;
let after_receiver = rest.get(receiver_end + 1..)?.trim_start();
let after_name = after_receiver.strip_prefix(name)?.trim_start();
if !after_name.starts_with('(') {
return None;
}
Some(after_name.to_string())
}
fn find_matching_paren(text: &str, open_index: usize) -> Option<usize> {
let mut depth = 0usize;
for (index, ch) in text
.char_indices()
.skip_while(|(index, _)| *index < open_index)
{
match ch {
'(' => depth = depth.saturating_add(1),
')' => {
depth = depth.checked_sub(1)?;
if depth == 0 {
return Some(index);
}
},
_ => {},
}
}
None
}
fn python_function_signature(header: &str) -> Option<String> {
let start = header.find('(')?;
let end = header.rfind(':').unwrap_or(header.len());
let signature = header.get(start..end)?.trim();
if signature.is_empty() {
None
} else {
Some(signature.to_string())
}
}
fn rust_function_signature(header: &str) -> Option<String> {
let start = header.find('(')?;
let end = header.rfind('{').unwrap_or(header.len());
let mut sig = header.get(start..end)?.trim();
if let Some(idx) = sig.find(" where ") {
sig = sig.get(..idx)?.trim();
}
if sig.is_empty() {
None
} else {
Some(sig.to_string())
}
}
// ── Trivia and attribute detection ───────────────────────────────────────
pub fn is_trivia_node(node: Node<'_>) -> bool {
shape::is_generic_trivia(node)
}
pub fn is_absorbable_attribute(kind: &str) -> bool {
shape::is_generic_absorbable_attr(kind)
}
// ── Other helpers ────────────────────────────────────────────────────────
pub fn looks_like_python_statement(node: Node<'_>, source: &str) -> bool {
let header = normalized_header(source, node.start_byte(), node.end_byte());
header.contains(':') && !header.contains('{')
}
pub fn detect_indent(source: &str, start_byte: usize) -> (u32, String) {
let line_start = source.as_bytes()[..start_byte]
.iter()
.rposition(|&b| b == b'\n')
.map_or(0, |pos| pos + 1);
let line_prefix = &source[line_start..start_byte];
let mut cols = 0u32;
let mut ch = String::new();
for byte in line_prefix.bytes() {
match byte {
b'\t' => {
cols += 1;
if ch.is_empty() {
ch = "\t".to_string();
}
},
b' ' => {
cols += 1;
if ch.is_empty() {
ch = " ".to_string();
}
},
_ => break,
}
}
(cols, ch)
}
pub fn is_root_wrapper_node(node: Node<'_>) -> bool {
shape::is_root_wrapper_node(node)
}
pub const fn line_span(start_line: usize, end_line: usize) -> usize {
end_line.saturating_sub(start_line) + 1
}
pub fn total_line_count(source: &str) -> usize {
if source.is_empty() {
0
} else {
source.bytes().filter(|byte| *byte == b'\n').count() + 1
}
}
pub fn first_scalar_child(node: Node<'_>) -> Option<Node<'_>> {
named_children(node).into_iter().find(|child| {
!matches!(
child.kind(),
"block_node"
| "flow_node"
| "block_mapping"
| "flow_mapping"
| "block_sequence"
| "flow_sequence"
)
})
}
#[cfg(test)]
mod tests {
use super::compute_body_inner_boundaries;
#[test]
fn compute_body_inner_boundaries_handles_brace_and_indent_bodies() {
let ts = "function main() {\n\treturn 1;\n}\n";
let ts_start = ts.find('{').expect("open brace");
let ts_end = ts.rfind('}').expect("close brace") + 1;
let (ts_inner_start, ts_inner_end) = compute_body_inner_boundaries(ts, ts_start, ts_end);
assert_eq!(&ts[ts_inner_start..ts_inner_end], "\treturn 1;\n");
let rust = "fn main() {\n println!(\"hi\");\n}\n";
let rust_start = rust.find('{').expect("open brace");
let rust_end = rust.rfind('}').expect("close brace") + 1;
let (rust_inner_start, rust_inner_end) =
compute_body_inner_boundaries(rust, rust_start, rust_end);
assert_eq!(&rust[rust_inner_start..rust_inner_end], " println!(\"hi\");\n");
let go = "func main() {\n\treturn\n}\n";
let go_start = go.find('{').expect("open brace");
let go_end = go.rfind('}').expect("close brace") + 1;
let (go_inner_start, go_inner_end) = compute_body_inner_boundaries(go, go_start, go_end);
assert_eq!(&go[go_inner_start..go_inner_end], "\treturn\n");
let py = "def main():\n return 1\n";
let py_body_start = py.find(" return 1").expect("body start");
let py_body_end = py_body_start + " return 1".len();
let (py_inner_start, py_inner_end) =
compute_body_inner_boundaries(py, py_body_start, py_body_end);
assert_eq!(&py[py_inner_start..py_inner_end], " return 1");
}
}
-690
View File
@@ -1,690 +0,0 @@
use std::collections::{HashMap, HashSet};
use super::{
chunk_checksum,
common::{detect_indent, total_line_count},
kind::ChunkKind,
line_start_offsets,
state::ConflictMeta,
types::{ChunkNode, ChunkTree},
};
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct ConflictRegion {
pub ours_start_line: usize,
pub ours_end_line: usize,
pub theirs_start_line: usize,
pub theirs_end_line: usize,
pub marker_start_line: usize,
pub marker_end_line: usize,
pub ours_content: String,
pub theirs_content: String,
pub base_content: Option<String>,
pub base_label: Option<String>,
pub ours_label: String,
pub theirs_label: String,
}
#[derive(Clone, Debug)]
pub struct CleanResult {
pub source: String,
pub conflicts: Vec<ConflictRegion>,
pub ours_byte_ranges: Vec<(usize, usize)>,
}
#[derive(Clone)]
struct PendingConflict {
path: Option<String>,
parent_path: Option<String>,
ours_start_byte: usize,
ours_end_byte: usize,
ours_content: String,
theirs_content: String,
base_content: Option<String>,
base_label: Option<String>,
ours_label: String,
theirs_label: String,
}
pub fn has_conflict_markers(source: &str) -> bool {
source.contains("<<<<<<<") && source.contains("=======") && source.contains(">>>>>>>")
}
pub fn detect_conflicts(source: &str) -> Vec<ConflictRegion> {
let lines = source_lines(source);
let mut conflicts = Vec::new();
let mut index = 0usize;
while index < lines.len() {
let Some(ours_label) = marker_label(lines[index], "<<<<<<<") else {
index += 1;
continue;
};
let marker_start_line = index + 1;
let ours_start_index = index + 1;
let mut separator_index = None;
let mut base_marker_index = None;
let mut base_label = None;
let mut cursor = ours_start_index;
while cursor < lines.len() {
if let Some(label) = marker_label(lines[cursor], "|||||||") {
base_marker_index = Some(cursor);
base_label = Some(label);
cursor += 1;
while cursor < lines.len() {
if is_marker_line(lines[cursor], "=======") {
separator_index = Some(cursor);
break;
}
cursor += 1;
}
break;
}
if is_marker_line(lines[cursor], "=======") {
separator_index = Some(cursor);
break;
}
cursor += 1;
}
let Some(separator_index) = separator_index else {
index += 1;
continue;
};
let theirs_start_index = separator_index + 1;
let mut end_index = theirs_start_index;
while end_index < lines.len() && marker_label(lines[end_index], ">>>>>>>").is_none() {
end_index += 1;
}
let Some(theirs_label) = lines
.get(end_index)
.and_then(|line| marker_label(line, ">>>>>>>"))
else {
index += 1;
continue;
};
let ours_end_index = base_marker_index.unwrap_or(separator_index);
let base_start_index = base_marker_index.map_or(separator_index, |value| value + 1);
let base_end_index = separator_index;
let (ours_start_line, ours_end_line) = content_line_range(ours_start_index, ours_end_index);
let (theirs_start_line, theirs_end_line) = content_line_range(theirs_start_index, end_index);
conflicts.push(ConflictRegion {
ours_start_line,
ours_end_line,
theirs_start_line,
theirs_end_line,
marker_start_line,
marker_end_line: end_index + 1,
ours_content: join_lines(&lines[ours_start_index..ours_end_index]),
theirs_content: join_lines(&lines[theirs_start_index..end_index]),
base_content: base_marker_index
.map(|_| join_lines(&lines[base_start_index..base_end_index])),
base_label,
ours_label,
theirs_label,
});
index = end_index + 1;
}
conflicts
}
pub fn accept_ours(source: &str, conflicts: &[ConflictRegion]) -> CleanResult {
if conflicts.is_empty() {
return CleanResult {
source: source.to_owned(),
conflicts: Vec::new(),
ours_byte_ranges: Vec::new(),
};
}
let lines = source_lines(source);
let mut clean_source = String::with_capacity(source.len());
let mut ours_byte_ranges = Vec::with_capacity(conflicts.len());
let mut cursor_line = 1usize;
for conflict in conflicts {
let marker_start = conflict.marker_start_line.saturating_sub(1);
for line in lines
.iter()
.take(marker_start)
.skip(cursor_line.saturating_sub(1))
{
clean_source.push_str(line);
}
let ours_start_byte = clean_source.len();
let ours_start_index = conflict.ours_start_line.saturating_sub(1);
let ours_end_index = if conflict.ours_start_line <= conflict.ours_end_line {
conflict.ours_end_line
} else {
ours_start_index
};
for line in lines.iter().take(ours_end_index).skip(ours_start_index) {
clean_source.push_str(line);
}
ours_byte_ranges.push((ours_start_byte, clean_source.len()));
cursor_line = conflict.marker_end_line + 1;
}
for line in lines.iter().skip(cursor_line.saturating_sub(1)) {
clean_source.push_str(line);
}
CleanResult { source: clean_source, conflicts: conflicts.to_vec(), ours_byte_ranges }
}
pub fn reconstruct_markers(
clean_source: &str,
conflict_meta: &HashMap<String, ConflictMeta>,
) -> String {
if conflict_meta.is_empty() {
return clean_source.to_owned();
}
let mut conflicts = conflict_meta.iter().collect::<Vec<_>>();
conflicts.sort_unstable_by_key(|(_, meta)| meta.ours_start_byte);
let mut rendered = String::with_capacity(clean_source.len() + conflict_meta.len() * 64);
let mut cursor = 0usize;
for (_, meta) in conflicts {
if meta.ours_start_byte > clean_source.len()
|| meta.ours_end_byte > clean_source.len()
|| meta.ours_start_byte > meta.ours_end_byte
{
continue;
}
rendered.push_str(&clean_source[cursor..meta.ours_start_byte]);
push_marker_line(&mut rendered, "<<<<<<<", Some(meta.ours_label.as_str()));
push_conflict_content(&mut rendered, &clean_source[meta.ours_start_byte..meta.ours_end_byte]);
if let Some(base_content) = meta.base_content.as_deref() {
push_marker_line(&mut rendered, "|||||||", meta.base_label.as_deref());
push_conflict_content(&mut rendered, base_content);
}
push_marker_line(&mut rendered, "=======", None);
push_conflict_content(&mut rendered, meta.theirs_content.as_str());
push_marker_line(&mut rendered, ">>>>>>>", Some(meta.theirs_label.as_str()));
cursor = meta.ours_end_byte;
}
rendered.push_str(&clean_source[cursor..]);
rendered
}
pub fn inject_conflict_chunks(
tree: &mut ChunkTree,
source: &str,
clean_result: &CleanResult,
) -> HashMap<String, ConflictMeta> {
let mut pending = Vec::with_capacity(clean_result.conflicts.len());
for (conflict, (ours_start_byte, ours_end_byte)) in clean_result
.conflicts
.iter()
.zip(clean_result.ours_byte_ranges.iter().copied())
{
pending.push(PendingConflict {
path: None,
parent_path: None,
ours_start_byte,
ours_end_byte,
ours_content: conflict.ours_content.clone(),
theirs_content: conflict.theirs_content.clone(),
base_content: conflict.base_content.clone(),
base_label: conflict.base_label.clone(),
ours_label: conflict.ours_label.clone(),
theirs_label: conflict.theirs_label.clone(),
});
}
inject_pending_conflicts(tree, source, pending)
}
pub fn reinject_conflict_chunks(
tree: &mut ChunkTree,
source: &str,
conflict_meta: &HashMap<String, ConflictMeta>,
) -> HashMap<String, ConflictMeta> {
let mut pending = conflict_meta
.iter()
.map(|(path, meta)| PendingConflict {
path: Some(path.clone()),
parent_path: conflict_parent_path(path),
ours_start_byte: meta.ours_start_byte,
ours_end_byte: meta.ours_end_byte,
ours_content: source
.get(meta.ours_start_byte..meta.ours_end_byte)
.unwrap_or_default()
.to_owned(),
theirs_content: meta.theirs_content.clone(),
base_content: meta.base_content.clone(),
base_label: meta.base_label.clone(),
ours_label: meta.ours_label.clone(),
theirs_label: meta.theirs_label.clone(),
})
.collect::<Vec<_>>();
pending.sort_unstable_by_key(|conflict| conflict.ours_start_byte);
inject_pending_conflicts(tree, source, pending)
}
fn inject_pending_conflicts(
tree: &mut ChunkTree,
source: &str,
pending: Vec<PendingConflict>,
) -> HashMap<String, ConflictMeta> {
let mut conflict_meta = HashMap::new();
let mut existing_paths = tree
.chunks
.iter()
.map(|chunk| chunk.path.clone())
.collect::<HashSet<_>>();
let line_starts = line_start_offsets(source);
let mut counters = HashMap::<String, usize>::new();
for pending_conflict in pending {
if pending_conflict.ours_start_byte > pending_conflict.ours_end_byte
|| pending_conflict.ours_end_byte > source.len()
{
continue;
}
let parent_path = match pending_conflict.parent_path.as_deref() {
Some(parent_path) => {
if tree.chunks.iter().any(|chunk| chunk.path == parent_path) {
parent_path.to_owned()
} else {
continue;
}
},
None => find_innermost_parent_path(
tree,
pending_conflict.ours_start_byte,
pending_conflict.ours_end_byte,
)
.unwrap_or_default(),
};
let conflict_path = pending_conflict.path.unwrap_or_else(|| {
next_conflict_path(parent_path.as_str(), &mut counters, &existing_paths)
});
let ours_path = format!("{conflict_path}.ours");
let theirs_path = format!("{conflict_path}.theirs");
existing_paths.insert(conflict_path.clone());
existing_paths.insert(ours_path.clone());
existing_paths.insert(theirs_path.clone());
let (start_line, end_line, line_count) = real_line_stats(
&line_starts,
pending_conflict.ours_start_byte,
pending_conflict.ours_end_byte,
);
let theirs_line_count = display_line_count(pending_conflict.theirs_content.as_str()) as u32;
let (indent, indent_char) =
detect_indent(source, pending_conflict.ours_start_byte.min(source.len()));
let conflict_identifier = conflict_path
.rsplit('.')
.next()
.and_then(|leaf| leaf.strip_prefix("conflict_"))
.map(ToOwned::to_owned);
let conflict_checksum = chunk_checksum(
format!(
"{}\0{}\0{}\0{}\0{}",
pending_conflict.ours_label,
pending_conflict.theirs_label,
pending_conflict.ours_content,
pending_conflict.theirs_content,
pending_conflict.base_content.as_deref().unwrap_or_default(),
)
.as_bytes(),
);
let theirs_start_line = start_line;
let theirs_end_line = if theirs_line_count == 0 {
theirs_start_line
} else {
theirs_start_line + theirs_line_count - 1
};
let conflict_end_line = end_line.max(theirs_end_line);
let conflict_line_count = if conflict_end_line >= start_line {
conflict_end_line - start_line + 1
} else {
0
};
tree.chunks.push(ChunkNode {
path: conflict_path.clone(),
identifier: conflict_identifier,
kind: ChunkKind::Conflict,
leaf: false,
virtual_content: None,
parent_path: Some(parent_path.clone()),
children: vec![ours_path.clone(), theirs_path.clone()],
signature: None,
start_line,
end_line: conflict_end_line,
line_count: conflict_line_count,
start_byte: pending_conflict.ours_start_byte as u32,
end_byte: pending_conflict.ours_end_byte as u32,
checksum_start_byte: pending_conflict.ours_start_byte as u32,
prologue_end_byte: None,
epilogue_start_byte: None,
checksum: conflict_checksum,
error: false,
indent,
indent_char: indent_char.clone(),
group: false,
});
tree.chunks.push(ChunkNode {
path: ours_path.clone(),
identifier: None,
kind: ChunkKind::Ours,
leaf: true,
virtual_content: (pending_conflict.ours_start_byte == pending_conflict.ours_end_byte)
.then(String::new),
parent_path: Some(conflict_path.clone()),
children: Vec::new(),
signature: None,
start_line,
end_line,
line_count,
start_byte: pending_conflict.ours_start_byte as u32,
end_byte: pending_conflict.ours_end_byte as u32,
checksum_start_byte: pending_conflict.ours_start_byte as u32,
prologue_end_byte: None,
epilogue_start_byte: None,
checksum: chunk_checksum(
source
.as_bytes()
.get(pending_conflict.ours_start_byte..pending_conflict.ours_end_byte)
.unwrap_or_default(),
),
error: false,
indent,
indent_char: indent_char.clone(),
group: false,
});
tree.chunks.push(ChunkNode {
path: theirs_path.clone(),
identifier: None,
kind: ChunkKind::Theirs,
leaf: true,
virtual_content: Some(pending_conflict.theirs_content.clone()),
parent_path: Some(conflict_path.clone()),
children: Vec::new(),
signature: None,
start_line: theirs_start_line,
end_line: theirs_end_line,
line_count: theirs_line_count,
start_byte: pending_conflict.ours_start_byte as u32,
end_byte: pending_conflict.ours_start_byte as u32,
checksum_start_byte: pending_conflict.ours_start_byte as u32,
prologue_end_byte: None,
epilogue_start_byte: None,
checksum: chunk_checksum(pending_conflict.theirs_content.as_bytes()),
error: false,
indent,
indent_char,
group: false,
});
insert_conflict_child(
tree,
parent_path.as_str(),
conflict_path.as_str(),
pending_conflict.ours_start_byte,
pending_conflict.ours_end_byte,
);
conflict_meta.insert(conflict_path, ConflictMeta {
theirs_content: pending_conflict.theirs_content,
ours_label: pending_conflict.ours_label,
theirs_label: pending_conflict.theirs_label,
base_content: pending_conflict.base_content,
base_label: pending_conflict.base_label,
ours_start_byte: pending_conflict.ours_start_byte,
ours_end_byte: pending_conflict.ours_end_byte,
});
}
conflict_meta
}
fn source_lines(source: &str) -> Vec<&str> {
if source.is_empty() {
Vec::new()
} else {
source.split_inclusive('\n').collect()
}
}
fn strip_line_ending(line: &str) -> &str {
line.trim_end_matches(['\n', '\r'])
}
fn marker_label(line: &str, prefix: &str) -> Option<String> {
let stripped = strip_line_ending(line);
let remainder = stripped.strip_prefix(prefix)?;
Some(remainder.trim_start().to_owned())
}
fn is_marker_line(line: &str, prefix: &str) -> bool {
strip_line_ending(line).starts_with(prefix)
}
fn join_lines(lines: &[&str]) -> String {
let mut joined = String::new();
for line in lines {
joined.push_str(line);
}
joined
}
const fn content_line_range(start_index: usize, end_index: usize) -> (usize, usize) {
if start_index < end_index {
(start_index + 1, end_index)
} else {
(start_index + 1, start_index)
}
}
fn push_marker_line(out: &mut String, marker: &str, label: Option<&str>) {
out.push_str(marker);
if let Some(label) = label
&& !label.is_empty()
{
out.push(' ');
out.push_str(label);
}
out.push('\n');
}
fn push_conflict_content(out: &mut String, content: &str) {
out.push_str(content);
if !content.is_empty() && !content.ends_with('\n') {
out.push('\n');
}
}
fn find_innermost_parent_path(tree: &ChunkTree, start: usize, end: usize) -> Option<String> {
tree
.chunks
.iter()
.filter(|chunk| {
(chunk.start_byte as usize) <= start
&& end <= (chunk.end_byte as usize)
&& (chunk.end_byte as usize).saturating_sub(chunk.start_byte as usize)
>= end.saturating_sub(start)
})
.min_by_key(|chunk| {
(
(chunk.end_byte as usize).saturating_sub(chunk.start_byte as usize),
chunk.path.split('.').count(),
)
})
.map(|chunk| chunk.path.clone())
}
fn next_conflict_path(
parent_path: &str,
counters: &mut HashMap<String, usize>,
existing_paths: &HashSet<String>,
) -> String {
let key = parent_path.to_owned();
let next = counters.entry(key).or_insert(1);
loop {
let leaf = format!("conflict_{next}");
let path = if parent_path.is_empty() {
leaf
} else {
format!("{parent_path}.{leaf}")
};
*next += 1;
if !existing_paths.contains(path.as_str()) {
return path;
}
}
}
fn insert_conflict_child(
tree: &mut ChunkTree,
parent_path: &str,
conflict_path: &str,
ours_start_byte: usize,
ours_end_byte: usize,
) {
let Some(parent_index) = tree
.chunks
.iter()
.position(|chunk| chunk.path == parent_path)
else {
return;
};
let current_children = tree.chunks[parent_index].children.clone();
let mut updated_children = Vec::with_capacity(current_children.len() + 1);
let mut inserted = false;
for child_path in current_children {
let Some(child) = tree.chunks.iter().find(|chunk| chunk.path == child_path) else {
continue;
};
let child_start = child.start_byte as usize;
let child_end = child.end_byte as usize;
let overlaps = child_start < ours_end_byte && ours_start_byte < child_end;
if overlaps {
continue;
}
if !inserted && child_start > ours_start_byte {
updated_children.push(conflict_path.to_owned());
inserted = true;
}
updated_children.push(child_path);
}
if !inserted {
updated_children.push(conflict_path.to_owned());
}
tree.chunks[parent_index]
.children
.clone_from(&updated_children);
if parent_path.is_empty() {
tree.root_children = updated_children;
}
}
fn byte_to_line(line_starts: &[usize], byte: usize) -> u32 {
if line_starts.is_empty() {
return 0;
}
line_starts.partition_point(|offset| *offset <= byte) as u32
}
fn real_line_stats(line_starts: &[usize], start: usize, end: usize) -> (u32, u32, u32) {
if start == end {
let line = byte_to_line(line_starts, start);
return (line, line, 0);
}
let start_line = byte_to_line(line_starts, start);
let end_line = byte_to_line(line_starts, end.saturating_sub(1));
let line_count = if end_line >= start_line {
end_line - start_line + 1
} else {
0
};
(start_line, end_line, line_count)
}
fn display_line_count(content: &str) -> usize {
if content.is_empty() {
0
} else if content.ends_with('\n') {
content.split_terminator('\n').count()
} else {
total_line_count(content)
}
}
fn conflict_parent_path(path: &str) -> Option<String> {
match path.rsplit_once('.') {
Some((parent, _)) => Some(parent.to_owned()),
None if !path.is_empty() => Some(String::new()),
None => None,
}
}
#[cfg(test)]
mod tests {
use std::collections::HashMap;
use super::*;
use crate::chunk::state::ConflictMeta;
#[test]
fn detects_standard_and_diff3_conflicts() {
let source = "\
one\n<<<<<<< HEAD\nours\n||||||| base\nbase\n=======\ntheirs\n>>>>>>> topic\ntwo\n<<<<<<< \
HEAD\nx\n=======\ny\n>>>>>>> other\n";
let conflicts = detect_conflicts(source);
assert_eq!(conflicts.len(), 2);
assert_eq!(conflicts[0].ours_content, "ours\n");
assert_eq!(conflicts[0].theirs_content, "theirs\n");
assert_eq!(conflicts[0].base_content.as_deref(), Some("base\n"));
assert_eq!(conflicts[0].base_label.as_deref(), Some("base"));
assert_eq!(conflicts[1].ours_content, "x\n");
assert_eq!(conflicts[1].theirs_content, "y\n");
assert!(conflicts[1].base_content.is_none());
}
#[test]
fn accept_ours_returns_clean_source_and_byte_ranges() {
let source = "\
fn a() {\n<<<<<<< HEAD\n\treturn foo();\n=======\n\treturn bar();\n>>>>>>> topic\n}\n";
let conflicts = detect_conflicts(source);
let clean = accept_ours(source, &conflicts);
assert_eq!(clean.source, "fn a() {\n\treturn foo();\n}\n");
assert_eq!(clean.ours_byte_ranges, vec![(9, 24)]);
}
#[test]
fn reconstructs_conflict_markers_from_clean_source() {
let clean_source = "fn a() {\n\treturn foo();\n}\n";
let mut conflict_meta = HashMap::new();
conflict_meta.insert("fn_a.conflict_1".to_owned(), ConflictMeta {
theirs_content: "\treturn bar();\n".to_owned(),
ours_label: "HEAD".to_owned(),
theirs_label: "topic".to_owned(),
base_content: Some("\treturn baz();\n".to_owned()),
base_label: Some("base".to_owned()),
ours_start_byte: 9,
ours_end_byte: 24,
});
let reconstructed = reconstruct_markers(clean_source, &conflict_meta);
assert_eq!(
reconstructed,
"fn a() {\n<<<<<<< HEAD\n\treturn foo();\n||||||| base\n\treturn \
baz();\n=======\n\treturn bar();\n>>>>>>> topic\n}\n",
);
}
}
-85
View File
@@ -1,85 +0,0 @@
//! Default (shared) classification logic.
//!
//! These are minimal catch-all fallbacks for node kinds not handled by any
//! per-language classifier. The real classification lives in the `ast_*`
//! modules; these defaults only fire for truly unrecognized node kinds.
use tree_sitter::Node;
use super::{common::*, kind::ChunkKind};
pub fn classify_root_default<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
infer_named_candidate(node, source)
}
pub fn classify_class_default<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
infer_named_candidate(node, source)
}
pub fn classify_function_default<'tree>(
node: Node<'tree>,
source: &str,
) -> RawChunkCandidate<'tree> {
let kind_name = sanitize_node_kind(node.kind());
let kind = ChunkKind::from_sanitized_kind(kind_name);
let candidate = auto_classify(node, kind, source);
if candidate.recurse.is_some() || candidate.identifier.is_some() {
candidate
} else {
group_candidate(node, kind, source)
}
}
pub fn classify_var_decl<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
if let Some(candidate) = promote_assigned_expression(node, node, source) {
return candidate;
}
if let Some(name) = extract_single_declarator_name(node, source) {
return make_kind_chunk(node, ChunkKind::Variable, Some(name), source, None);
}
group_candidate(node, ChunkKind::Declarations, source)
}
pub fn promote_assigned_expression<'tree>(
range_node: Node<'tree>,
declaration_node: Node<'tree>,
source: &str,
) -> Option<RawChunkCandidate<'tree>> {
let declarators: Vec<Node<'tree>> = named_children(declaration_node)
.into_iter()
.filter(|c| c.kind() == "variable_declarator")
.collect();
if declarators.len() != 1 {
return None;
}
let decl = declarators[0];
let value = decl.child_by_field_name("value")?;
let name = extract_identifier(decl, source).unwrap_or_else(|| "anonymous".to_string());
match value.kind() {
"arrow_function" | "function_expression" | "function" => {
let recurse = recurse_body(value, ChunkContext::FunctionBody);
Some(make_kind_chunk_from(
range_node,
value,
ChunkKind::Function,
Some(name),
source,
recurse,
))
},
"class" | "class_expression" => {
let recurse = recurse_class(value);
Some(make_container_chunk_from(
range_node,
value,
ChunkKind::Class,
Some(name),
source,
recurse,
))
},
_ => None,
}
}
File diff suppressed because it is too large Load Diff
-618
View File
@@ -1,618 +0,0 @@
use crate::chunk::{HASHLINE_BIGRAMS, types::ChunkTree};
const DEFAULT_SPACE_INDENT_STEP: usize = 4;
const MAX_REASONABLE_INDENT_STEP: usize = 8;
pub fn dedent_python_style(text: &str) -> String {
let mut margin: Option<&str> = None;
for line in text.split('\n') {
if line.trim().is_empty() {
continue;
}
let indent = leading_whitespace(line);
margin = Some(match margin {
None => indent,
Some(current) if indent.starts_with(current) => current,
Some(current) if current.starts_with(indent) => indent,
Some(current) => common_prefix(current, indent),
});
}
let Some(margin) = margin else {
return text.to_owned();
};
if margin.is_empty() {
return text.to_owned();
}
text
.split('\n')
.map(|line| line.strip_prefix(margin).unwrap_or(line))
.collect::<Vec<_>>()
.join("\n")
}
pub fn indent_non_empty_lines(text: &str, prefix: &str) -> String {
if prefix.is_empty() {
return text.to_owned();
}
text
.split('\n')
.map(|line| {
if line.trim().is_empty() {
line.to_owned()
} else {
format!("{prefix}{line}")
}
})
.collect::<Vec<_>>()
.join("\n")
}
pub fn detect_space_indent_step(text: &str) -> usize {
let mut min = usize::MAX;
for line in text.split('\n') {
if line.trim().is_empty() {
continue;
}
let count = line.chars().take_while(|ch| *ch == ' ').count();
if count > 0 {
min = min.min(count);
}
}
if min == usize::MAX || min == 0 || min > MAX_REASONABLE_INDENT_STEP {
DEFAULT_SPACE_INDENT_STEP
} else {
min
}
}
pub fn count_indent_columns(whitespace: &str, space_step: usize) -> usize {
whitespace
.chars()
.map(|ch| if ch == '\t' { space_step } else { 1 })
.sum()
}
pub fn normalize_to_tabs(line: &str, indent_char: char, indent_step: usize) -> String {
if indent_char == '\t' {
return line.to_owned();
}
let whitespace = leading_whitespace(line);
if whitespace.is_empty() {
return line.to_owned();
}
let step = indent_step.max(1);
let total_columns = count_indent_columns(whitespace, step);
let tabs = total_columns / step;
let remainder = total_columns % step;
format!("{}{}{}", "\t".repeat(tabs), " ".repeat(remainder), &line[whitespace.len()..])
}
pub fn denormalize_from_tabs(
line: &str,
file_indent_char: char,
file_indent_step: usize,
) -> String {
if file_indent_char != ' ' && file_indent_char != '\t' {
return line.to_owned();
}
let whitespace = leading_whitespace(line);
if whitespace.is_empty() {
return line.to_owned();
}
let step = file_indent_step.max(1);
let mut converted = String::with_capacity(whitespace.len() * step.max(1));
for ch in whitespace.chars() {
match ch {
'\t' if file_indent_char == '\t' => converted.push('\t'),
'\t' => converted.push_str(&file_indent_char.to_string().repeat(step)),
' ' => converted.push(' '),
_ => converted.push(ch),
}
}
format!("{converted}{}", &line[whitespace.len()..])
}
pub fn normalize_target_indent(target_indent: &str, sample_text: &str) -> String {
if target_indent.is_empty() {
return String::new();
}
let has_tabs = target_indent.contains('\t');
let has_spaces = target_indent.contains(' ');
if !has_tabs || !has_spaces {
return target_indent.to_owned();
}
let space_step = detect_space_indent_step(sample_text);
let total_columns = count_indent_columns(target_indent, space_step);
let normalized_levels = round_to_nearest_step(total_columns, space_step) / space_step;
if normalized_levels == 0 {
return String::new();
}
match target_indent.chars().next().unwrap_or(' ') {
'\t' => "\t".repeat(normalized_levels),
_ => " ".repeat(normalized_levels * space_step),
}
}
pub fn normalize_leading_whitespace_char(
text: &str,
target_char: char,
file_indent_step: Option<usize>,
) -> String {
if target_char != ' ' && target_char != '\t' {
return text.to_owned();
}
let other_char = if target_char == ' ' { '\t' } else { ' ' };
let mut needs_conversion = false;
for line in text.split('\n') {
if line.trim().is_empty() {
continue;
}
let ws = leading_whitespace(line);
if ws.is_empty() {
continue;
}
needs_conversion |= ws.contains(other_char);
}
if !needs_conversion {
return text.to_owned();
}
let space_step = if target_char == ' ' {
file_indent_step
.filter(|step| *step > 1)
.unwrap_or_else(|| detect_space_indent_step(text))
} else {
file_indent_step.unwrap_or_else(|| detect_space_indent_step(text))
};
text
.split('\n')
.map(|line| {
let ws = leading_whitespace(line);
if ws.is_empty() {
return line.to_owned();
}
let rest = &line[ws.len()..];
let total_spaces = count_indent_columns(ws, space_step);
if target_char == ' ' {
format!("{}{}", " ".repeat(total_spaces), rest)
} else {
let tabs = total_spaces / space_step;
let remainder = total_spaces % space_step;
format!("{}{}{}", "\t".repeat(tabs), " ".repeat(remainder), rest)
}
})
.collect::<Vec<_>>()
.join("\n")
}
pub fn reindent_inserted_block(
content: &str,
target_indent: &str,
file_indent_step: Option<usize>,
) -> String {
if content.is_empty() {
return String::new();
}
let normalized_target_indent = normalize_target_indent(target_indent, content);
let mut dedented = dedent_python_style(content);
if let Some(target_char) = normalized_target_indent.chars().next() {
let target_step = if target_char == ' ' {
file_indent_step.unwrap_or_else(|| normalized_target_indent.chars().count())
} else {
file_indent_step.unwrap_or_else(|| detect_space_indent_step(&dedented))
};
dedented = normalize_leading_whitespace_char(&dedented, target_char, Some(target_step));
}
indent_non_empty_lines(&dedented, &normalized_target_indent)
}
/// Detect the file's indent character.
/// Prefer chunk metadata, then fall back to scanning source lines.
/// Returns `' '` when the file provides no indentation signal.
pub fn detect_file_indent_char(source: &str, tree: &ChunkTree) -> char {
for chunk in &tree.chunks {
if chunk.indent > 0 && !chunk.indent_char.is_empty() {
return chunk.indent_char.chars().next().unwrap_or(' ');
}
}
for line in source.split('\n') {
if line.trim().is_empty() {
continue;
}
if let Some(ch) = leading_whitespace(line).chars().next()
&& matches!(ch, ' ' | '\t')
{
return ch;
}
}
' '
}
/// Detect spaces-per-indent-level from parent→child indent differences.
/// Falls back to scanning `source` for the minimum indented-line width,
/// then to `DEFAULT_SPACE_INDENT_STEP` when neither gives a signal.
pub fn detect_file_indent_step(source: &str, tree: &ChunkTree) -> u32 {
for chunk in &tree.chunks {
if chunk.children.is_empty() {
continue;
}
for child_path in &chunk.children {
let Some(child) = tree
.chunks
.iter()
.find(|candidate| &candidate.path == child_path)
else {
continue;
};
if child.indent <= chunk.indent || child.indent_char != " " {
continue;
}
let step = child.indent - chunk.indent;
if step > 0 && step <= MAX_REASONABLE_INDENT_STEP as u32 {
return step;
}
}
}
detect_space_indent_step(source) as u32
}
pub fn strip_content_prefixes(content: &str) -> String {
let lines = content.split('\n').collect::<Vec<_>>();
let mut line_num_count = 0usize;
let mut non_empty = 0usize;
for line in &lines {
if line.trim().is_empty() {
continue;
}
non_empty += 1;
if parse_chunk_gutter_code_row(line).is_some() {
line_num_count += 1;
}
}
if non_empty == 0 {
return content.to_owned();
}
let without_line_numbers = if line_num_count * 10 > non_empty * 6 {
lines
.iter()
.map(|line| strip_chunk_gutter_line(line))
.collect::<Vec<_>>()
} else {
lines
.iter()
.map(|line| (*line).to_owned())
.collect::<Vec<_>>()
};
strip_new_line_prefixes(&without_line_numbers).join("\n")
}
fn leading_whitespace(line: &str) -> &str {
let end = line
.char_indices()
.find_map(|(index, ch)| (!matches!(ch, ' ' | '\t')).then_some(index))
.unwrap_or(line.len());
&line[..end]
}
fn common_prefix<'a>(left: &'a str, right: &'a str) -> &'a str {
let mut matched = 0usize;
for ((left_index, left_char), (_, right_char)) in left.char_indices().zip(right.char_indices()) {
if left_char != right_char {
break;
}
matched = left_index + left_char.len_utf8();
}
&left[..matched]
}
const fn round_to_nearest_step(value: usize, step: usize) -> usize {
if step == 0 {
return value;
}
((value + (step / 2)) / step) * step
}
fn parse_chunk_gutter_code_row(line: &str) -> Option<&str> {
let trimmed = line.trim_start_matches([' ', '\t']);
let digits = trimmed.chars().take_while(|ch| ch.is_ascii_digit()).count();
if digits == 0 {
return None;
}
let after_digits = &trimmed[digits..];
let after_spaces = after_digits.trim_start_matches([' ', '\t']);
let after_pipe = after_spaces
.strip_prefix('|')
.or_else(|| after_spaces.strip_prefix('│'))?;
Some(
after_pipe
.strip_prefix(' ')
.or_else(|| after_pipe.strip_prefix('\t'))
.unwrap_or(after_pipe),
)
}
fn strip_chunk_gutter_line(line: &str) -> String {
if let Some(rest) = parse_chunk_gutter_code_row(line) {
return rest.to_owned();
}
let trimmed = line.trim_start_matches([' ', '\t']);
if trimmed.starts_with('|') || trimmed.starts_with('│') {
return String::new();
}
line.to_owned()
}
fn strip_new_line_prefixes(lines: &[String]) -> Vec<String> {
let non_empty = lines.iter().filter(|line| !line.trim().is_empty()).count();
if non_empty == 0 {
return lines.to_vec();
}
let hash_prefixed = lines
.iter()
.filter(|line| !line.trim().is_empty())
.filter(|line| hashline_prefix_len(line).is_some())
.count();
if hash_prefixed == non_empty {
return lines
.iter()
.map(|line| match hashline_prefix_len(line) {
Some(prefix_len) => line[prefix_len..].to_owned(),
None => line.clone(),
})
.collect();
}
lines.to_vec()
}
fn hashline_prefix_len(line: &str) -> Option<usize> {
let mut offset = line.len() - line.trim_start_matches([' ', '\t']).len();
let mut remainder = &line[offset..];
if let Some(stripped) = remainder.strip_prefix(">>>") {
offset += 3;
remainder = stripped;
} else if let Some(stripped) = remainder.strip_prefix(">>") {
offset += 2;
remainder = stripped;
}
let ws = remainder.len() - remainder.trim_start_matches([' ', '\t']).len();
offset += ws;
remainder = &remainder[ws..];
if let Some(stripped) = remainder.strip_prefix('+') {
offset += 1;
remainder = stripped;
let inner_ws = remainder.len() - remainder.trim_start_matches([' ', '\t']).len();
offset += inner_ws;
remainder = &remainder[inner_ws..];
}
// Line number digits are mandatory (no `#`-only form in the new format).
let digits = remainder
.chars()
.take_while(|ch| ch.is_ascii_digit())
.count();
if digits == 0 {
return None;
}
offset += digits;
remainder = &remainder[digits..];
// Match exactly one BPE bigram (2 ASCII chars) from HASHLINE_BIGRAMS,
// directly adjacent to the line-number digits (no `#` separator).
// Use char-boundary-safe slicing to avoid panicking on multi-byte content.
let bigram_end = 2;
if remainder.len() < bigram_end || !remainder.is_char_boundary(bigram_end) {
return None;
}
let bigram = &remainder[..bigram_end];
if !bigram.is_ascii() || !HASHLINE_BIGRAMS.contains(&bigram) {
return None;
}
offset += bigram_end;
remainder = &remainder[bigram_end..];
// Anchor terminator is a single colon character.
remainder.strip_prefix(':').map(|_| offset + 1)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::chunk::{kind::ChunkKind, types::ChunkNode};
fn chunk(
path: &str,
parent_path: Option<&str>,
children: &[&str],
indent: u32,
indent_char: &str,
) -> ChunkNode {
let kind = match path.split_once('_').map_or(path, |(prefix, _)| prefix) {
"fn" => ChunkKind::Function,
"class" => ChunkKind::Class,
"stmts" => ChunkKind::Statements,
_ => ChunkKind::Chunk,
};
ChunkNode {
path: path.to_owned(),
identifier: path
.split_once('_')
.and_then(|(_, identifier)| (!identifier.is_empty()).then_some(identifier.to_owned())),
kind,
leaf: children.is_empty(),
virtual_content: None,
parent_path: parent_path.map(str::to_owned),
children: children.iter().map(|child| (*child).to_owned()).collect(),
signature: None,
start_line: 1,
end_line: 1,
line_count: 1,
start_byte: 0,
end_byte: 0,
checksum_start_byte: 0,
prologue_end_byte: None,
epilogue_start_byte: None,
checksum: "ABCD".to_owned(),
error: false,
indent,
indent_char: indent_char.to_owned(),
group: false,
}
}
#[test]
fn dedent_preserves_mixed_common_margin() {
let input = "\t foo\n\t bar\n\t baz";
assert_eq!(dedent_python_style(input), "foo\n bar\nbaz");
}
#[test]
fn normalize_leading_whitespace_char_uses_file_step() {
let input = "\t alpha\n\t beta";
assert_eq!(
normalize_leading_whitespace_char(input, ' ', Some(4)),
" alpha\n beta"
);
}
#[test]
fn canonical_indent_round_trips_common_profiles() {
let cases = [
(" value()", ' ', 4, "\tvalue()", " value()"),
(" value()", ' ', 3, "\t\tvalue()", " value()"),
(" value()", ' ', 2, "\tvalue()", " value()"),
("\tvalue()", '\t', 4, "\tvalue()", "\tvalue()"),
(" \t value()", ' ', 4, "\t value()", " value()"),
];
for (input, indent_char, indent_step, canonical, restored) in cases {
let normalized = normalize_to_tabs(input, indent_char, indent_step);
assert_eq!(normalized, canonical, "unexpected canonical indent for {input:?}");
assert_eq!(
denormalize_from_tabs(&normalized, indent_char, indent_step),
restored,
"unexpected restored indent for {input:?}"
);
}
}
#[test]
fn reindent_inserted_block_preserves_relative_indentation() {
// Agent sends tab-based content: call(\n\talpha,\n\tbeta,\n)
// After denormalize_from_tabs (4 spaces/tab): call(\n alpha,\n beta,\n)
// Should add target indent to all lines, preserving relative offsets.
let input = "call(\n alpha,\n beta,\n)";
assert_eq!(
reindent_inserted_block(input, " ", Some(4)),
" call(\n alpha,\n beta,\n )"
);
}
#[test]
fn strip_content_prefixes_removes_gutter_and_meta_rows() {
let input = "10 | fn main() {\n │ <.fn_main#ABCD>\n11 | println!(\"hi\");\n12 | }";
assert_eq!(strip_content_prefixes(input), "fn main() {\n\n println!(\"hi\");\n}");
}
#[test]
fn strip_content_prefixes_removes_colon_hashline_prefixes() {
let input = "1th:fn main() {\n2er:\tprintln!(\"hi\");\n3in:}";
assert_eq!(strip_content_prefixes(input), "fn main() {\n\tprintln!(\"hi\");\n}");
}
#[test]
fn detect_file_indent_step_prefers_space_children() {
let tree = ChunkTree {
language: "rust".to_owned(),
checksum: "ABCD".to_owned(),
line_count: 1,
parse_errors: 0,
parse_error_lines: Vec::new(),
fallback: false,
root_path: String::new(),
root_children: vec!["cls_A".to_owned()],
chunks: vec![
chunk("cls_A", Some(""), &["fn_b"], 0, " "),
chunk("fn_b", Some("cls_A"), &[], 2, " "),
],
};
assert_eq!(detect_file_indent_step("", &tree), 2);
}
#[test]
fn detect_file_indent_step_falls_back_to_source_scan() {
// When the chunk tree has no parent->child pairs (all leaves),
// fall back to scanning source lines for the minimum indent width.
let tree = ChunkTree {
language: "yaml".to_owned(),
checksum: "ABCD".to_owned(),
line_count: 3,
parse_errors: 0,
parse_error_lines: Vec::new(),
fallback: false,
root_path: String::new(),
root_children: vec!["key_ser".to_owned()],
chunks: vec![chunk("key_ser", Some(""), &[], 0, " ")],
};
let source = "server:\n host: localhost\n port: 5432\n";
assert_eq!(detect_file_indent_step(source, &tree), 2);
}
#[test]
fn detect_file_indent_char_falls_back_to_source_lines() {
let tree = ChunkTree {
language: "rust".to_owned(),
checksum: "ABCD".to_owned(),
line_count: 3,
parse_errors: 0,
parse_error_lines: Vec::new(),
fallback: false,
root_path: String::new(),
root_children: vec!["fn_mai".to_owned()],
chunks: vec![chunk("fn_mai", Some(""), &[], 0, "")],
};
let source = "fn main() {\n println!(\"hi\");\n}\n";
assert_eq!(detect_file_indent_char(source, &tree), ' ');
}
#[test]
fn detect_file_indent_char_defaults_to_spaces_without_signal() {
let tree = ChunkTree {
language: "rust".to_owned(),
checksum: "ABCD".to_owned(),
line_count: 0,
parse_errors: 0,
parse_error_lines: Vec::new(),
fallback: false,
root_path: String::new(),
root_children: Vec::new(),
chunks: Vec::new(),
};
assert_eq!(detect_file_indent_char("", &tree), ' ');
}
}
-699
View File
@@ -1,699 +0,0 @@
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
pub enum ChunkKind {
Add,
After,
Alias,
Algo,
Arg,
Argv,
Array,
At,
Attr,
AttrExpr,
Attrs,
Block,
BlockIf,
BlockLocals,
Body,
Case,
Cases,
Catch,
Cell,
Class,
Clause,
Cmd,
Code,
Cond,
Constructor,
Contract,
Copy,
Custom,
Declarations,
Decl,
DefaultExport,
Define,
Directive,
Either,
Elif,
Else,
Enum,
Env,
Error,
Except,
Exports,
Expose,
Expression,
Field,
Fields,
File,
Frame,
Function,
For,
ForIn,
ForOf,
Form,
Frontmatter,
Group,
GroupBy,
Headers,
Healthcheck,
Html,
Hunks,
Hunk,
If,
Iface,
Impl,
Imports,
Includes,
InlineFragment,
Install,
Interface,
Interpolation,
Item,
Join,
Key,
KeyScripts,
Label,
Let,
List,
Loop,
Macro,
Map,
Markdown,
Match,
Method,
Methods,
Module,
Mustache,
Object,
Operation,
Operator,
Option,
Options,
OrderBy,
Parameters,
Preamble,
Proc,
Process,
Project,
Proto,
Py,
Python,
Query,
Receive,
Recipe,
Relations,
Render,
Return,
Row,
Root,
Rule,
Schema,
Script,
ScriptModule,
ScriptSetup,
Section,
Select,
Setting,
Shebang,
Shell,
Slot,
Snippet,
Source,
Stage,
StaticInit,
Statements,
Struct,
Style,
StyleScoped,
Switch,
Table,
Tag,
Target,
Template,
Text,
Trait,
Translation,
Try,
Ts,
Type,
Typescript,
Union,
User,
Val,
Variable,
Variant,
Variants,
VersionGate,
When,
Where,
While,
With,
Workdir,
Conflict,
Ours,
Theirs,
Chunk,
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum SummaryStyle {
Function,
Variable,
Imports,
Default,
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub struct ChunkTraits {
pub groupable: bool,
pub packed: bool,
pub addressable_leaf: bool,
pub always_preserve_children: bool,
pub has_addressable_members: bool,
pub summary: SummaryStyle,
pub container: bool,
}
const DEFAULT_TRAITS: ChunkTraits = ChunkTraits {
groupable: false,
packed: false,
addressable_leaf: false,
always_preserve_children: false,
has_addressable_members: false,
summary: SummaryStyle::Default,
container: false,
};
const GROUP_TRAITS: ChunkTraits = ChunkTraits { groupable: true, ..DEFAULT_TRAITS };
const CONTAINER_TRAITS: ChunkTraits = ChunkTraits { container: true, ..DEFAULT_TRAITS };
const ADDRESSABLE_CONTAINER_TRAITS: ChunkTraits =
ChunkTraits { container: true, has_addressable_members: true, ..DEFAULT_TRAITS };
const PRESERVE_CHILDREN_TRAITS: ChunkTraits =
ChunkTraits { container: true, always_preserve_children: true, ..DEFAULT_TRAITS };
const PACKED_LEAF_TRAITS: ChunkTraits =
ChunkTraits { packed: true, addressable_leaf: true, ..DEFAULT_TRAITS };
const FUNCTION_TRAITS: ChunkTraits =
ChunkTraits { summary: SummaryStyle::Function, ..DEFAULT_TRAITS };
const VARIABLE_TRAITS: ChunkTraits = ChunkTraits {
packed: true,
addressable_leaf: true,
summary: SummaryStyle::Variable,
..DEFAULT_TRAITS
};
const IMPORTS_TRAITS: ChunkTraits =
ChunkTraits { groupable: true, summary: SummaryStyle::Imports, ..DEFAULT_TRAITS };
impl ChunkKind {
pub const fn prefix(self) -> &'static str {
match self {
Self::Add => "add",
Self::After => "aft",
Self::Alias => "al",
Self::Algo => "algo",
Self::Arg => "arg",
Self::Argv => "argv",
Self::Array => "ar",
Self::At => "at",
Self::Attr => "attr",
Self::AttrExpr => "aex",
Self::Attrs => "ats",
Self::Block => "blk",
Self::BlockIf => "bif",
Self::BlockLocals => "blc",
Self::Body => "b",
Self::Case => "case",
Self::Cases => "cs",
Self::Catch => "ctc",
Self::Cell => "cell",
Self::Class => "cls",
Self::Clause => "cla",
Self::Cmd => "cmd",
Self::Code => "code",
Self::Cond => "cond",
Self::Constructor => "ctor",
Self::Contract => "ctr",
Self::Copy => "copy",
Self::Custom => "cus",
Self::Declarations => "d",
Self::Decl => "decl",
Self::DefaultExport => "dex",
Self::Define => "def",
Self::Directive => "dir",
Self::Either => "eth",
Self::Elif => "elif",
Self::Else => "else",
Self::Enum => "en",
Self::Env => "env",
Self::Error => "err",
Self::Except => "exc",
Self::Exports => "exp",
Self::Expose => "exo",
Self::Expression => "ex",
Self::Field => "fld",
Self::Fields => "flds",
Self::File => "file",
Self::Frame => "fr",
Self::Function => "fn",
Self::For => "for",
Self::ForIn => "fri",
Self::ForOf => "fro",
Self::Form => "form",
Self::Frontmatter => "fm",
Self::Group => "grp",
Self::GroupBy => "gby",
Self::Headers => "hdrs",
Self::Healthcheck => "hck",
Self::Html => "html",
Self::Hunks => "hks",
Self::Hunk => "hunk",
Self::If => "if",
Self::Iface => "ifc",
Self::Impl => "ipl",
Self::Imports => "imp",
Self::Includes => "incl",
Self::InlineFragment => "ifr",
Self::Install => "inst",
Self::Interface => "intf",
Self::Interpolation => "itp",
Self::Item => "item",
Self::Join => "join",
Self::Key => "key",
Self::KeyScripts => "ksc",
Self::Label => "lbl",
Self::Let => "let",
Self::List => "list",
Self::Loop => "loop",
Self::Macro => "mc",
Self::Map => "map",
Self::Markdown => "md",
Self::Match => "mt",
Self::Method => "m",
Self::Methods => "ms",
Self::Module => "mod",
Self::Mustache => "mst",
Self::Object => "obj",
Self::Operation => "op",
Self::Operator => "oper",
Self::Option => "opt",
Self::Options => "opts",
Self::OrderBy => "oby",
Self::Parameters => "p",
Self::Preamble => "pre",
Self::Proc => "proc",
Self::Process => "prcs",
Self::Project => "proj",
Self::Proto => "pt",
Self::Py => "py",
Self::Python => "pyt",
Self::Query => "qry",
Self::Receive => "recv",
Self::Recipe => "rcp",
Self::Relations => "rels",
Self::Render => "rnd",
Self::Return => "ret",
Self::Row => "row",
Self::Root => "root",
Self::Rule => "rule",
Self::Schema => "sch",
Self::Script => "scr",
Self::ScriptModule => "smo",
Self::ScriptSetup => "sse",
Self::Section => "sct",
Self::Select => "sel",
Self::Setting => "stn",
Self::Shebang => "shb",
Self::Shell => "sh",
Self::Slot => "slot",
Self::Snippet => "snip",
Self::Source => "src",
Self::Stage => "stg",
Self::StaticInit => "sni",
Self::Statements => "st",
Self::Struct => "stc",
Self::Style => "sty",
Self::StyleScoped => "sco",
Self::Switch => "sw",
Self::Table => "tbl",
Self::Tag => "tag",
Self::Target => "tgt",
Self::Template => "tmpl",
Self::Text => "text",
Self::Trait => "tr",
Self::Translation => "trs",
Self::Try => "try",
Self::Ts => "ts",
Self::Type => "ty",
Self::Typescript => "tysc",
Self::Union => "u",
Self::User => "user",
Self::Val => "val",
Self::Variable => "var",
Self::Variant => "vr",
Self::Variants => "vrs",
Self::VersionGate => "vg",
Self::When => "when",
Self::Where => "wh",
Self::While => "wl",
Self::With => "with",
Self::Workdir => "wd",
Self::Conflict => "cfl",
Self::Ours => "ours",
Self::Theirs => "ths",
Self::Chunk => "ch",
}
}
pub const fn traits(self) -> &'static ChunkTraits {
match self {
Self::Add => &GROUP_TRAITS,
Self::After => &GROUP_TRAITS,
Self::Alias => &DEFAULT_TRAITS,
Self::Algo => &CONTAINER_TRAITS,
Self::Arg => &DEFAULT_TRAITS,
Self::Argv => &DEFAULT_TRAITS,
Self::Array => &DEFAULT_TRAITS,
Self::At => &DEFAULT_TRAITS,
Self::Attr => &DEFAULT_TRAITS,
Self::AttrExpr => &DEFAULT_TRAITS,
Self::Attrs => &CONTAINER_TRAITS,
Self::Block => &DEFAULT_TRAITS,
Self::BlockIf => &DEFAULT_TRAITS,
Self::BlockLocals => &DEFAULT_TRAITS,
Self::Body => &DEFAULT_TRAITS,
Self::Case => &DEFAULT_TRAITS,
Self::Cases => &GROUP_TRAITS,
Self::Catch => &DEFAULT_TRAITS,
Self::Cell => &CONTAINER_TRAITS,
Self::Class => &CONTAINER_TRAITS,
Self::Clause => &DEFAULT_TRAITS,
Self::Cmd => &GROUP_TRAITS,
Self::Code => &GROUP_TRAITS,
Self::Cond => &DEFAULT_TRAITS,
Self::Constructor => &FUNCTION_TRAITS,
Self::Contract => &DEFAULT_TRAITS,
Self::Copy => &GROUP_TRAITS,
Self::Custom => &DEFAULT_TRAITS,
Self::Declarations => &GROUP_TRAITS,
Self::Decl => &DEFAULT_TRAITS,
Self::DefaultExport => &DEFAULT_TRAITS,
Self::Define => &DEFAULT_TRAITS,
Self::Directive => &DEFAULT_TRAITS,
Self::Either => &DEFAULT_TRAITS,
Self::Elif => &DEFAULT_TRAITS,
Self::Else => &DEFAULT_TRAITS,
Self::Enum => &ADDRESSABLE_CONTAINER_TRAITS,
Self::Env => &DEFAULT_TRAITS,
Self::Error => &DEFAULT_TRAITS,
Self::Except => &DEFAULT_TRAITS,
Self::Exports => &GROUP_TRAITS,
Self::Expose => &DEFAULT_TRAITS,
Self::Expression => &DEFAULT_TRAITS,
Self::Field => &PACKED_LEAF_TRAITS,
Self::Fields => &GROUP_TRAITS,
Self::File => &DEFAULT_TRAITS,
Self::Frame => &DEFAULT_TRAITS,
Self::Function => &FUNCTION_TRAITS,
Self::For => &DEFAULT_TRAITS,
Self::ForIn => &DEFAULT_TRAITS,
Self::ForOf => &DEFAULT_TRAITS,
Self::Form => &DEFAULT_TRAITS,
Self::Frontmatter => &DEFAULT_TRAITS,
Self::Group => &DEFAULT_TRAITS,
Self::GroupBy => &DEFAULT_TRAITS,
Self::Headers => &GROUP_TRAITS,
Self::Healthcheck => &DEFAULT_TRAITS,
Self::Html => &GROUP_TRAITS,
Self::Hunks => &GROUP_TRAITS,
Self::Hunk => &DEFAULT_TRAITS,
Self::If => &DEFAULT_TRAITS,
Self::Iface => &PRESERVE_CHILDREN_TRAITS,
Self::Impl => &CONTAINER_TRAITS,
Self::Imports => &IMPORTS_TRAITS,
Self::Includes => &GROUP_TRAITS,
Self::InlineFragment => &DEFAULT_TRAITS,
Self::Install => &DEFAULT_TRAITS,
Self::Interface => &PRESERVE_CHILDREN_TRAITS,
Self::Interpolation => &GROUP_TRAITS,
Self::Item => &DEFAULT_TRAITS,
Self::Join => &DEFAULT_TRAITS,
Self::Key => &PACKED_LEAF_TRAITS,
Self::KeyScripts => &DEFAULT_TRAITS,
Self::Label => &DEFAULT_TRAITS,
Self::Let => &DEFAULT_TRAITS,
Self::List => &DEFAULT_TRAITS,
Self::Loop => &DEFAULT_TRAITS,
Self::Macro => &DEFAULT_TRAITS,
Self::Map => &DEFAULT_TRAITS,
Self::Markdown => &DEFAULT_TRAITS,
Self::Match => &DEFAULT_TRAITS,
Self::Method => &DEFAULT_TRAITS,
Self::Methods => &GROUP_TRAITS,
Self::Module => &CONTAINER_TRAITS,
Self::Mustache => &DEFAULT_TRAITS,
Self::Object => &DEFAULT_TRAITS,
Self::Operation => &DEFAULT_TRAITS,
Self::Operator => &DEFAULT_TRAITS,
Self::Option => &DEFAULT_TRAITS,
Self::Options => &GROUP_TRAITS,
Self::OrderBy => &DEFAULT_TRAITS,
Self::Parameters => &GROUP_TRAITS,
Self::Preamble => &DEFAULT_TRAITS,
Self::Proc => &CONTAINER_TRAITS,
Self::Process => &CONTAINER_TRAITS,
Self::Project => &DEFAULT_TRAITS,
Self::Proto => &DEFAULT_TRAITS,
Self::Py => &DEFAULT_TRAITS,
Self::Python => &DEFAULT_TRAITS,
Self::Query => &DEFAULT_TRAITS,
Self::Receive => &DEFAULT_TRAITS,
Self::Recipe => &DEFAULT_TRAITS,
Self::Relations => &DEFAULT_TRAITS,
Self::Render => &DEFAULT_TRAITS,
Self::Return => &DEFAULT_TRAITS,
Self::Row => &PACKED_LEAF_TRAITS,
Self::Root => &DEFAULT_TRAITS,
Self::Rule => &DEFAULT_TRAITS,
Self::Schema => &DEFAULT_TRAITS,
Self::Script => &DEFAULT_TRAITS,
Self::ScriptModule => &DEFAULT_TRAITS,
Self::ScriptSetup => &DEFAULT_TRAITS,
Self::Section => &DEFAULT_TRAITS,
Self::Select => &DEFAULT_TRAITS,
Self::Setting => &DEFAULT_TRAITS,
Self::Shebang => &DEFAULT_TRAITS,
Self::Shell => &DEFAULT_TRAITS,
Self::Slot => &DEFAULT_TRAITS,
Self::Snippet => &DEFAULT_TRAITS,
Self::Source => &DEFAULT_TRAITS,
Self::Stage => &DEFAULT_TRAITS,
Self::StaticInit => &DEFAULT_TRAITS,
Self::Statements => &GROUP_TRAITS,
Self::Struct => &ADDRESSABLE_CONTAINER_TRAITS,
Self::Style => &DEFAULT_TRAITS,
Self::StyleScoped => &DEFAULT_TRAITS,
Self::Switch => &DEFAULT_TRAITS,
Self::Table => &DEFAULT_TRAITS,
Self::Tag => &DEFAULT_TRAITS,
Self::Target => &DEFAULT_TRAITS,
Self::Template => &DEFAULT_TRAITS,
Self::Text => &GROUP_TRAITS,
Self::Trait => &PRESERVE_CHILDREN_TRAITS,
Self::Translation => &DEFAULT_TRAITS,
Self::Try => &DEFAULT_TRAITS,
Self::Ts => &DEFAULT_TRAITS,
Self::Type => &ADDRESSABLE_CONTAINER_TRAITS,
Self::Typescript => &DEFAULT_TRAITS,
Self::Union => &DEFAULT_TRAITS,
Self::User => &DEFAULT_TRAITS,
Self::Val => &DEFAULT_TRAITS,
Self::Variable => &VARIABLE_TRAITS,
Self::Variant => &PACKED_LEAF_TRAITS,
Self::Variants => &DEFAULT_TRAITS,
Self::VersionGate => &DEFAULT_TRAITS,
Self::When => &DEFAULT_TRAITS,
Self::Where => &DEFAULT_TRAITS,
Self::While => &DEFAULT_TRAITS,
Self::With => &GROUP_TRAITS,
Self::Workdir => &DEFAULT_TRAITS,
Self::Conflict => &DEFAULT_TRAITS,
Self::Ours => &PACKED_LEAF_TRAITS,
Self::Theirs => &PACKED_LEAF_TRAITS,
Self::Chunk => &DEFAULT_TRAITS,
}
}
pub fn path_segment(self, identifier: Option<&str>) -> String {
match identifier {
Some(identifier) => format!("{}_{identifier}", self.prefix()),
None => self.prefix().to_string(),
}
}
pub fn from_sanitized_kind(kind: &str) -> Self {
match kind {
"add" => Self::Add,
"after" => Self::After,
"alias" => Self::Alias,
"algo" => Self::Algo,
"arg" => Self::Arg,
"argv" => Self::Argv,
"array" => Self::Array,
"at" => Self::At,
"attr" => Self::Attr,
"attr_expr" => Self::AttrExpr,
"attrs" => Self::Attrs,
"block" => Self::Block,
"block_if" => Self::BlockIf,
"block_locals" => Self::BlockLocals,
"body" => Self::Body,
"case" => Self::Case,
"cases" => Self::Cases,
"catch" => Self::Catch,
"cell" => Self::Cell,
"class" => Self::Class,
"clause" => Self::Clause,
"cmd" => Self::Cmd,
"code" => Self::Code,
"cond" => Self::Cond,
"constructor" => Self::Constructor,
"contract" => Self::Contract,
"copy" => Self::Copy,
"custom" => Self::Custom,
"decls" | "declarations" => Self::Declarations,
"decl" => Self::Decl,
"default_export" => Self::DefaultExport,
"define" => Self::Define,
"directive" => Self::Directive,
"either" => Self::Either,
"elif" => Self::Elif,
"else" => Self::Else,
"enum" => Self::Enum,
"env" => Self::Env,
"error" => Self::Error,
"except" => Self::Except,
"exports" => Self::Exports,
"expose" => Self::Expose,
"expr" | "expression" => Self::Expression,
"field" => Self::Field,
"fields" => Self::Fields,
"file" => Self::File,
"frame" => Self::Frame,
"fn" | "function" => Self::Function,
"for" => Self::For,
"for_in" => Self::ForIn,
"for_of" => Self::ForOf,
"form" => Self::Form,
"frontmatter" => Self::Frontmatter,
"group" => Self::Group,
"group_by" => Self::GroupBy,
"headers" => Self::Headers,
"healthcheck" => Self::Healthcheck,
"html" => Self::Html,
"hunks" => Self::Hunks,
"hunk" => Self::Hunk,
"if" => Self::If,
"iface" => Self::Iface,
"impl" => Self::Impl,
"imports" => Self::Imports,
"includes" => Self::Includes,
"inline_fragment" => Self::InlineFragment,
"install" => Self::Install,
"interface" => Self::Interface,
"interpolation" => Self::Interpolation,
"item" => Self::Item,
"join" => Self::Join,
"key" => Self::Key,
"key_scripts" => Self::KeyScripts,
"label" => Self::Label,
"let" => Self::Let,
"list" => Self::List,
"loop" => Self::Loop,
"macro" => Self::Macro,
"map" => Self::Map,
"markdown" => Self::Markdown,
"match" => Self::Match,
"meth" | "method" => Self::Method,
"methods" => Self::Methods,
"mod" | "module" => Self::Module,
"mustache" => Self::Mustache,
"object" => Self::Object,
"operation" => Self::Operation,
"operator" => Self::Operator,
"option" => Self::Option,
"options" => Self::Options,
"order_by" => Self::OrderBy,
"params" | "parameters" => Self::Parameters,
"preamble" => Self::Preamble,
"proc" => Self::Proc,
"process" => Self::Process,
"project" => Self::Project,
"proto" => Self::Proto,
"py" => Self::Py,
"python" => Self::Python,
"query" => Self::Query,
"receive" => Self::Receive,
"recipe" => Self::Recipe,
"relations" => Self::Relations,
"render" => Self::Render,
"ret" | "return" => Self::Return,
"row" => Self::Row,
"root" => Self::Root,
"rule" => Self::Rule,
"schema" => Self::Schema,
"script" => Self::Script,
"script_module" => Self::ScriptModule,
"script_setup" => Self::ScriptSetup,
"section" => Self::Section,
"select" => Self::Select,
"setting" => Self::Setting,
"shebang" => Self::Shebang,
"shell" => Self::Shell,
"slot" => Self::Slot,
"snippet" => Self::Snippet,
"source" => Self::Source,
"stage" => Self::Stage,
"static_init" => Self::StaticInit,
"stmts" | "statements" => Self::Statements,
"struct" => Self::Struct,
"style" => Self::Style,
"style_scoped" => Self::StyleScoped,
"switch" => Self::Switch,
"table" => Self::Table,
"tag" => Self::Tag,
"target" => Self::Target,
"template" => Self::Template,
"text" => Self::Text,
"trait" => Self::Trait,
"translation" => Self::Translation,
"try" => Self::Try,
"ts" => Self::Ts,
"type" => Self::Type,
"typescript" => Self::Typescript,
"union" => Self::Union,
"user" => Self::User,
"val" => Self::Val,
"var" | "variable" => Self::Variable,
"variant" => Self::Variant,
"variants" => Self::Variants,
"version_gate" => Self::VersionGate,
"when" => Self::When,
"where" => Self::Where,
"while" => Self::While,
"with" => Self::With,
"workdir" => Self::Workdir,
"conflict" => Self::Conflict,
"ours" => Self::Ours,
"theirs" => Self::Theirs,
"chunk" => Self::Chunk,
_ => Self::Chunk,
}
}
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
-130
View File
@@ -1,130 +0,0 @@
use std::{cell::RefCell, collections::HashMap};
use serde::Deserialize;
#[derive(Debug, Deserialize)]
struct GeneratedSchema {
languages: HashMap<String, HashMap<String, NodeTypeSchema>>,
}
#[derive(Clone, Debug, Deserialize)]
pub struct NodeTypeSchema {
pub identifier_fields: Vec<String>,
pub body_fields: Vec<String>,
pub promotion_fields: Vec<String>,
pub container_child_kinds: Vec<String>,
pub is_supertype: bool,
pub has_structural_children: bool,
}
impl NodeTypeSchema {
pub const fn is_structural(&self) -> bool {
self.is_supertype
|| !self.identifier_fields.is_empty()
|| !self.body_fields.is_empty()
|| !self.container_child_kinds.is_empty()
|| self.has_structural_children
}
}
thread_local! {
static CURRENT_LANGUAGE: RefCell<Option<&'static str>> = const { RefCell::new(None) };
}
static GENERATED_SCHEMA: std::sync::LazyLock<HashMap<String, HashMap<String, NodeTypeSchema>>> =
std::sync::LazyLock::new(|| {
let raw = include_str!(concat!(env!("OUT_DIR"), "/chunk_schema.json"));
let generated: GeneratedSchema =
serde_json::from_str(raw).expect("generated chunk schema should parse");
generated.languages
});
pub struct SchemaLanguageGuard {
previous: Option<&'static str>,
}
impl Drop for SchemaLanguageGuard {
fn drop(&mut self) {
CURRENT_LANGUAGE.with(|current| {
*current.borrow_mut() = self.previous;
});
}
}
pub fn enter_language(language: &'static str) -> SchemaLanguageGuard {
let previous = CURRENT_LANGUAGE.with(|current| current.replace(Some(language)));
SchemaLanguageGuard { previous }
}
pub fn current_language() -> Option<&'static str> {
CURRENT_LANGUAGE.with(|current| *current.borrow())
}
pub fn schema_for(language: &str, kind: &str) -> Option<&'static NodeTypeSchema> {
GENERATED_SCHEMA
.get(language)
.and_then(|schemas| schemas.get(kind))
}
pub fn schema_for_current(kind: &str) -> Option<&'static NodeTypeSchema> {
current_language().and_then(|language| schema_for(language, kind))
}
#[cfg(test)]
pub fn has_schema(language: &str) -> bool {
GENERATED_SCHEMA.contains_key(language)
}
#[cfg(test)]
mod tests {
use super::{has_schema, schema_for};
#[test]
fn python_function_definition_schema_has_name_and_body() {
let schema = schema_for("python", "function_definition")
.expect("python function_definition schema should exist");
assert_eq!(schema.identifier_fields, vec!["name".to_string()]);
assert_eq!(schema.body_fields, vec!["body".to_string()]);
assert!(schema.promotion_fields.is_empty());
assert!(schema.is_structural());
}
#[test]
fn nix_let_expression_schema_exposes_binding_set_child() {
let schema =
schema_for("nix", "let_expression").expect("nix let_expression schema should exist");
assert_eq!(schema.body_fields, vec!["body".to_string()]);
assert!(
schema
.container_child_kinds
.iter()
.any(|kind| kind == "binding_set"),
"let_expression should surface binding_set children"
);
assert!(schema.has_structural_children);
}
#[test]
fn wrapper_schemas_preserve_promotable_definition_fields() {
let python = schema_for("python", "decorated_definition")
.expect("python decorated_definition schema should exist");
assert_eq!(python.promotion_fields, vec!["definition".to_string()]);
let typescript = schema_for("typescript", "export_statement")
.expect("typescript export_statement schema should exist");
assert!(
typescript
.promotion_fields
.iter()
.any(|field| field == "declaration"),
"export_statement should preserve declaration field for promotion"
);
}
#[test]
fn generated_schema_covers_expected_languages() {
for language in ["python", "nix", "toml", "typescript", "rust", "yaml", "handlebars"] {
assert!(has_schema(language), "{language} should have generated schema data");
}
}
}
-354
View File
@@ -1,354 +0,0 @@
use tree_sitter::Node;
use super::{atom_list, schema};
const IDENTIFIER_FIELD_PRIORITY: &[&str] = &[
"name",
"identifier",
"attrpath",
"key",
"label",
"alias",
"field",
"member",
"property",
"tag",
"target",
"variable",
];
const BODY_FIELD_PRIORITY: &[&str] = &["body", "value", "declaration_list", "block", "members"];
pub fn signature_end_byte(node: Node<'_>) -> Option<usize> {
recurse_target(node)
.filter(|child| child.start_byte() > node.start_byte())
.map(|child| child.start_byte())
}
pub fn identifier_node(node: Node<'_>) -> Option<Node<'_>> {
schema_identifier_node(node)
.or_else(|| probe_field_child(node, IDENTIFIER_FIELD_PRIORITY))
.or_else(|| {
local_named_children(node)
.into_iter()
.find(|child| looks_like_identifier_kind(child.kind()))
})
}
pub fn recurse_target(node: Node<'_>) -> Option<Node<'_>> {
schema_body_child(node)
.or_else(|| schema_container_child(node))
.or_else(|| probe_field_child(node, BODY_FIELD_PRIORITY))
.or_else(|| probe_structural_child(node))
.and_then(descend_to_structural_target)
}
pub fn value_container_target(node: Node<'_>) -> Option<Node<'_>> {
for field in ["value", "body"] {
if let Some(child) = node.child_by_field_name(field)
&& let Some(target) = descend_to_structural_target(child)
{
return Some(target);
}
}
recurse_target(node)
}
pub fn is_root_wrapper_node(node: Node<'_>) -> bool {
if is_atom_node(node) {
return false;
}
if let Some(schema) = schema::schema_for_current(node.kind()) {
if schema.is_supertype {
return true;
}
if schema.is_structural() && !is_transparent_recurse_wrapper(node) {
return false;
}
}
if identifier_node(node).is_some() {
return false;
}
let non_trivia = local_named_children(node)
.into_iter()
.filter(|child| !is_generic_trivia(*child))
.collect::<Vec<_>>();
non_trivia.len() == 1 && node_looks_structural(non_trivia[0])
}
pub fn is_generic_trivia(node: Node<'_>) -> bool {
node.is_extra() || kind_looks_like_comment(node.kind())
}
pub fn is_generic_absorbable_attr(kind: &str) -> bool {
matches!(kind, "attribute_item" | "inner_attribute_item")
}
pub fn is_atom_node(node: Node<'_>) -> bool {
atom_list::is_atom_node_current(node.kind())
}
fn schema_identifier_node(node: Node<'_>) -> Option<Node<'_>> {
let schema = schema::schema_for_current(node.kind())?;
for field in &schema.identifier_fields {
if let Some(child) = node.child_by_field_name(field) {
return Some(child);
}
}
None
}
fn schema_body_child(node: Node<'_>) -> Option<Node<'_>> {
let schema = schema::schema_for_current(node.kind())?;
for field in &schema.body_fields {
if let Some(child) = node.child_by_field_name(field) {
return Some(child);
}
}
None
}
fn schema_container_child(node: Node<'_>) -> Option<Node<'_>> {
let schema = schema::schema_for_current(node.kind())?;
local_named_children(node).into_iter().find(|child| {
schema
.container_child_kinds
.iter()
.any(|kind| child.kind() == kind)
})
}
fn probe_field_child<'tree>(node: Node<'tree>, fields: &[&str]) -> Option<Node<'tree>> {
fields
.iter()
.find_map(|field| node.child_by_field_name(field))
}
fn probe_structural_child(node: Node<'_>) -> Option<Node<'_>> {
local_named_children(node).into_iter().find(|child| {
!is_generic_trivia(*child)
&& !looks_like_identifier_kind(child.kind())
&& !is_atom_node(*child)
&& node_looks_structural(*child)
})
}
fn descend_to_structural_target(node: Node<'_>) -> Option<Node<'_>> {
if is_atom_node(node) {
return None;
}
if is_transparent_recurse_wrapper(node) {
let child = local_named_children(node)
.into_iter()
.find(|child| !is_generic_trivia(*child) && node_looks_structural(*child))?;
return descend_to_structural_target(child).or(Some(child));
}
if node_looks_structural(node) {
return Some(node);
}
probe_structural_child(node).and_then(descend_to_structural_target)
}
fn is_transparent_recurse_wrapper(node: Node<'_>) -> bool {
if is_atom_node(node) || identifier_node(node).is_some() {
return false;
}
schema::schema_for_current(node.kind()).is_some_and(|schema| schema.is_supertype)
|| matches!(node.kind(), "document" | "stream")
|| node.kind().ends_with("_node")
}
fn node_looks_structural(node: Node<'_>) -> bool {
if is_atom_node(node) {
return false;
}
if schema::schema_for_current(node.kind()).is_some_and(|schema| schema.is_structural()) {
return true;
}
let non_trivia_children = local_named_children(node)
.into_iter()
.filter(|child| !is_generic_trivia(*child))
.collect::<Vec<_>>();
!non_trivia_children.is_empty()
&& non_trivia_children
.iter()
.any(|child| !looks_like_identifier_kind(child.kind()) || child.named_child_count() > 0)
}
fn local_named_children(node: Node<'_>) -> Vec<Node<'_>> {
let mut children = Vec::new();
for index in 0..node.child_count() {
if let Some(child) = node.child(index)
&& (child.is_named() || child.is_error() || child.kind() == "ERROR")
{
children.push(child);
}
}
children
}
fn looks_like_identifier_kind(kind: &str) -> bool {
kind == "identifier"
|| kind == "name"
|| kind.ends_with("_identifier")
|| kind.ends_with("_name")
|| kind.ends_with("_label")
}
fn kind_looks_like_comment(kind: &str) -> bool {
kind == "comment" || kind.contains("comment")
}
/// Detect a call-with-trailing-callback pattern inside a node.
///
/// Returns `(call_text_node, body_node)` where `call_text_node` is the function
/// being called (for name extraction) and `body_node` is the block body of the
/// trailing callback argument.
///
/// This handles patterns like:
/// - JS/TS: `describe('x', () => { ... })`, `app.use(handler)`
/// - Go: `t.Run("x", func(t *testing.T) { ... })`
/// - Rust: `tokio::spawn(async { ... })`
/// - Ruby: `describe 'x' do ... end` (via `do_block`)
///
/// Only matches when the last argument has a structural block body, so
/// expression-body arrows and simple value arguments are not promoted.
pub fn trailing_callback_body(node: Node<'_>) -> Option<(Node<'_>, Node<'_>)> {
// Look through the node's children for a call-like node.
let call = local_named_children(node)
.into_iter()
.find(|c| is_call_like(*c))?;
// Find the arguments / parameter list.
let args_node = call
.child_by_field_name("arguments")
.or_else(|| call.child_by_field_name("args"))
.or_else(|| call.child_by_field_name("block"))?;
// Get the last named child of the arguments list.
let last_arg = local_named_children(args_node).into_iter().last()?;
// Try to find a structural body inside the last argument.
let body = recurse_target(last_arg)?;
// Extract the call target node for name purposes.
let func_node = call
.child_by_field_name("function")
.or_else(|| call.child_by_field_name("method"))
.or_else(|| call.child_by_field_name("name"))
.unwrap_or(call);
Some((func_node, body))
}
/// Returns `true` if the node looks like a function/method call.
fn is_call_like(node: Node<'_>) -> bool {
let kind = node.kind();
// Explicit known call kinds.
if matches!(
kind,
"call_expression"
| "call"
| "function_call"
| "method_call"
| "method_call_expression"
| "invocation_expression"
) {
return true;
}
// Heuristic: has an `arguments` field.
node.child_by_field_name("arguments").is_some()
}
#[cfg(test)]
mod tests {
use ast_grep_core::tree_sitter::LanguageExt;
use tree_sitter::{Node, Parser};
use super::{
identifier_node, is_root_wrapper_node, recurse_target, signature_end_byte,
trailing_callback_body,
};
use crate::language::SupportLang;
fn parse_tree_with_language(
source: &str,
language: SupportLang,
) -> (crate::chunk::schema::SchemaLanguageGuard, tree_sitter::Tree) {
let schema_language = crate::chunk::schema::enter_language(language.canonical_name());
let mut parser = Parser::new();
parser
.set_language(&language.get_ts_language())
.expect("language should parse");
let tree = parser.parse(source, None).expect("tree should parse");
(schema_language, tree)
}
fn find_named_node<'tree>(node: Node<'tree>, kind: &str) -> Option<Node<'tree>> {
if node.kind() == kind {
return Some(node);
}
for index in 0..node.child_count() {
let Some(child) = node.child(index) else {
continue;
};
if let Some(found) = find_named_node(child, kind) {
return Some(found);
}
}
None
}
#[test]
fn python_function_definition_resolves_identifier_and_body() {
let (_schema_language, tree) =
parse_tree_with_language("def greet(name):\n return name\n", SupportLang::Python);
let node =
find_named_node(tree.root_node(), "function_definition").expect("function_definition");
assert_eq!(identifier_node(node).expect("name").kind(), "identifier");
assert_eq!(recurse_target(node).expect("body").kind(), "block");
assert!(signature_end_byte(node).is_some());
}
#[test]
fn typescript_class_declaration_resolves_body() {
let (_schema_language, tree) =
parse_tree_with_language("class Greeter {\n hello() {}\n}\n", SupportLang::TypeScript);
let node = find_named_node(tree.root_node(), "class_declaration").expect("class_declaration");
assert_eq!(recurse_target(node).expect("body").kind(), "class_body");
}
#[test]
fn yaml_wrapper_nodes_are_structural_without_shared_kind_lists() {
let (_schema_language, tree) =
parse_tree_with_language("root:\n child: 1\n", SupportLang::Yaml);
let node = find_named_node(tree.root_node(), "block_node").expect("block_node");
assert!(is_root_wrapper_node(node), "block_node should be treated as a structural wrapper");
}
#[test]
fn js_expression_statement_with_callback_detects_body() {
let source = "describe(\"suite\", () => {\n\tit(\"a\", () => {});\n});\n";
let (_schema_language, tree) = parse_tree_with_language(source, SupportLang::TypeScript);
let expr_stmt =
find_named_node(tree.root_node(), "expression_statement").expect("expression_statement");
let (func_node, body) =
trailing_callback_body(expr_stmt).expect("should detect trailing callback body");
assert_eq!(body.kind(), "statement_block");
assert!(
matches!(func_node.kind(), "identifier" | "member_expression"),
"func_node should be an identifier or member_expression, got {}",
func_node.kind()
);
}
}
-689
View File
@@ -1,689 +0,0 @@
use std::{
collections::HashMap,
sync::{Arc, LazyLock},
};
use napi::{Error, Result};
use napi_derive::napi;
use regex::Regex;
use super::{
build_chunk_tree,
indent::{detect_file_indent_char, detect_file_indent_step, normalize_to_tabs},
resolve::{
ParsedSelector, chunk_region_range, format_region_ref, format_selector_tree,
resolve_chunk_selector, resolve_chunk_with_crc, split_selector_crc_and_region,
},
};
use crate::chunk::types::{
ChunkInfo, ChunkNode, ChunkReadStatus, ChunkReadTarget, ChunkRegion, ChunkTree, EditParams,
EditResult, ReadRenderParams, ReadResult, RenderParams, VisibleLineRange,
};
const LINE_RANGE_SELECTOR_RE: &str = r"^L(\d+)(?:-L?(\d+))?$";
const TLAPLUS_BEGIN_TRANSLATION_RE: &str = r"^\s*\\\*\s*BEGIN TRANSLATION\s*$";
const TLAPLUS_END_TRANSLATION_RE: &str = r"^\s*\\\*\s*END TRANSLATION\s*$";
static LINE_RANGE_SELECTOR_REGEX: LazyLock<Regex> =
LazyLock::new(|| Regex::new(LINE_RANGE_SELECTOR_RE).expect("line range regex must compile"));
static TLAPLUS_BEGIN_TRANSLATION_REGEX: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(TLAPLUS_BEGIN_TRANSLATION_RE).expect("tlaplus begin regex must compile")
});
static TLAPLUS_END_TRANSLATION_REGEX: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(TLAPLUS_END_TRANSLATION_RE).expect("tlaplus end regex must compile")
});
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct ConflictMeta {
pub theirs_content: String,
pub ours_label: String,
pub theirs_label: String,
pub base_content: Option<String>,
pub base_label: Option<String>,
pub ours_start_byte: usize,
pub ours_end_byte: usize,
}
#[derive(Clone)]
pub struct ChunkStateInner {
pub(crate) source: String,
pub(crate) language: String,
pub(crate) tree: ChunkTree,
pub(crate) notebook: Option<crate::chunk::ast_ipynb::SharedNotebookContext>,
pub(crate) conflict_meta: HashMap<String, ConflictMeta>,
lookup: HashMap<String, usize>,
checksum_lookup: HashMap<String, Vec<usize>>,
leaf_lookup: HashMap<String, Vec<usize>>,
suffix_lookup: HashMap<String, Vec<usize>>,
}
impl ChunkStateInner {
pub(crate) fn parse(source: String, language: String) -> Result<Self> {
let normalized_language = normalize_language(language.as_str());
if normalized_language == "ipynb" {
let parsed =
crate::chunk::ast_ipynb::parse_notebook(&source).map_err(napi::Error::from_reason)?;
let kernel_lang = parsed.context.kernel_language.clone();
let tree = crate::chunk::ast_ipynb::build_notebook_tree_from_virtual(
parsed.virtual_source.as_str(),
kernel_lang.as_str(),
)
.map_err(napi::Error::from_reason)?;
let ctx = std::sync::Arc::new(parsed.context);
let mut inner = Self::new(parsed.virtual_source, normalized_language, tree);
inner.notebook = Some(ctx);
return Ok(inner);
}
if crate::chunk::conflict::has_conflict_markers(source.as_str()) {
let conflicts = crate::chunk::conflict::detect_conflicts(source.as_str());
if !conflicts.is_empty() {
let clean_result = crate::chunk::conflict::accept_ours(source.as_str(), &conflicts);
let mut tree =
build_chunk_tree(clean_result.source.as_str(), normalized_language.as_str())?;
let conflict_meta = crate::chunk::conflict::inject_conflict_chunks(
&mut tree,
clean_result.source.as_str(),
&clean_result,
);
let mut inner = Self::new(clean_result.source, normalized_language, tree);
inner.conflict_meta = conflict_meta;
return Ok(inner);
}
}
let tree = build_chunk_tree(source.as_str(), normalized_language.as_str())?;
Ok(Self::new(source, normalized_language, tree))
}
pub(crate) fn new(source: String, language: String, tree: ChunkTree) -> Self {
let mut lookup = HashMap::new();
let mut checksum_lookup = HashMap::new();
let mut leaf_lookup = HashMap::new();
let mut suffix_lookup = HashMap::new();
for (index, chunk) in tree.chunks.iter().enumerate() {
lookup.insert(chunk.path.clone(), index);
checksum_lookup
.entry(chunk.checksum.clone())
.or_insert_with(Vec::new)
.push(index);
if chunk.path.is_empty() {
continue;
}
if let Some(leaf) = chunk.path.rsplit('.').next() {
leaf_lookup
.entry(leaf.to_string())
.or_insert_with(Vec::new)
.push(index);
}
let segments = chunk.path.split('.').collect::<Vec<_>>();
for start in 1..segments.len() {
suffix_lookup
.entry(segments[start..].join("."))
.or_insert_with(Vec::new)
.push(index);
}
}
Self {
source,
language,
tree,
notebook: None,
conflict_meta: HashMap::new(),
lookup,
checksum_lookup,
leaf_lookup,
suffix_lookup,
}
}
pub(crate) const fn source(&self) -> &str {
self.source.as_str()
}
pub(crate) const fn language(&self) -> &str {
self.language.as_str()
}
pub(crate) const fn tree(&self) -> &ChunkTree {
&self.tree
}
pub(crate) fn root(&self) -> Option<&ChunkNode> {
self.chunk("")
}
pub(crate) fn chunk(&self, path: &str) -> Option<&ChunkNode> {
self
.lookup
.get(path)
.and_then(|index| self.tree.chunks.get(*index))
}
pub(crate) fn chunk_by_index(&self, index: usize) -> Option<&ChunkNode> {
self.tree.chunks.get(index)
}
pub(crate) fn chunks_by_checksum(&self, checksum: &str) -> Vec<&ChunkNode> {
self
.checksum_lookup
.get(checksum)
.into_iter()
.flatten()
.filter_map(|index| self.chunk_by_index(*index))
.collect()
}
pub(crate) fn chunks_by_leaf(&self, leaf: &str) -> Vec<&ChunkNode> {
self
.leaf_lookup
.get(leaf)
.into_iter()
.flatten()
.filter_map(|index| self.chunk_by_index(*index))
.collect()
}
pub(crate) fn chunks_by_suffix(&self, suffix: &str) -> Vec<&ChunkNode> {
self
.suffix_lookup
.get(suffix)
.into_iter()
.flatten()
.filter_map(|index| self.chunk_by_index(*index))
.collect()
}
pub(crate) fn chunks(&self) -> impl Iterator<Item = &ChunkNode> {
self.tree.chunks.iter()
}
pub(crate) fn child_chunks(&self, parent_path: &str) -> Vec<&ChunkNode> {
self
.chunk(parent_path)
.map(|parent| {
parent
.children
.iter()
.filter_map(|path| self.chunk(path.as_str()))
.collect()
})
.unwrap_or_default()
}
pub(crate) fn line_to_containing_chunk_path(&self, line: u32) -> Option<String> {
crate::chunk::line_to_chunk_path(&self.tree, line)
}
}
/// Parsed file as a chunk tree: query nodes, render views, format grep hits,
/// and apply edits.
#[napi]
#[derive(Clone)]
pub struct ChunkState {
inner: Arc<ChunkStateInner>,
}
impl ChunkState {
pub(crate) fn from_inner(inner: ChunkStateInner) -> Self {
Self { inner: Arc::new(inner) }
}
pub(crate) fn inner(&self) -> &ChunkStateInner {
self.inner.as_ref()
}
}
#[napi]
impl ChunkState {
/// Build chunk state by parsing `source` with the given `language` id (e.g.
/// `typescript`).
#[napi(factory)]
pub fn parse(source: String, language: String) -> Result<Self> {
ChunkStateInner::parse(source, language).map(Self::from_inner)
}
/// Normalized language identifier used for the tree-sitter parse.
#[napi(getter)]
pub fn language(&self) -> String {
self.inner.language().to_string()
}
/// Full source text for this file.
#[napi(getter)]
pub fn source(&self) -> String {
self.inner.source().to_string()
}
/// Stable checksum for the entire file contents.
#[napi(getter)]
pub fn checksum(&self) -> String {
self.inner.tree().checksum.clone()
}
/// Line count of the source buffer.
#[napi(getter)]
pub fn line_count(&self) -> u32 {
self.inner.tree().line_count
}
/// Count of tree-sitter error nodes seen while building the tree.
#[napi(getter)]
pub fn parse_errors(&self) -> u32 {
self.inner.tree().parse_errors
}
/// True when a fallback classifier produced the tree.
#[napi(getter)]
pub fn fallback(&self) -> bool {
self.inner.tree().fallback
}
/// Selector path string for the synthetic root (often empty).
#[napi(getter)]
pub fn root_path(&self) -> String {
self.inner.tree().root_path.clone()
}
/// Top-level child chunk paths under the root.
#[napi(getter)]
pub fn root_children(&self) -> Vec<String> {
self.inner.tree().root_children.clone()
}
/// Total number of chunk nodes.
#[napi(getter)]
pub fn chunk_count(&self) -> u32 {
self.inner.tree().chunks.len() as u32
}
/// True when the parsed file contains unresolved merge conflicts.
#[napi]
pub fn has_conflicts(&self) -> bool {
!self.inner.conflict_meta.is_empty()
}
/// Count of unresolved merge conflicts represented in the chunk tree.
#[napi]
pub fn conflict_count(&self) -> u32 {
self.inner.conflict_meta.len() as u32
}
/// Summary for the root chunk, if it exists.
#[napi]
pub fn root(&self) -> Option<ChunkInfo> {
self.inner.root().map(chunk_info)
}
/// Look up [`ChunkInfo`] for a chunk selector path.
#[napi]
pub fn chunk(&self, chunk_path: String) -> Option<ChunkInfo> {
let mut warnings = Vec::new();
resolve_chunk_selector(self.inner(), Some(chunk_path.as_str()), &mut warnings)
.ok()
.map(chunk_info)
}
/// Every chunk node as a [`ChunkInfo`] list.
#[napi]
pub fn chunks(&self) -> Vec<ChunkInfo> {
self.inner.chunks().map(chunk_info).collect()
}
/// Direct children of `chunkPath` (use empty or omit for root); errors if
/// the path is missing.
#[napi]
pub fn children(&self, chunk_path: Option<String>) -> Result<Vec<ChunkInfo>> {
let parent = if let Some(chunk_path) = chunk_path {
let mut warnings = Vec::new();
resolve_chunk_selector(self.inner(), Some(chunk_path.as_str()), &mut warnings)
.map_err(Error::from_reason)?
} else {
self
.inner
.root()
.ok_or_else(|| Error::from_reason("Chunk tree is missing the root chunk".to_string()))?
};
Ok(self
.inner
.child_chunks(parent.path.as_str())
.into_iter()
.map(chunk_info)
.collect())
}
/// Chunk selector path that contains 1-based source line `line`, if any.
#[napi]
pub fn line_to_containing_chunk_path(&self, line: u32) -> Option<String> {
self.inner.line_to_containing_chunk_path(line)
}
/// Render a chunk subtree or listing as UTF-8 text for tools.
#[napi]
pub fn render(&self, params: RenderParams) -> String {
crate::chunk::render::render_state(self.inner(), &params)
}
/// Parse `readPath` (selector, line scope, etc.) and return rendered text or
/// errors.
#[napi]
pub fn render_read(&self, params: ReadRenderParams) -> Result<ReadResult> {
let ParsedChunkReadPath { selector, crc, region } =
match parse_chunk_read_path(params.read_path.as_str()) {
Ok(parsed) => parsed,
Err(err) => {
return Ok(ReadResult {
text: format!("{}\n\n{}", params.display_path, err),
chunk: Some(ChunkReadTarget {
status: ChunkReadStatus::UnsupportedRegion,
selector: params.read_path.clone(),
}),
});
},
};
let visible_range = selector.as_deref().and_then(parse_visible_line_range);
let Some(root) = self.inner.root() else {
return Ok(ReadResult {
text: format!("{}\n\n[Chunk tree root missing]", params.display_path),
chunk: None,
});
};
if let Some(visible_range) = visible_range {
if visible_range.start_line > self.inner.tree().line_count {
let suggestion = if self.inner.tree().line_count == 0 {
"The file is empty.".to_string()
} else {
format!(
"Use sel=L1 to read from the start, or sel=L{} to read the last line.",
self.inner.tree().line_count
)
};
return Ok(ReadResult {
text: format!(
"Line {} is beyond end of file ({} lines total). {suggestion}",
visible_range.start_line,
self.inner.tree().line_count,
),
chunk: None,
});
}
let clamped_range = VisibleLineRange {
start_line: visible_range.start_line,
end_line: visible_range.end_line.min(self.inner.tree().line_count),
};
let notice = format!(
"[Notice: chunk view scoped to requested lines L{}-L{}; clipped chunks keep head/tail \
context and collapse non-overlapping children.]",
clamped_range.start_line, clamped_range.end_line
);
let text = self.render(RenderParams {
chunk_path: Some(root.path.clone()),
title: params.display_path.clone(),
language_tag: params.language_tag.clone(),
visible_range: Some(clamped_range),
render_children_only: true,
omit_checksum: params.omit_checksum,
anchor_style: params.anchor_style,
show_leaf_preview: true,
tab_replacement: params.tab_replacement,
normalize_indent: params.normalize_indent,
focused_paths: None,
});
return Ok(ReadResult { text: format!("{notice}\n\n{text}"), chunk: None });
}
if selector.as_deref().is_none_or(str::is_empty) && crc.is_none() && region.is_none() {
return Ok(ReadResult {
text: self.render(RenderParams {
chunk_path: Some(root.path.clone()),
title: params.display_path.clone(),
language_tag: params.language_tag.clone(),
visible_range: None,
render_children_only: true,
omit_checksum: params.omit_checksum,
anchor_style: params.anchor_style,
show_leaf_preview: true,
tab_replacement: params.tab_replacement,
normalize_indent: params.normalize_indent,
focused_paths: None,
}),
chunk: None,
});
}
if selector.as_deref() == Some("?") {
let mut lines = vec![format!("{} chunks (dot-joined paths):", params.display_path)];
lines.extend(format_selector_tree(
self.inner.tree(),
&self.inner.tree().root_children,
false,
));
return Ok(ReadResult { text: lines.join("\n"), chunk: None });
}
let mut warnings = Vec::new();
let resolved = match resolve_chunk_with_crc(
self.inner(),
selector.as_deref(),
crc.as_deref(),
&mut warnings,
) {
Ok(resolved) => resolved,
Err(err) => {
let sel = selector.unwrap_or_default();
return Ok(ReadResult {
text: format!("{}:{}\n\n{}", params.display_path, sel, err),
chunk: Some(ChunkReadTarget { status: ChunkReadStatus::NotFound, selector: sel }),
});
},
};
let chunk = resolved.chunk;
// Use the region from parse_chunk_read_path, NOT region,
// because resolve_chunk_with_crc re-parses the already-cleaned
// selector and loses the region suffix.
let selector_ref = format_region_ref(chunk, region);
// Fall back to whole-chunk read when the chunk has no real region
// boundaries (e.g. markdown fenced blocks, leaf chunks without
// prologue/epilogue). Without this, `^` on such chunks returns
// "[Empty @^ region]" instead of the whole chunk.
let region = if region.is_some()
&& (chunk.prologue_end_byte.is_none() || chunk.epilogue_start_byte.is_none())
{
None
} else {
region
};
if let Some(absolute_line_range) = params.absolute_line_range {
let req_start = absolute_line_range.start_line;
let req_end = absolute_line_range.end_line;
let low = chunk.start_line.max(req_start.min(req_end));
let high = chunk.end_line.min(req_start.max(req_end));
if low > high {
let requested = if req_start == req_end {
format!("L{req_start}")
} else {
format!("L{req_start}-L{req_end}")
};
return Ok(ReadResult {
text: format!(
"Requested range {requested} does not overlap {}:{} (lines {}-{}).",
params.display_path, chunk.path, chunk.start_line, chunk.end_line
),
chunk: Some(ChunkReadTarget {
status: ChunkReadStatus::Ok,
selector: selector_ref,
}),
});
}
}
if let Some(target_region) = region {
let masked_source = mask_chunk_display_source(self.inner.source(), self.inner.language());
let (start, end) = chunk_region_range(chunk, target_region);
let tab_replacement = params.tab_replacement.as_deref().unwrap_or(" ");
let normalize_indent = params.normalize_indent.unwrap_or(false).then(|| {
(
detect_file_indent_char(self.inner.source(), self.inner.tree()),
detect_file_indent_step(self.inner.source(), self.inner.tree()) as usize,
)
});
// Extend the region start to the beginning of the line so that the
// leading indentation of the first line is included. Without this,
// regions whose start_byte is mid-line (e.g. a decorator `@property`
// inside a class) would show the first line without indentation,
// making the normalization inconsistent with subsequent lines.
let display_start = masked_source[..start].rfind('\n').map_or(0, |nl| nl + 1);
let region_text = masked_source
.get(display_start..end)
.unwrap_or_default()
.split('\n')
.map(|line| match normalize_indent {
Some((indent_char, indent_step)) => {
normalize_to_tabs(line, indent_char, indent_step)
},
None => line.replace('\t', tab_replacement),
})
.collect::<Vec<_>>()
.join("\n");
let text = if region_text.is_empty() {
format!("{selector_ref}\n\n[Empty @{} region]", target_region.as_str())
} else {
format!("{selector_ref}\n\n{region_text}")
};
return Ok(ReadResult {
text,
chunk: Some(ChunkReadTarget { status: ChunkReadStatus::Ok, selector: selector_ref }),
});
}
Ok(ReadResult {
text: self.render(RenderParams {
chunk_path: Some(chunk.path.clone()),
title: format!("{}:{}", params.display_path, chunk.path),
language_tag: params.language_tag.clone(),
visible_range: None,
render_children_only: false,
omit_checksum: params.omit_checksum,
anchor_style: params.anchor_style,
show_leaf_preview: true,
tab_replacement: params.tab_replacement,
normalize_indent: params.normalize_indent,
focused_paths: None,
}),
chunk: Some(ChunkReadTarget { status: ChunkReadStatus::Ok, selector: selector_ref }),
})
}
/// Prefix a grep line with `display_path` and the chunk path for
/// `line_number`, when known.
#[napi]
pub fn format_grep_line(&self, display_path: String, line_number: u32, line: String) -> String {
let chunk_path = self.inner.line_to_containing_chunk_path(line_number);
let location = chunk_path.map_or_else(
|| display_path.clone(),
|chunk_path| {
if chunk_path.is_empty() {
display_path.clone()
} else {
format!("{display_path}:{chunk_path}")
}
},
);
format!("{location}>{line_number}|{line}")
}
/// Apply batch edits, re-parse, write files, and return updated state and
/// messaging.
#[napi]
pub fn apply_edits(&self, params: EditParams) -> Result<EditResult> {
crate::chunk::edit::apply_edits(self, &params).map_err(Error::from_reason)
}
}
#[derive(Clone)]
struct ParsedChunkReadPath {
selector: Option<String>,
crc: Option<String>,
region: Option<ChunkRegion>,
}
fn normalize_language(language: &str) -> String {
language.trim().to_ascii_lowercase()
}
fn chunk_info(chunk: &ChunkNode) -> ChunkInfo {
ChunkInfo {
path: chunk.path.clone(),
identifier: chunk.identifier.clone(),
checksum: chunk.checksum.clone(),
start_line: chunk.start_line,
end_line: chunk.end_line,
leaf: chunk.leaf,
}
}
fn chunk_read_path_separator_index(read_path: &str) -> Option<usize> {
if read_path.len() >= 3 {
let bytes = read_path.as_bytes();
if bytes[0].is_ascii_alphabetic() && bytes[1] == b':' && matches!(bytes[2], b'/' | b'\\') {
return read_path[2..].find(':').map(|index| index + 2);
}
}
read_path.find(':')
}
fn parse_chunk_read_path(read_path: &str) -> std::result::Result<ParsedChunkReadPath, String> {
let raw_selector =
chunk_read_path_separator_index(read_path).map(|index| &read_path[(index + 1)..]);
let ParsedSelector { selector, crc, region, .. } =
split_selector_crc_and_region(raw_selector, None, None)?;
Ok(ParsedChunkReadPath { selector, crc, region })
}
fn parse_visible_line_range(selector: &str) -> Option<VisibleLineRange> {
let captures = LINE_RANGE_SELECTOR_REGEX.captures(selector)?;
let start_line = captures.get(1)?.as_str().parse::<u32>().ok()?.max(1);
let end_line = captures
.get(2)
.and_then(|m| m.as_str().parse::<u32>().ok())
.unwrap_or(start_line)
.max(start_line);
Some(VisibleLineRange { start_line, end_line })
}
pub fn mask_chunk_display_source(source: &str, language: &str) -> String {
if language != "tlaplus" {
return source.to_string();
}
let lines = source.split('\n').collect::<Vec<_>>();
let mut masked = lines
.iter()
.map(|line| (*line).to_string())
.collect::<Vec<_>>();
let mut index = 0usize;
while index < lines.len() {
if !TLAPLUS_BEGIN_TRANSLATION_REGEX.is_match(lines[index]) {
index += 1;
continue;
}
let begin_index = index;
let mut end_index = begin_index + 1;
while end_index < lines.len() && !TLAPLUS_END_TRANSLATION_REGEX.is_match(lines[end_index]) {
end_index += 1;
}
if begin_index + 1 < lines.len() {
masked[begin_index + 1] = "\\* [translation hidden]".to_string();
for line in masked
.iter_mut()
.take(end_index.min(lines.len()))
.skip(begin_index + 2)
{
line.clear();
}
}
index = end_index + 1;
}
masked.join("\n")
}
-403
View File
@@ -1,403 +0,0 @@
//! Shared types for the chunk-tree system.
use napi_derive::napi;
use crate::chunk::{kind::ChunkKind, state::ChunkState};
#[derive(Clone)]
pub struct ChunkNode {
pub path: String,
pub identifier: Option<String>,
pub kind: ChunkKind,
pub leaf: bool,
/// For virtual chunks (for example `theirs` branches in conflicts), content
/// that is rendered instead of slicing `source`.
pub virtual_content: Option<String>,
pub parent_path: Option<String>,
pub children: Vec<String>,
pub signature: Option<String>,
pub start_line: u32,
pub end_line: u32,
pub line_count: u32,
pub start_byte: u32,
pub end_byte: u32,
/// Start byte of the semantic declaration used for checksums. This can be
/// later than `start_byte` when the chunk absorbs attached leading trivia
/// such as doc comments or attributes.
pub checksum_start_byte: u32,
/// End byte of the prologue region. `None` means the chunk does not expose
/// regions.
pub prologue_end_byte: Option<u32>,
/// Start byte of the epilogue region. `None` means the chunk does not expose
/// regions.
pub epilogue_start_byte: Option<u32>,
pub checksum: String,
pub error: bool,
pub indent: u32,
pub indent_char: String,
/// True for group-candidate chunks (e.g. `stmts`, `imports`, `decls`) that
/// represent an ordered list of similar items. Append/prepend is valid on
/// these even when they are leaf nodes.
pub group: bool,
}
#[derive(Clone)]
pub struct ChunkTree {
pub language: String,
pub checksum: String,
pub line_count: u32,
pub parse_errors: u32,
/// 1-indexed line numbers of tree-sitter ERROR / MISSING nodes surfaced
/// during parsing. Used to focus error messages around failure locations
/// when an edit introduces a parse error far from the edit target.
pub parse_error_lines: Vec<u32>,
pub fallback: bool,
pub root_path: String,
pub root_children: Vec<String>,
pub chunks: Vec<ChunkNode>,
}
/// Summary of a single chunk node for tool output and navigation.
#[derive(Clone)]
#[napi(object)]
pub struct ChunkInfo {
/// Chunk selector path within the tree.
pub path: String,
/// Bare chunk identifier (without kind prefix), if available.
pub identifier: Option<String>,
/// Stable checksum anchor for this chunk.
pub checksum: String,
/// 1-based start line in the source file (inclusive).
pub start_line: u32,
/// 1-based end line in the source file (inclusive).
pub end_line: u32,
/// Whether this node is a leaf (no child chunks).
pub leaf: bool,
}
/// Result of resolving a chunk read request against the tree.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
#[napi(string_enum)]
pub enum ChunkReadStatus {
/// Selector matched a chunk and content was produced.
#[napi(value = "ok")]
Ok,
/// No chunk matched the requested selector.
#[napi(value = "not_found")]
NotFound,
/// Chunk matched but does not support the requested region.
#[napi(value = "unsupported_region")]
UnsupportedRegion,
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
#[napi(string_enum)]
pub enum ChunkRegion {
#[napi(value = "^")]
Head,
#[napi(value = "~")]
Body,
}
impl ChunkRegion {
pub const fn as_str(self) -> &'static str {
match self {
Self::Head => "^",
Self::Body => "~",
}
}
}
/// Structural edit to apply relative to a chunk anchor.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
#[napi(string_enum)]
pub enum ChunkEditOp {
/// Put new content into the targeted region.
#[napi(value = "put")]
Put,
/// Find and replace a literal substring within the targeted region.
#[napi(value = "replace")]
Replace,
/// Remove the targeted region.
#[napi(value = "delete")]
Delete,
/// Insert `content` before the targeted region span.
#[napi(value = "before")]
Before,
/// Insert `content` after the targeted region span.
#[napi(value = "after")]
After,
/// Insert `content` at the start inside the targeted region.
#[napi(value = "prepend")]
Prepend,
/// Insert `content` at the end inside the targeted region.
#[napi(value = "append")]
Append,
}
impl ChunkEditOp {
pub const fn as_str(self) -> &'static str {
match self {
Self::Put => "put",
Self::Replace => "replace",
Self::Delete => "delete",
Self::Before => "before",
Self::After => "after",
Self::Prepend => "prepend",
Self::Append => "append",
}
}
}
/// Outcome of resolving which chunk was read for a `renderRead`-style request.
#[derive(Clone)]
#[napi(object)]
pub struct ChunkReadTarget {
/// Whether the selector matched.
pub status: ChunkReadStatus,
/// Sanitized selector string that was applied.
pub selector: String,
}
/// Inclusive 1-based line range within a source file (used for scoped chunk
/// rendering).
#[derive(Clone)]
#[napi(object)]
pub struct VisibleLineRange {
/// First line to include.
pub start_line: u32,
/// Last line to include.
pub end_line: u32,
}
/// How chunk anchors are formatted in rendered output (name and checksum
/// visibility).
#[derive(Clone, Copy, Default)]
#[napi(string_enum)]
pub enum ChunkAnchorStyle {
/// `[.name#crc]` style anchor.
#[default]
#[napi(value = "full")]
Full,
/// `[.kind#crc]` style anchor (kind is the name prefix before `_`).
#[napi(value = "kind")]
Kind,
/// `[#crc]` style anchor.
#[napi(value = "bare")]
Bare,
/// `[.name]` without checksum.
#[napi(value = "full-omit")]
FullOmit,
/// `[.kind]` without checksum.
#[napi(value = "kind-omit")]
KindOmit,
/// Minimal anchor without name or checksum.
#[napi(value = "none")]
None,
}
impl ChunkAnchorStyle {
pub const fn with_omit_checksum(self, omit: bool) -> Self {
if !omit {
return self;
}
match self {
Self::Full => Self::FullOmit,
Self::Kind => Self::KindOmit,
Self::Bare => Self::None,
Self::FullOmit => Self::FullOmit,
Self::KindOmit => Self::KindOmit,
Self::None => Self::None,
}
}
fn render_i(
&self,
marker: (&str, &str),
indent: &str,
name: &str,
crc: &str,
line_count_suffix: &str,
) -> String {
fn extract_kind(name: &str) -> &str {
name.find('_').map_or_else(|| name, |index| &name[..index])
}
let (open, close) = marker;
match self {
Self::Full => format!("{indent}{open}{name}#{crc}{close}{line_count_suffix}"),
Self::Kind => {
format!(
"{indent}{open}{kind}#{crc}{close}{line_count_suffix}",
kind = extract_kind(name)
)
},
Self::Bare => format!("{indent}{open}#{crc}{close}{line_count_suffix}"),
Self::FullOmit => format!("{indent}{open}{name}{close}{line_count_suffix}"),
Self::KindOmit => {
format!("{indent}{open}{kind}{close}{line_count_suffix}", kind = extract_kind(name))
},
Self::None => String::new(),
}
}
/// Render an opening anchor: `{prefix}@{name}#{crc}`.
pub fn render(&self, prefix: &str, name: &str, crc: &str) -> String {
self.render_i(("@", ""), prefix, name, crc, "")
}
}
/// How a chunk participates in a focus-scoped render pass.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
#[napi(string_enum)]
pub enum ChunkFocusMode {
/// Emit full content and recurse normally.
#[napi(value = "expanded")]
Expanded,
/// Emit just the opening anchor; do not recurse or emit body.
#[napi(value = "collapsed")]
Collapsed,
/// Emit opening + closing anchors; recurse into focused children only.
/// Interior gap lines between children are suppressed.
#[napi(value = "container")]
Container,
}
/// Path + focus mode pair for the N-API boundary (`HashMap` doesn't cross FFI).
#[derive(Clone, Debug)]
#[napi(object)]
pub struct FocusedPath {
pub path: String,
pub mode: ChunkFocusMode,
}
/// Options for `ChunkState.render`: which subtree to show and how anchors
/// appear.
#[derive(Clone)]
#[napi(object)]
pub struct RenderParams {
/// Path of the chunk to render; `None` uses the tree root.
pub chunk_path: Option<String>,
/// Title line shown above the tree (often the file path).
pub title: String,
/// Optional language label for the header block.
pub language_tag: Option<String>,
/// Restrict output to an inclusive line range of the file.
pub visible_range: Option<VisibleLineRange>,
/// When true, list only direct children instead of a full subtree.
pub render_children_only: bool,
/// Hide checksums in anchors when true.
pub omit_checksum: bool,
/// Anchor formatting style for chunk headers.
pub anchor_style: Option<ChunkAnchorStyle>,
/// Include a one-line preview for leaf chunks.
pub show_leaf_preview: bool,
/// Replace tab characters in displayed previews (e.g. two spaces).
pub tab_replacement: Option<String>,
/// When true, normalize displayed indentation to canonical tabs.
pub normalize_indent: Option<bool>,
/// When set, restrict rendering to these chunks with their specified focus
/// modes. Everything not in this list is skipped.
pub focused_paths: Option<Vec<FocusedPath>>,
}
/// Options for `ChunkState.renderRead`: selector path, display path, and
/// optional line scoping.
#[derive(Clone)]
#[napi(object)]
pub struct ReadRenderParams {
/// Read selector (`sel=...` path, line range, or empty for whole tree).
pub read_path: String,
/// Path shown in titles and error messages (often the file path).
pub display_path: String,
/// Optional language label for the rendered block.
pub language_tag: Option<String>,
/// Hide checksums in rendered anchors.
pub omit_checksum: bool,
/// Anchor formatting style.
pub anchor_style: Option<ChunkAnchorStyle>,
/// Optional absolute file line range to intersect with the resolved chunk.
pub absolute_line_range: Option<VisibleLineRange>,
/// Replace tabs in embedded previews.
pub tab_replacement: Option<String>,
/// When true, normalize displayed indentation to canonical tabs.
pub normalize_indent: Option<bool>,
}
/// Rendered chunk text plus optional resolution metadata for the read request.
#[derive(Clone)]
#[napi(object)]
pub struct ReadResult {
/// Rendered UTF-8 text (chunk tree, notice, or error message).
pub text: String,
/// When a selector was used, whether it matched and which selector applied.
pub chunk: Option<ChunkReadTarget>,
}
/// One edit in a batch; targets a chunk via `sel`/`crc` (with params-level
/// defaults).
#[derive(Clone)]
#[napi(object)]
pub struct EditOperation {
/// Edit kind (replace, delete, insert relative to anchor).
pub op: ChunkEditOp,
/// Chunk selector path; falls back to `EditParams.defaultSelector` when
/// omitted.
pub sel: Option<String>,
/// Optional checksum anchor; falls back to `EditParams.defaultCrc` when
/// omitted.
pub crc: Option<String>,
/// Region to target. When omitted, targets the full chunk.
pub region: Option<ChunkRegion>,
/// Replacement or inserted text (meaning depends on `op`).
pub content: Option<String>,
/// For `replace` op: literal substring to find inside the target chunk.
/// Must match exactly once.
pub find: Option<String>,
}
/// Arguments for applying a batch of chunk edits to a file.
#[derive(Clone)]
#[napi(object)]
pub struct EditParams {
/// Edits to apply in order.
pub operations: Vec<EditOperation>,
/// When true, normalize indentation for response rendering and inserted
/// content. When false, preserve literal tabs/spaces.
pub normalize_indent: Option<bool>,
/// Default chunk selector when an `EditOperation` omits `sel`.
pub default_selector: Option<String>,
/// Default checksum when an `EditOperation` omits `crc`.
pub default_crc: Option<String>,
/// Anchor formatting for rendered response text.
pub anchor_style: Option<ChunkAnchorStyle>,
/// Working directory used to resolve `filePath` and display paths.
pub cwd: String,
/// Path to the source file to edit (often relative to `cwd`).
pub file_path: String,
}
/// Result of applying edits: new parse state plus before/after source and
/// messaging.
#[derive(Clone)]
#[napi(object, object_from_js = false)]
pub struct EditResult {
/// Chunk tree state after applying edits and re-parsing.
pub state: ChunkState,
/// Full file text before edits.
pub diff_before: String,
/// Full file text after edits.
pub diff_after: String,
/// Rendered summary for tooling (hunks, anchors), driven by `anchorStyle`.
pub response_text: String,
/// Whether the on-disk source changed.
pub changed: bool,
/// Whether the updated source re-parsed without fatal issues.
pub parse_valid: bool,
/// Absolute or normalized paths that were written or touched.
pub touched_paths: Vec<String>,
/// Non-fatal issues (e.g. selector warnings) collected during apply.
pub warnings: Vec<String>,
}
-1
View File
@@ -27,7 +27,6 @@ static GLOBAL: MiMalloc = MiMalloc;
pub mod appearance;
pub mod ast;
pub mod chunk;
pub mod clipboard;
pub mod fd;
pub mod fs_cache;
+9
View File
@@ -1,6 +1,15 @@
# Changelog
## [Unreleased]
### Removed
- Removed the `chunk` edit mode, chunk-aware `read` selectors, chunk-aware `grep` rendering, and the `omp read` chunk CLI subcommand
- Removed the `read.prosechunks`, `read.explorechunks`, and `read.anchorstyle` settings
- Removed the underlying `chunk` native module and AST-based chunk schema generation from `pi-natives`
### Fixed
- Fixed `poll` wait duration parsing to fall back to `30s` when the provided value is an empty string
## [14.4.1] - 2026-04-26
### Breaking Changes
-1
View File
@@ -49,7 +49,6 @@ const commands: CommandEntry[] = [
{ name: "config", load: () => import("./commands/config").then(m => m.default) },
{ name: "grep", load: () => import("./commands/grep").then(m => m.default) },
{ name: "grievances", load: () => import("./commands/grievances").then(m => m.default) },
{ name: "read", load: () => import("./commands/read").then(m => m.default) },
{ name: "jupyter", load: () => import("./commands/jupyter").then(m => m.default) },
{ name: "plugin", load: () => import("./commands/plugin").then(m => m.default) },
{ name: "setup", load: () => import("./commands/setup").then(m => m.default) },
-67
View File
@@ -1,67 +0,0 @@
/**
* Read CLI command handler.
*
* Handles `omp read` subcommand — emits chunk-mode read output for files,
* and delegates URL reads through the read tool pipeline.
*/
import * as path from "node:path";
import chalk from "chalk";
import { Settings } from "../config/settings";
import { formatChunkedRead, resolveAnchorStyle } from "../edit/modes/chunk";
import { getLanguageFromPath } from "../modes/theme/theme";
import type { ToolSession } from "../tools";
import { parseReadUrlTarget } from "../tools/fetch";
import { ReadTool } from "../tools/read";
export interface ReadCommandArgs {
path: string;
sel?: string;
}
function createCliReadSession(cwd: string, settings: Settings): ToolSession {
return {
cwd,
hasUI: false,
hasEditTool: true,
getSessionFile: () => null,
getSessionSpawns: () => null,
settings,
};
}
export async function runReadCommand(cmd: ReadCommandArgs): Promise<void> {
const cwd = process.cwd();
const parsedUrlTarget = parseReadUrlTarget(cmd.path, cmd.sel);
if (parsedUrlTarget) {
const settings = await Settings.init({ cwd });
const tool = new ReadTool(createCliReadSession(cwd, settings));
const result = await tool.execute("cli-read", { path: cmd.path, sel: cmd.sel });
const text = result.content.find((content): content is { type: "text"; text: string } => content.type === "text");
console.log(text?.text ?? "");
return;
}
const filePath = path.resolve(cmd.path);
const file = Bun.file(filePath);
if (!(await file.exists())) {
console.error(chalk.red(`Error: File not found: ${cmd.path}`));
process.exit(1);
}
const readPath = cmd.sel ? `${filePath}:${cmd.sel}` : filePath;
const language = getLanguageFromPath(filePath);
try {
const result = await formatChunkedRead({
filePath,
readPath,
cwd,
language,
anchorStyle: resolveAnchorStyle(),
});
console.log(result.text);
} catch (err) {
console.error(chalk.red(`Error: ${err instanceof Error ? err.message : String(err)}`));
process.exit(1);
}
}
@@ -1,33 +0,0 @@
/**
* Chunk-mode read tool.
*/
import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli";
import { type ReadCommandArgs, runReadCommand } from "../cli/read-cli";
import { initTheme } from "../modes/theme/theme";
export default class Read extends Command {
static description = "Read a file as a chunk tree";
static args = {
path: Args.string({ description: "File path to read", required: true }),
};
static flags = {
sel: Flags.string({
char: "s",
description: "Chunk selector or line range (e.g. class_Foo.fn_bar, L10-L20)",
}),
};
async run(): Promise<void> {
const { args, flags } = await this.parse(Read);
const cmd: ReadCommandArgs = {
path: args.path ?? "",
sel: flags.sel,
};
await initTheme();
await runReadCommand(cmd);
}
}
@@ -1,6 +1,5 @@
import * as fs from "node:fs";
import * as path from "node:path";
import { type ChunkAnchorStyle, formatAnchor } from "@oh-my-pi/pi-natives";
import {
getProjectDir,
getProjectPromptsDir,
@@ -62,35 +61,6 @@ prompt.registerHelper("hline", (lineNum: unknown, content: unknown): string => {
return `${ref}${HASHLINE_CONTENT_SEPARATOR}${text}`;
});
/**
* {{anchor name checksum}} — render a branch anchor tag using the current anchor style.
* Style is resolved from the template context (`anchorStyle`) or defaults to "full".
*/
prompt.registerHelper("anchor", function (this: prompt.TemplateContext, name: string, checksum: string): string {
const style = (this.anchorStyle as ChunkAnchorStyle) ?? "full";
return formatAnchor(name, checksum, style);
});
/**
* {{sel "parent_Name.child_Name"}} — render a chunk path for `sel` fields in examples.
* In `full` style the path is returned as-is (`class_Server.fn_start`).
* In `kind` style each segment is trimmed to its kind prefix (`class.fn`).
* In `bare` style the path is omitted (the model uses only `crc` to identify chunks).
*/
prompt.registerHelper("sel", function (this: prompt.TemplateContext, chunkPath: string): string {
const style = (this.anchorStyle as ChunkAnchorStyle) ?? "full";
if (style === "full") return chunkPath;
if (style === "bare") return "";
// kind: trim each segment to its kind prefix (before the first `_`)
return chunkPath
.split(".")
.map(seg => {
const idx = seg.indexOf("_");
return idx === -1 ? seg : seg.slice(0, idx);
})
.join(".");
});
const INLINE_ARG_SHELL_PATTERN = /\$(?:ARGUMENTS|@(?:\[\d+(?::\d*)?\])?|\d+)/;
const INLINE_ARG_TEMPLATE_PATTERN = /\{\{[\s\S]*?(?:\b(?:arguments|ARGUMENTS|args)\b|\barg\s+[^}]+)[\s\S]*?\}\}/;
@@ -955,12 +955,12 @@ export const SETTINGS_SCHEMA = {
// Edit tool
"edit.mode": {
type: "enum",
values: ["replace", "patch", "hashline", "chunk", "vim", "apply_patch", "atom"] as const,
values: ["replace", "patch", "hashline", "vim", "apply_patch", "atom"] as const,
default: "hashline",
ui: {
tab: "editing",
label: "Edit Mode",
description: "Select the edit tool variant (replace, patch, hashline, chunk, vim, or apply_patch)",
description: "Select the edit tool variant (replace, patch, hashline, vim, or apply_patch)",
},
},
@@ -1046,38 +1046,6 @@ export const SETTINGS_SCHEMA = {
},
},
"read.prosechunks": {
type: "boolean",
default: false,
ui: {
tab: "editing",
label: "Prose Chunks",
description: "Enable chunk rendering for prose files in chunk edit mode",
},
},
"read.explorechunks": {
type: "boolean",
default: false,
ui: {
tab: "editing",
label: "Explore Chunks",
description: "Show chunk tree without checksums for read-only agents like explore",
},
},
"read.anchorstyle": {
type: "enum",
values: ["full", "kind", "bare"],
default: "full",
ui: {
tab: "editing",
label: "Anchor Style",
description: "Render chunk anchors with full names, kind prefixes, or checksum-only tags",
submenu: true,
},
},
// LSP
"lsp.enabled": {
type: "boolean",
+1 -1
View File
@@ -326,7 +326,7 @@ export class Settings {
/**
* Get the edit variant for a specific model.
* Returns "patch", "replace", "hashline", "chunk", "vim", "apply_patch", or null (use global default).
* Returns "patch", "replace", "hashline", "vim", "apply_patch", or null (use global default).
*/
getEditVariantForModel(model: string | undefined): EditMode | null {
if (!model) return null;
-46
View File
@@ -10,7 +10,6 @@ import {
} from "../lsp";
import applyPatchDescription from "../prompts/tools/apply-patch.md" with { type: "text" };
import atomDescription from "../prompts/tools/atom.md" with { type: "text" };
import chunkEditDescription from "../prompts/tools/chunk-edit.md" with { type: "text" };
import hashlineDescription from "../prompts/tools/hashline.md" with { type: "text" };
import patchDescription from "../prompts/tools/patch.md" with { type: "text" };
import replaceDescription from "../prompts/tools/replace.md" with { type: "text" };
@@ -27,15 +26,6 @@ import {
executeAtomSingle,
resolveAtomEntryPaths,
} from "./modes/atom";
import {
type ChunkParams,
type ChunkToolEdit,
chunkEditParamsSchema,
executeChunkSingle,
parseChunkEditPath,
resolveAnchorStyle,
resolveChunkAutoIndent,
} from "./modes/chunk";
import {
executeHashlineSingle,
HashlineMismatchError,
@@ -53,7 +43,6 @@ export * from "./diff";
export * from "./line-hash";
export * from "./modes/apply-patch";
export * from "./modes/atom";
export * from "./modes/chunk";
export * from "./modes/hashline";
export * from "./modes/patch";
export * from "./modes/replace";
@@ -66,7 +55,6 @@ type TInput =
| typeof patchEditSchema
| typeof hashlineEditParamsSchema
| typeof atomEditParamsSchema
| typeof chunkEditParamsSchema
| typeof vimSchema
| typeof applyPatchSchema;
@@ -76,7 +64,6 @@ type EditParams =
| PatchParams
| HashlineParams
| AtomParams
| ChunkParams
| VimParams
| ApplyPatchParams;
type EditToolResultDetails = EditToolDetails | VimToolDetails;
@@ -325,39 +312,6 @@ export class EditTool implements AgentTool<TInput> {
#getModeDefinition(): EditModeDefinition {
return {
chunk: {
description: (session: ToolSession) =>
prompt.render(chunkEditDescription, {
anchorStyle: resolveAnchorStyle(session.settings),
chunkAutoIndent: resolveChunkAutoIndent(),
}),
parameters: chunkEditParamsSchema,
execute: (
tool: EditTool,
params: EditParams,
signal: AbortSignal | undefined,
batchRequest: LspBatchRequest | undefined,
onUpdate?: (partialResult: AgentToolResult<EditToolDetails, TInput>) => void,
) => {
const { edits, path: topPath } = params as ChunkParams & { path?: string };
const resolved = resolveEntryPaths(edits as ChunkToolEdit[], topPath);
const byFile = groupBy(resolved, (e: ChunkToolEdit) => parseChunkEditPath(e.path).filePath);
const entries = [...byFile.entries()].map(([filePath, fileEdits]) => ({
path: filePath,
run: (br: LspBatchRequest | undefined) =>
executeChunkSingle({
session: tool.session,
path: filePath,
edits: fileEdits,
signal,
batchRequest: br,
writethrough: tool.#writethrough,
beginDeferredDiagnosticsForPath: p => tool.#beginDeferredDiagnosticsForPath(p),
}),
}));
return executePerFile(entries, batchRequest, onUpdate);
},
},
patch: {
description: () => prompt.render(patchDescription),
parameters: patchEditSchema,
@@ -667,59 +667,6 @@ export const HASHLINE_BIGRAMS = [
export const HASHLINE_BIGRAMS_COUNT = HASHLINE_BIGRAMS.length;
/**
* 40 common English BPE bigrams used by chunk checksums (`path#checksum`).
* Kept separate from {@link HASHLINE_BIGRAMS} because the chunk checksum
* format is `path#bigram1bigram2` (4 chars from a 1600-code namespace) and
* is independent of the line-anchor format.
*
* Order is stable forever — changing it invalidates every saved chunk path.
*/
export const CHUNK_BIGRAMS = [
"th",
"he",
"in",
"er",
"an",
"re",
"on",
"at",
"en",
"nd",
"ti",
"es",
"or",
"te",
"of",
"ed",
"is",
"it",
"al",
"ar",
"st",
"to",
"nt",
"ng",
"se",
"ha",
"as",
"ou",
"io",
"le",
"ve",
"co",
"me",
"de",
"hi",
"ri",
"ro",
"ic",
"ne",
"ea",
] as const;
export const CHUNK_BIGRAMS_COUNT = CHUNK_BIGRAMS.length;
/**
* Regex source matching exactly one bigram from {@link HASHLINE_BIGRAMS}.
* Used by hashline parsers — keep in sync with the alphabet array above.
@@ -1,832 +0,0 @@
import * as fs from "node:fs/promises";
import * as nodePath from "node:path";
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import { StringEnum } from "@oh-my-pi/pi-ai";
import {
ChunkAnchorStyle,
ChunkEditOp,
type ChunkInfo,
ChunkReadStatus,
type ChunkReadTarget,
ChunkRegion,
ChunkState,
type EditOperation as NativeEditOperation,
} from "@oh-my-pi/pi-natives";
import { $envpos } from "@oh-my-pi/pi-utils";
import { type Static, Type } from "@sinclair/typebox";
import type { BunFile } from "bun";
import { LRUCache } from "lru-cache";
import type { Settings } from "../../config/settings";
import type { WritethroughCallback, WritethroughDeferredHandle } from "../../lsp";
import { getLanguageFromPath } from "../../modes/theme/theme";
import type { ToolSession } from "../../tools";
import { assertEditableFileContent } from "../../tools/auto-generated-guard";
import { invalidateFsScanAfterWrite } from "../../tools/fs-cache-invalidation";
import { outputMeta } from "../../tools/output-meta";
import { enforcePlanModeWrite, resolvePlanPath } from "../../tools/plan-mode-guard";
import { generateUnifiedDiffString } from "../diff";
import { HASHLINE_BIGRAMS } from "../line-hash";
import { detectLineEnding, normalizeToLF, restoreLineEndings, stripBom } from "../normalize";
import type { EditToolDetails, LspBatchRequest } from "../renderer";
export type { ChunkReadTarget };
export type ChunkEditOperation =
| { op: "put"; sel?: string; content: string }
| { op: "delete"; sel?: string }
| { op: "before"; sel?: string; content: string }
| { op: "after"; sel?: string; content: string }
| { op: "prepend"; sel?: string; content: string }
| { op: "append"; sel?: string; content: string };
type ChunkEditResult = {
diffSourceBefore: string;
diffSourceAfter: string;
responseText: string;
changed: boolean;
parseValid: boolean;
touchedPaths: string[];
warnings: string[];
};
export type ParsedChunkReadPath = {
filePath: string;
selector?: string;
};
type ChunkCacheEntry = {
mtimeMs: number;
size: number;
source: string;
state: ChunkState;
};
const validAnchorStyles: Record<string, ChunkAnchorStyle> = {
full: ChunkAnchorStyle.Full,
kind: ChunkAnchorStyle.Kind,
bare: ChunkAnchorStyle.Bare,
};
export function resolveChunkAutoIndent(rawValue = Bun.env.PI_CHUNK_AUTOINDENT): boolean {
if (!rawValue) return true;
const normalized = rawValue.trim().toLowerCase();
switch (normalized) {
case "1":
case "true":
case "yes":
case "on":
return true;
case "0":
case "false":
case "no":
case "off":
return false;
default:
throw new Error(`Invalid PI_CHUNK_AUTOINDENT: ${rawValue}`);
}
}
function getChunkRenderIndentOptions(): {
normalizeIndent: boolean;
tabReplacement: string;
} {
return resolveChunkAutoIndent()
? { normalizeIndent: true, tabReplacement: " " }
: { normalizeIndent: false, tabReplacement: "\t" };
}
export function resolveAnchorStyle(settings?: Settings): ChunkAnchorStyle {
const envStyle = Bun.env.PI_ANCHOR_STYLE;
return (
(envStyle && validAnchorStyles[envStyle]) ||
(settings?.get("read.anchorstyle") as ChunkAnchorStyle | undefined) ||
ChunkAnchorStyle.Full
);
}
const chunkStateCache = new LRUCache<string, ChunkCacheEntry>({
max: $envpos("PI_CHUNK_CACHE_MAX_ENTRIES", 200),
});
export function invalidateChunkCache(filePath: string): void {
chunkStateCache.delete(filePath);
}
type ChunkSourceContext = {
resolvedPath: string;
sourceFile: BunFile;
sourceExists: boolean;
rawContent: string;
chunkLanguage: string | undefined;
};
type ChunkSourceIntent = "read" | "write";
function normalizeLanguage(language: string | undefined): string {
return language?.trim().toLowerCase() || "";
}
function normalizeChunkSource(text: string): string {
return normalizeToLF(stripBom(text).text);
}
function displayPathForFile(filePath: string, cwd: string): string {
const relative = nodePath.relative(cwd, filePath).replace(/\\/g, "/");
return relative && !relative.startsWith("..") ? relative : filePath.replace(/\\/g, "/");
}
function fileLanguageTag(filePath: string, language?: string): string | undefined {
const normalizedLanguage = normalizeLanguage(language);
if (normalizedLanguage.length > 0) return normalizedLanguage;
const ext = nodePath.extname(filePath).replace(/^\./, "").toLowerCase();
return ext.length > 0 ? ext : undefined;
}
async function resolveChunkSourceContext(
session: ToolSession,
path: string,
options?: { intent?: ChunkSourceIntent },
): Promise<ChunkSourceContext> {
const resolvedPath = resolvePlanPath(session, path);
const sourceFile = Bun.file(resolvedPath);
const sourceExists = await sourceFile.exists();
if ((options?.intent ?? "write") === "write") {
enforcePlanModeWrite(session, path, { op: sourceExists ? "update" : "create" });
}
let rawContent = "";
if (sourceExists) {
rawContent = await sourceFile.text();
assertEditableFileContent(rawContent, path);
}
return {
resolvedPath,
sourceFile,
sourceExists,
rawContent,
chunkLanguage: getLanguageFromPath(resolvedPath),
};
}
/**
* Preview-safe loader: read raw source without plan-mode enforcement or
* editable-file guards. Used by streaming diff previews that must not throw
* side-effecting errors while args are still being streamed.
*/
export async function loadChunkSource(params: {
cwd: string;
path: string;
}): Promise<{ resolvedPath: string; rawContent: string; language: string | undefined; exists: boolean }> {
const resolvedPath = nodePath.isAbsolute(params.path) ? params.path : nodePath.resolve(params.cwd, params.path);
const sourceFile = Bun.file(resolvedPath);
const exists = await sourceFile.exists();
const rawContent = exists ? await sourceFile.text() : "";
return { resolvedPath, rawContent, language: getLanguageFromPath(resolvedPath), exists };
}
/**
* Compute a unified diff preview for a chunk edit without applying it.
* Used for streaming previews while args are still arriving. Returns
* `{ error }` on any failure so callers can decide whether to surface it.
*/
export async function computeChunkDiff(
input: { path: string; edits: ChunkToolEdit[] },
cwd: string,
options?: { anchorStyle?: ChunkAnchorStyle; signal?: AbortSignal },
): Promise<{ diff: string; firstChangedLine: number | undefined } | { error: string }> {
try {
options?.signal?.throwIfAborted?.();
const { filePath } = parseChunkEditPath(input.path);
if (!filePath) return { error: "chunk edit path is empty" };
const { resolvedPath, rawContent, language } = await loadChunkSource({ cwd, path: filePath });
options?.signal?.throwIfAborted?.();
const { operations } = normalizeChunkEditOperations(input.edits);
const result = applyChunkEdits({
source: rawContent,
language,
cwd,
filePath: resolvedPath,
operations,
anchorStyle: options?.anchorStyle,
});
options?.signal?.throwIfAborted?.();
if (!result.changed) {
return { diff: "", firstChangedLine: undefined };
}
return generateUnifiedDiffString(result.diffSourceBefore, result.diffSourceAfter);
} catch (err) {
return { error: err instanceof Error ? err.message : String(err) };
}
}
function normalizeChunkRegionSyntax(text: string): string {
return text.replaceAll("@body", "~").replaceAll("@head", "^");
}
function buildChunkEditResult(result: {
diffBefore: string;
diffAfter: string;
responseText: string;
changed: boolean;
parseValid: boolean;
touchedPaths: string[];
warnings: string[];
}): ChunkEditResult {
return {
diffSourceBefore: result.diffBefore,
diffSourceAfter: result.diffAfter,
responseText: result.responseText,
changed: result.changed,
parseValid: result.parseValid,
touchedPaths: result.touchedPaths,
warnings: result.warnings.map(normalizeChunkRegionSyntax),
};
}
function chunkReadPathSeparatorIndex(readPath: string): number {
if (/^[a-zA-Z]:[/\\]/.test(readPath)) {
return readPath.indexOf(":", 2);
}
const urlMatch = readPath.match(/^([a-z][a-z0-9+.-]*):\/\//i);
if (urlMatch) {
const scheme = urlMatch[1].toLowerCase();
const urlPrefixEnd = urlMatch[0].length;
if (scheme === "local") {
const index = readPath.lastIndexOf(":");
return index >= urlPrefixEnd ? index : -1;
}
const pathStart = readPath.indexOf("/", urlPrefixEnd);
if (pathStart === -1) return -1;
const index = readPath.lastIndexOf(":");
return index >= pathStart ? index : -1;
}
return readPath.indexOf(":");
}
export function parseChunkSelector(selector: string | undefined): { selector?: string } {
if (!selector || selector.length === 0) {
return {};
}
return { selector };
}
/** Split a combined `file:selector` path into file path and chunk selector. */
export function parseChunkEditPath(editPath: string | undefined): { filePath: string; selector?: string } {
if (!editPath) return { filePath: "" };
const colonIndex = chunkReadPathSeparatorIndex(editPath);
if (colonIndex === -1) {
return { filePath: editPath };
}
const sel = editPath.slice(colonIndex + 1) || undefined;
return { filePath: editPath.slice(0, colonIndex), selector: sel };
}
export function parseChunkReadPath(readPath: string): ParsedChunkReadPath {
const colonIndex = chunkReadPathSeparatorIndex(readPath);
if (colonIndex === -1) {
return { filePath: readPath };
}
const parsedSelector = parseChunkSelector(readPath.slice(colonIndex + 1) || undefined);
return {
filePath: readPath.slice(0, colonIndex),
selector: parsedSelector.selector,
};
}
export function isChunkReadablePath(readPath: string): boolean {
return parseChunkReadPath(readPath).selector !== undefined;
}
export async function loadChunkStateForFile(filePath: string, language: string | undefined): Promise<ChunkCacheEntry> {
const file = Bun.file(filePath);
const stat = await file.stat();
const cached = chunkStateCache.get(filePath);
if (cached && cached.mtimeMs === stat.mtimeMs && cached.size === stat.size) {
return cached;
}
const source = normalizeChunkSource(await file.text());
const state = ChunkState.parse(source, normalizeLanguage(language));
const entry = { mtimeMs: stat.mtimeMs, size: stat.size, source, state };
chunkStateCache.set(filePath, entry);
return entry;
}
export async function formatChunkedRead(params: {
filePath: string;
readPath: string;
cwd: string;
language?: string;
omitChecksum?: boolean;
anchorStyle?: ChunkAnchorStyle;
absoluteLineRange?: { startLine: number; endLine?: number };
}): Promise<{ text: string; resolvedPath?: string; chunk?: ChunkReadTarget }> {
const { filePath, readPath, cwd, language, omitChecksum = false, anchorStyle, absoluteLineRange } = params;
const normalizedLanguage = normalizeLanguage(language);
const { state } = await loadChunkStateForFile(filePath, normalizedLanguage);
const displayPath = displayPathForFile(filePath, cwd);
const renderIndentOptions = getChunkRenderIndentOptions();
const result = state.renderRead({
readPath,
displayPath,
languageTag: fileLanguageTag(filePath, normalizedLanguage),
omitChecksum,
anchorStyle,
absoluteLineRange: absoluteLineRange
? { startLine: absoluteLineRange.startLine, endLine: absoluteLineRange.endLine ?? absoluteLineRange.startLine }
: undefined,
tabReplacement: renderIndentOptions.tabReplacement,
normalizeIndent: renderIndentOptions.normalizeIndent,
});
return { text: result.text, resolvedPath: filePath, chunk: result.chunk };
}
export type ChunkedGrepMatch = {
displayPath: string;
fileLineCount: number;
chunkPath?: string;
chunkChecksum?: string;
lineNumber: number;
line: string;
};
export async function describeChunkedGrepMatch(params: {
filePath: string;
lineNumber: number;
line: string;
cwd: string;
language?: string;
}): Promise<ChunkedGrepMatch> {
const { filePath, lineNumber, line, cwd, language } = params;
const { state } = await loadChunkStateForFile(filePath, language);
const chunkPath = state.lineToContainingChunkPath(lineNumber) || undefined;
const chunkInfo = chunkPath ? state.chunk(chunkPath) : null;
return {
displayPath: displayPathForFile(filePath, cwd),
fileLineCount: state.lineCount,
chunkPath,
chunkChecksum: chunkInfo?.checksum,
lineNumber,
line,
};
}
const CHUNK_CHECKSUM_BIGRAMS = new Set<string>(HASHLINE_BIGRAMS);
type NativeChunkRegion = "head" | "body";
function isChunkChecksumToken(value: string): boolean {
if (value.length !== 4) return false;
const lower = value.toLowerCase();
return CHUNK_CHECKSUM_BIGRAMS.has(lower.slice(0, 2)) && CHUNK_CHECKSUM_BIGRAMS.has(lower.slice(2, 4));
}
function parseChunkEditSelector(selector: string | undefined): {
selector?: string;
crc?: string;
region?: NativeChunkRegion;
} {
if (!selector) {
return {};
}
let trimmed = selector.trim();
if (trimmed.length === 0) {
return {};
}
let region: NativeChunkRegion | undefined;
const suffix = trimmed.at(-1);
if (suffix === "~" || suffix === "^") {
region = suffix === "~" ? "body" : "head";
trimmed = trimmed.slice(0, -1).trimEnd();
}
let selectorPart = trimmed;
let crc: string | undefined;
const hashIndex = selectorPart.lastIndexOf("#");
if (hashIndex >= 0) {
const suffix = selectorPart.slice(hashIndex + 1).trim();
if (isChunkChecksumToken(suffix)) {
crc = suffix.toLowerCase();
selectorPart = selectorPart.slice(0, hashIndex).trimEnd();
}
} else if (isChunkChecksumToken(selectorPart)) {
crc = selectorPart.toLowerCase();
selectorPart = "";
}
return { selector: selectorPart || undefined, crc, region };
}
type NativeChunkRegionEncoding = "named" | "symbolic";
function toNativeEditRegion(
region: NativeChunkRegion | undefined,
encoding: NativeChunkRegionEncoding,
): NativeEditOperation["region"] | undefined {
if (!region) {
return undefined;
}
if (encoding === "symbolic") {
return region === "body" ? ChunkRegion.Body : ChunkRegion.Head;
}
return region as unknown as NativeEditOperation["region"] | undefined;
}
function toNativeEditOperation(
operation: ChunkEditOperation,
defaultRegion: NativeChunkRegion | undefined,
encoding: NativeChunkRegionEncoding,
): NativeEditOperation {
const { selector, crc, region } = parseChunkEditSelector(operation.sel);
const nativeRegion = toNativeEditRegion(operation.sel === undefined ? (region ?? defaultRegion) : region, encoding);
switch (operation.op) {
case "put":
return {
op: ChunkEditOp.Put,
sel: selector,
crc,
region: nativeRegion,
content: operation.content,
};
case "before":
return { op: ChunkEditOp.Before, sel: selector, crc, region: nativeRegion, content: operation.content };
case "after":
return { op: ChunkEditOp.After, sel: selector, crc, region: nativeRegion, content: operation.content };
case "prepend":
return { op: ChunkEditOp.Prepend, sel: selector, crc, region: nativeRegion, content: operation.content };
case "append":
return { op: ChunkEditOp.Append, sel: selector, crc, region: nativeRegion, content: operation.content };
case "delete":
return { op: ChunkEditOp.Delete, sel: selector, crc, region: nativeRegion };
default: {
const exhaustive: never = operation;
return exhaustive;
}
}
}
function buildNativeChunkEditRequest(
params: { defaultSelector?: string; defaultCrc?: string; operations: ChunkEditOperation[] },
encoding: NativeChunkRegionEncoding,
): Pick<Parameters<ChunkState["applyEdits"]>[0], "operations" | "defaultSelector" | "defaultCrc"> {
const parsedDefaultSelector = parseChunkEditSelector(params.defaultSelector);
const operations = params.operations.map(operation =>
toNativeEditOperation(operation, parsedDefaultSelector.region, encoding),
);
return {
operations,
defaultSelector: parsedDefaultSelector.selector,
defaultCrc: params.defaultCrc ?? parsedDefaultSelector.crc,
};
}
function isChunkRegionEncodingError(error: unknown): error is Error {
return (
error instanceof Error &&
/value `"(body|head|~|\^)"` does not match any variant of enum `ChunkRegion`/.test(error.message)
);
}
export function applyChunkEdits(params: {
source: string;
language?: string;
cwd: string;
filePath: string;
operations: ChunkEditOperation[];
defaultSelector?: string;
defaultCrc?: string;
anchorStyle?: ChunkAnchorStyle;
}): ChunkEditResult {
const normalizedSource = normalizeChunkSource(params.source);
const applyNativeEdits = (encoding: NativeChunkRegionEncoding): ChunkEditResult => {
const request = buildNativeChunkEditRequest(params, encoding);
const state = ChunkState.parse(normalizedSource, normalizeLanguage(params.language));
return buildChunkEditResult(
state.applyEdits({
operations: request.operations,
normalizeIndent: resolveChunkAutoIndent(),
defaultSelector: request.defaultSelector,
defaultCrc: request.defaultCrc,
anchorStyle: params.anchorStyle,
cwd: params.cwd,
filePath: params.filePath,
}),
);
};
try {
return applyNativeEdits("named");
} catch (error) {
if (isChunkRegionEncodingError(error)) {
try {
return applyNativeEdits("symbolic");
} catch (fallbackError) {
if (fallbackError instanceof Error) {
throw new Error(normalizeChunkRegionSyntax(fallbackError.message));
}
throw fallbackError;
}
}
if (error instanceof Error) {
throw new Error(normalizeChunkRegionSyntax(error.message));
}
throw error;
}
}
export async function getChunkInfoForFile(
filePath: string,
language: string | undefined,
chunkPath: string,
): Promise<ChunkInfo | undefined> {
const { state } = await loadChunkStateForFile(filePath, language);
return state.chunk(chunkPath) ?? undefined;
}
export function missingChunkReadTarget(selector: string): ChunkReadTarget {
return { status: ChunkReadStatus.NotFound, selector };
}
export const chunkToolEditSchema = Type.Object(
{
path: Type.Optional(
Type.String({
description: "File path with chunk selector. Examples: 'src/app.ts:fn_foo#thth~', 'src/app.ts:class_Bar'.",
}),
),
write: Type.Optional(
Type.Union([Type.String(), Type.Null()], {
description:
"Write complete new content to the targeted region. Null is rejected; use delete: true for deletion.",
}),
),
delete: Type.Optional(
Type.Boolean({
description: "Explicitly delete the targeted chunk. Must be true; include the current chunk ID.",
}),
),
insert: Type.Optional(
Type.Object(
{
loc: StringEnum(["append", "prepend"] as const),
body: Type.String({ description: "Content to insert." }),
},
{ description: "Insert content relative to the chunk." },
),
),
},
{ additionalProperties: false },
);
export const chunkEditParamsSchema = Type.Object(
{
path: Type.Optional(Type.String({ description: "Default file path used when an edit omits its own `path`" })),
edits: Type.Array(chunkToolEditSchema, {
description: "Chunk edits",
minItems: 1,
}),
},
{ additionalProperties: false },
);
export type ChunkToolEdit = Static<typeof chunkToolEditSchema>;
export type ChunkParams = Static<typeof chunkEditParamsSchema>;
export interface ExecuteChunkSingleOptions {
session: ToolSession;
path: string;
edits: ChunkToolEdit[];
signal?: AbortSignal;
batchRequest?: LspBatchRequest;
writethrough: WritethroughCallback;
beginDeferredDiagnosticsForPath: (path: string) => WritethroughDeferredHandle;
}
/** Auto-correct indentation for content targeting a body region (`~`) when autoIndent is on.
* Handles two patterns:
* 1. Tab-based over-indentation: models include the function's base \t indent.
* 2. Space-based indentation: models use literal spaces instead of \t.
* Returns the corrected content and any warnings. */
function autoCorrectBodyIndent(content: string, index: number): { content: string; warnings: string[] } {
const warnings: string[] = [];
if (!content || !resolveChunkAutoIndent()) return { content, warnings };
const lines = content.split("\n");
const nonEmpty = lines.filter(l => l.length > 0);
if (nonEmpty.length <= 1) return { content, warnings };
// 1. Tab-based over-indentation: strip common leading tabs.
const minTabs = Math.min(...nonEmpty.map(l => l.match(/^\t*/)?.[0].length ?? 0));
if (minTabs >= 1) {
const fixed = lines.map(l => (l.length === 0 ? l : l.slice(minTabs))).join("\n");
warnings.push(
`Edit ${index + 1}: auto-corrected body indentation \u2014 stripped ${minTabs} leading tab(s). When writing to \`~\`, write at column 0; the tool adds the function's base indent.`,
);
return { content: fixed, warnings };
}
// 2. Space-based indentation: strip common leading spaces and convert to tabs.
const spaceIndents = nonEmpty.map(l => l.match(/^ */)?.[0].length ?? 0);
const minSpaces = Math.min(...spaceIndents);
if (minSpaces >= 2) {
const indentDiffs = spaceIndents.map(s => s - minSpaces).filter(d => d > 0);
const indentUnit = indentDiffs.length > 0 ? Math.min(...indentDiffs) : 4;
const unit = indentUnit >= 2 && indentUnit <= 8 ? indentUnit : 4;
const fixed = lines
.map(line => {
if (line.length === 0) return line;
const stripped = line.slice(minSpaces);
const leadingSpaces = stripped.match(/^ */)?.[0].length ?? 0;
const tabs = Math.floor(leadingSpaces / unit);
const rem = leadingSpaces % unit;
return "\t".repeat(tabs) + " ".repeat(rem) + stripped.slice(leadingSpaces);
})
.join("\n");
warnings.push(
`Edit ${index + 1}: auto-converted space indentation to tabs \u2014 stripped ${minSpaces} common leading spaces and converted ${unit}-space indent to tabs. When auto-indent is on, use \\t for indentation.`,
);
return { content: fixed, warnings };
}
return { content, warnings };
}
function chunkEditOperationFields(edit: ChunkToolEdit): string[] {
const fields: string[] = [];
if (edit.write !== undefined) fields.push("write");
if (edit.insert != null) fields.push("insert");
if (edit.delete === true) fields.push("delete");
return fields;
}
function assertSingleChunkOperation(edit: ChunkToolEdit, index: number): string {
const fields = chunkEditOperationFields(edit);
if (fields.length === 0) {
throw new Error(
`Edit ${index + 1}: no operation specified. Use write:"..." to replace, insert:{loc,body} to insert, or delete:true to delete. Use the open tool to inspect chunks.`,
);
}
if (fields.length > 1) {
throw new Error(
`Edit ${index + 1}: multiple operation fields set (${fields.join(", ")}). Each chunk edit entry must have exactly one operation.`,
);
}
return fields[0];
}
function normalizeChunkEditOperations(edits: ChunkToolEdit[]): {
operations: ChunkEditOperation[];
warnings: string[];
} {
const warnings: string[] = [];
const operations = edits.map((edit, index): ChunkEditOperation => {
const { selector } = parseChunkEditPath(edit.path);
const operation = assertSingleChunkOperation(edit, index);
if (operation === "write") {
if (edit.write === null) {
throw new Error(
`Edit ${index + 1}: write:null no longer deletes chunks. Use delete:true to delete, or open the chunk to inspect its content without modifying the file.`,
);
}
if (typeof edit.write !== "string") {
throw new Error(`Edit ${index + 1}: write must be a string.`);
}
if (edit.write.length === 0) {
throw new Error(
`Edit ${index + 1}: write:"" is a destructive empty replacement. Use delete:true to delete the chunk, or open the chunk to inspect its content without modifying the file.`,
);
}
let writeContent = edit.write;
if (selector?.endsWith("~")) {
const corrected = autoCorrectBodyIndent(writeContent, index);
writeContent = corrected.content;
warnings.push(...corrected.warnings);
}
return { op: "put", sel: selector, content: writeContent };
}
if (operation === "insert") {
if (edit.insert == null || typeof edit.insert.body !== "string" || edit.insert.body.length === 0) {
throw new Error(`Edit ${index + 1}: insert.body must be a non-empty string.`);
}
const op = edit.insert.loc === "prepend" ? "before" : "after";
let insertContent = edit.insert.body;
if (selector?.endsWith("~")) {
const corrected = autoCorrectBodyIndent(insertContent, index);
insertContent = corrected.content;
warnings.push(...corrected.warnings);
}
return { op, sel: selector, content: insertContent };
}
if (operation !== "delete") {
throw new Error(`Edit ${index + 1}: unsupported chunk edit operation "${operation}".`);
}
return { op: "delete", sel: selector };
});
return { operations, warnings };
}
async function writeChunkResult(params: {
result: ChunkEditResult;
resolvedPath: string;
sourceFile: BunFile;
sourceText: string;
sourceExists: boolean;
signal?: AbortSignal;
batchRequest?: LspBatchRequest;
writethrough: WritethroughCallback;
beginDeferredDiagnosticsForPath: (path: string) => WritethroughDeferredHandle;
}): Promise<AgentToolResult<EditToolDetails, typeof chunkEditParamsSchema>> {
const {
result,
resolvedPath,
sourceFile,
sourceText,
sourceExists,
signal,
batchRequest,
writethrough,
beginDeferredDiagnosticsForPath,
} = params;
const { bom, text } = stripBom(sourceText);
const originalEnding = detectLineEnding(text);
const finalContent = bom + restoreLineEndings(result.diffSourceAfter, originalEnding);
const diagnostics = await writethrough(resolvedPath, finalContent, signal, sourceFile, batchRequest, dst =>
dst === resolvedPath ? beginDeferredDiagnosticsForPath(resolvedPath) : undefined,
);
invalidateFsScanAfterWrite(resolvedPath);
const diffResult = generateUnifiedDiffString(result.diffSourceBefore, result.diffSourceAfter);
const warningsBlock = result.warnings.length > 0 ? `\n\nWarnings:\n${result.warnings.join("\n")}` : "";
const meta = outputMeta()
.diagnostics(diagnostics?.summary ?? "", diagnostics?.messages ?? [])
.get();
return {
content: [{ type: "text", text: `${result.responseText}${warningsBlock}` }],
details: {
diff: diffResult.diff,
firstChangedLine: diffResult.firstChangedLine,
diagnostics,
op: sourceExists ? "update" : "create",
meta,
},
};
}
export async function executeChunkSingle(
options: ExecuteChunkSingleOptions,
): Promise<AgentToolResult<EditToolDetails, typeof chunkEditParamsSchema>> {
const { session, path, edits, signal, batchRequest, writethrough, beginDeferredDiagnosticsForPath } = options;
const { resolvedPath, sourceFile, sourceExists, rawContent, chunkLanguage } = await resolveChunkSourceContext(
session,
path,
{ intent: "write" },
);
const parentDir = nodePath.dirname(resolvedPath);
if (parentDir && parentDir !== ".") {
await fs.mkdir(parentDir, { recursive: true });
}
const { operations: normalizedOperations, warnings: normWarnings } = normalizeChunkEditOperations(edits);
if (!sourceExists && normalizedOperations.some(op => op.sel)) {
throw new Error(
`File does not exist: ${path}. Cannot resolve chunk selectors on a non-existent file. Use the write tool to create a new file, or check the path for typos.`,
);
}
const chunkResult = applyChunkEdits({
source: rawContent,
language: chunkLanguage,
cwd: session.cwd,
filePath: resolvedPath,
operations: normalizedOperations,
anchorStyle: resolveAnchorStyle(session.settings),
});
chunkResult.warnings.push(...normWarnings);
if (!chunkResult.changed) {
const warningsBlock = chunkResult.warnings.length > 0 ? `\n\nWarnings:\n${chunkResult.warnings.join("\n")}` : "";
return {
content: [{ type: "text", text: `[No changes needed — content already matches.]${warningsBlock}` }],
details: {
diff: "",
op: sourceExists ? "update" : "create",
meta: outputMeta().get(),
},
};
}
return writeChunkResult({
result: chunkResult,
resolvedPath,
sourceFile,
sourceText: rawContent,
sourceExists,
signal,
batchRequest,
writethrough,
beginDeferredDiagnosticsForPath,
});
}
+5 -35
View File
@@ -29,7 +29,6 @@ import type { VimToolDetails } from "../vim/types";
import type { DiffError, DiffResult } from "./diff";
import { expandApplyPatchToEntries, expandApplyPatchToPreviewEntries } from "./modes/apply-patch";
import type { Operation, PatchEditEntry } from "./modes/patch";
import type { PerFileDiffPreview } from "./streaming";
// ═══════════════════════════════════════════════════════════════════════════
// LSP Batching
@@ -94,7 +93,7 @@ interface EditRenderArgs {
*/
previewDiff?: string;
__partialJson?: string;
// Hashline / chunk mode fields
// Hashline mode fields
edits?: EditRenderEntry[];
}
@@ -141,8 +140,6 @@ export interface EditRenderContext {
editMode?: EditMode;
/** Pre-computed diff preview (computed before tool executes) */
editDiffPreview?: DiffResult | DiffError;
/** Multi-file streaming diff preview (chunk edits spanning several files) */
perFileDiffPreview?: PerFileDiffPreview[];
/** Function to render diff text with syntax highlighting */
renderDiff?: (diffText: string, options?: { filePath?: string }) => string;
}
@@ -151,11 +148,9 @@ const EDIT_STREAMING_PREVIEW_LINES = 12;
const CALL_TEXT_PREVIEW_LINES = 6;
const CALL_TEXT_PREVIEW_WIDTH = 80;
/** Extract file path from an edit entry's path (handles chunk's file:selector format). */
/** Extract file path from an edit entry. */
function filePathFromEditEntry(p: string | undefined): string | undefined {
if (!p) return undefined;
const ci = /^[a-zA-Z]:[/\\]/.test(p) ? p.indexOf(":", 2) : p.indexOf(":");
return ci === -1 ? p : p.slice(0, ci);
return p ?? undefined;
}
function decodePartialJsonStringFragment(fragment: string): string {
@@ -280,32 +275,7 @@ function formatMetadataLine(lineCount: number | null, language: string | undefin
return uiTheme.fg("dim", `${icon}`);
}
function formatMultiFileStreamingDiff(previews: PerFileDiffPreview[], uiTheme: Theme): string {
const parts: string[] = [];
for (const preview of previews) {
if (!preview.diff && !preview.error) continue;
const header = uiTheme.fg("dim", `\n\n── ${shortenPath(preview.path)} ──`);
if (preview.error) {
parts.push(`${header}\n${uiTheme.fg("error", replaceTabs(preview.error))}`);
continue;
}
if (preview.diff) {
parts.push(`${header}${formatStreamingDiff(preview.diff, preview.path, uiTheme, "preview")}`);
}
}
return parts.join("");
}
function getCallPreview(
args: EditRenderArgs,
rawPath: string,
uiTheme: Theme,
renderContext: EditRenderContext | undefined,
): string {
const multi = renderContext?.perFileDiffPreview;
if (multi && multi.length > 0 && multi.some(p => p.diff || p.error)) {
return formatMultiFileStreamingDiff(multi, uiTheme);
}
function getCallPreview(args: EditRenderArgs, rawPath: string, uiTheme: Theme): string {
if (args.previewDiff) {
return formatStreamingDiff(args.previewDiff, rawPath, uiTheme, "preview");
}
@@ -438,7 +408,7 @@ export const editToolRenderer = {
if (fileCount > 1) {
text += uiTheme.fg("dim", ` (+${fileCount - 1} more)`);
}
text += getCallPreview(editArgs, rawPath, uiTheme, renderContext);
text += getCallPreview(editArgs, rawPath, uiTheme);
if (applyPatchSummary?.error) {
text += `\n\n${uiTheme.fg("error", truncateToWidth(replaceTabs(applyPatchSummary.error), CALL_TEXT_PREVIEW_WIDTH))}`;
}
@@ -16,7 +16,6 @@ import type { Theme } from "../modes/theme/theme";
import { type EditMode, resolveEditMode } from "../utils/edit-mode";
import { computeEditDiff, type DiffError, type DiffResult } from "./diff";
import { expandApplyPatchToEntries, expandApplyPatchToPreviewEntries } from "./modes/apply-patch";
import { type ChunkToolEdit, computeChunkDiff, parseChunkEditPath } from "./modes/chunk";
import { computeHashlineDiff, type HashlineToolEdit } from "./modes/hashline";
import { computePatchDiff, type PatchEditEntry } from "./modes/patch";
import type { ReplaceEditEntry } from "./modes/replace";
@@ -226,70 +225,6 @@ const hashlineStrategy: EditStreamingStrategy<HashlineArgs> = {
},
};
interface ChunkArgs {
path?: string;
edits?: ChunkToolEdit[];
__partialJson?: string;
}
const chunkStrategy: EditStreamingStrategy<ChunkArgs> = {
extractCompleteEdits(args, partialJson) {
if (!args?.edits) return args;
let edits = dropIncompleteLastEdit(args.edits, partialJson, "edits");
// Extra guard: if partial JSON still contains `":nu` / `":nul` (partial
// `null` literals), `partial-json` may have already surfaced the last
// entry with `write === null`. When that entry's `}` hasn't closed
// yet, it has already been dropped above. But if dropping was not
// triggered (e.g. list still open and no new `{` after), also drop the
// trailing null-write entry so the preview does not flicker with an
// error for an incomplete string/null literal.
if (partialJson && edits.length > 0) {
const last = edits[edits.length - 1] as Partial<ChunkToolEdit> | undefined;
const endsInPartialNull = /:\s*nu?l?\s*$/.test(partialJson.trimEnd());
if (last && endsInPartialNull && last.write === null) {
edits = edits.slice(0, -1);
}
}
return { ...args, edits };
},
async computeDiffPreview(args, ctx) {
const edits = args.edits ?? [];
if (edits.length === 0) return null;
// Group edits by file path
const groups = new Map<string, ChunkToolEdit[]>();
const fileOrder: string[] = [];
for (const edit of edits) {
if (!edit) continue;
const editPath = edit.path ?? args.path;
if (!editPath) continue;
const { filePath } = parseChunkEditPath(editPath);
if (!filePath) continue;
let bucket = groups.get(filePath);
if (!bucket) {
bucket = [];
groups.set(filePath, bucket);
fileOrder.push(filePath);
}
bucket.push({ ...edit, path: editPath });
}
if (fileOrder.length === 0) return null;
const MAX_FILES = 5;
const selected = fileOrder.slice(0, MAX_FILES);
const previews: PerFileDiffPreview[] = [];
for (const filePath of selected) {
ctx.signal.throwIfAborted();
const fileEdits = groups.get(filePath) ?? [];
const result = await computeChunkDiff({ path: filePath, edits: fileEdits }, ctx.cwd, { signal: ctx.signal });
previews.push(toPerFilePreview(filePath, result));
}
return previews;
},
renderStreamingFallback() {
return "";
},
};
interface ApplyPatchArgs {
input?: string;
}
@@ -365,7 +300,6 @@ export const EDIT_MODE_STRATEGIES: Record<EditMode, EditStreamingStrategy<unknow
replace: replaceStrategy as EditStreamingStrategy<unknown>,
patch: patchStrategy as EditStreamingStrategy<unknown>,
hashline: hashlineStrategy as EditStreamingStrategy<unknown>,
chunk: chunkStrategy as EditStreamingStrategy<unknown>,
apply_patch: applyPatchStrategy as EditStreamingStrategy<unknown>,
vim: vimStrategy,
atom: atomStrategy as EditStreamingStrategy<unknown>,
@@ -298,11 +298,6 @@ const OPTION_PROVIDERS: Partial<Record<SettingPath, OptionProvider>> = {
{ value: "1000", label: "1000 lines" },
{ value: "5000", label: "5000 lines" },
],
"read.anchorstyle": [
{ value: "full", label: "Full", description: "Show the kind prefix and identifier" },
{ value: "kind", label: "Kind", description: "Show only the kind prefix plus checksum" },
{ value: "bare", label: "Bare", description: "Show only the checksum" },
],
// Todo auto-clear delay
"tasks.todoClearDelay": [
{ value: "0", label: "Instant" },
@@ -109,7 +109,7 @@ export class ToolExecutionComponent extends Container {
isError?: boolean;
details?: any;
};
// Edit preview state (single-file for legacy modes, multi-file for chunk)
// Edit preview state
#editMode?: EditMode;
#editDiffPreview?: PerFileDiffPreview[];
#editDiffScheduleTimer?: NodeJS.Timeout;
@@ -639,10 +639,7 @@ export class ToolExecutionComponent extends Container {
return this.#args;
}
// Single-file previews feed the existing `previewDiff` channel consumed
// by `formatStreamingDiff` in the renderer. Multi-file previews are
// piped via `renderContext.perFileDiffPreview`, so the args we hand to
// `renderCall` only need the first file's diff to preserve prior
// single-file behavior.
// by `formatStreamingDiff` in the renderer.
const first = previews[0];
if (!first?.diff) {
return this.#args;
@@ -687,9 +684,6 @@ export class ToolExecutionComponent extends Container {
? { error: first.error }
: { diff: first.diff ?? "", firstChangedLine: first.firstChangedLine };
}
if (previews.length > 1) {
context.perFileDiffPreview = previews;
}
}
context.renderDiff = renderDiff;
}
@@ -1,158 +0,0 @@
Edits files via syntax-aware chunks. Use `read(path="file.ts")` to read and discover chunks before editing.
- `read` is the canonical read path for chunk source and `sel="?"` tree listings.
- `write` rewrites the entire targeted region — best for most edits.
- `insert` adds content before/after a chunk.
- `delete` deletes a targeted chunk and must be explicit.
Call format: `{"edits": [{"path": "file:chunk#ID~", "write": "new body"}, …]}`
<rules>
- **MUST** inspect first with `read`. Never invent chunk paths or IDs. Copy them from the latest `read` output or edit response.
- `path` format: `file:selector` — e.g. `src/app.ts:fn_foo#thth~`. Append `~` for body, `^` for head, or nothing for the whole chunk. Include `#ID` for `write`/`delete`.
- If the exact chunk path is unclear, run `read(path="file", sel="?")` and copy a selector from that listing.
{{#if chunkAutoIndent}}
- Use `\t` for indentation in `content`. Write content at indent-level 0 — the tool re-indents it to match the chunk's position in the file. For example, to replace `~` of a method, write the body starting at column 0:
```
content: "if (x) {\n\treturn true;\n}"
```
The tool adds the correct base indent automatically. Never manually pad with the chunk's own indentation.
Multiple sibling body lines at the same level all start at column 0: `"print(a)\nprint(b)\nprint(c)\n"`. Only use `\t` when nesting deeper (e.g. `"if cond:\n\tinner\nouter\n"`).
Before applying the target's base indent, the tool strips any common leading whitespace shared by all non-empty `write` lines as a safety net. Do not rely on that cleanup for mixed indentation; write `~` bodies at column 0 and use one `\t` per relative nesting level.
Multi-line replacements use the same relative-indentation model: the replacement text is dedented, then re-indented to the matched source line. Do not include the chunk's base indentation in replacement text.
**Common mistake** when replacing `~` of a function body: do NOT include the function's own indentation.
Wrong: `"if b == 0:\n\t\treturn None\n\treturn a / b\n"` — adds the function's base `\t` to every line.
Correct: `"if b == 0:\n\treturn None\nreturn a / b\n"` — `if` and `return a / b` at column 0, only `return None` gets `\t` for nesting.
{{else}}
- Match the file's literal tabs/spaces in `content`. Do not convert indentation to canonical `\t`.
- Write content at indent-level 0 relative to the target region. For example, to replace `~` of a method, write:
```
content: "if (x) {\n return true;\n}"
```
The tool adds the correct base indent automatically, then preserves the tabs/spaces you used inside the snippet. Never manually pad with the chunk's own indentation.
Before applying the target's base indent, the tool strips any common leading whitespace shared by all non-empty `write` lines as a safety net. Do not rely on that cleanup for mixed indentation; write `~` bodies at column 0.
Multi-line replacements use the same relative-indentation model: the replacement text is dedented, then re-indented to the matched source line. Do not include the chunk's base indentation in replacement text.
{{/if}}
- Region suffixes only apply to chunks with a real head/body boundary (classes, functions, impl blocks, and similar containers). On code leaf chunks (enum variants, fields, single statements, and compound statements like `if`/`for`/`while`/`match`/`try`), `~` and `^` are rejected. Use the unsuffixed selector and supply the complete replacement content, or edit the parent container's `~` body.
- Unsuffixed `write` on a leaf chunk uses your content verbatim after normal replacement; it is not a body-region rewrite. Include the exact indentation and punctuation the leaf needs in the file.
- `^` head writes and `~` body writes use the same base-indent model: write content at column 0 relative to the target region, and the tool applies the chunk's file indentation.
- `write` and `delete` require the current ID. `prepend`/`append` do not.
- **IDs change after every edit.** The edit response always carries the new IDs — use those for the next call or run `read(path="file", sel="?")` to refresh. Never reuse an ID from before the latest edit.
- Same-file edit batches are transactional: if any operation in that file fails, no changes from that file's batch are saved. Multi-file edit calls run per file, so a later file error does not roll back earlier files that already succeeded.
</rules>
<critical>
You **MUST** use the narrowest region that covers your change. Putting without a region overwrites the **entire chunk including leading comments, decorators, and attributes** — omitting them from `content` deletes them.
**`put` is total, not surgical.** The `content` you supply becomes the *complete* new content for the targeted region. Everything in the original region that you omit from `content` is deleted. Before using `put` on any chunk's `~`, verify the chunk does not contain children you intend to keep. If a chunk spans hundreds of lines and your change touches only a few, target a specific child chunk — not the parent.
**Group chunks (`stmts_*`, `imports_*`, `decls_*`) are containers.** They hold many sibling items (test functions, import statements, declarations). `put` on a group chunk's `~` overwrites **all** of its children. To edit one item inside a group, target that item's own chunk path. If no child chunk exists, use the specific child's chunk selector from `read` output — do not `put` the parent group.
</critical>
<regions>
In `read` output, lines marked `^` between the line number and `|` are **head** lines (doc comments, attributes/decorators, signature). Lines without `^` are **body** lines. Use this to decide which region to target:
- `fn_foo#ID~` — **body only (the default choice for most edits).** Head lines (`^`) are preserved automatically — doc comments, attributes, and signature stay untouched. On code leaf chunks, this is rejected because there is no safe body boundary.
- `fn_foo#ID^` — head only (decorators, attributes, doc comments, signature, opening delimiter). Body stays untouched.
- `fn_foo#ID` — entire chunk including leading trivia. **You must include doc comments and attributes in `content`; omitting them deletes them.**
- `chunk~` + `append`/`prepend` inserts *inside* the container. `chunk` + `append`/`prepend` inserts *outside*. Appending to a container without `~` emits a warning because it lands after the closing delimiter, not before it.
**Note on leading trivia:** whether a decorator/doc comment belongs to `^` depends on the parser. In Rust and Python, attributes and decorators are attached to the function chunk, so `^` covers them. In TypeScript/JavaScript, a `@decorator` + `/** jsdoc */` block immediately above a method often surfaces as a **separate sibling chunk** (shown as `chunk#ID` in the `?` listing) rather than as part of the function's `^`. JSDoc directly above a plain function is more likely to be absorbed into that function's `^`. If you need to rewrite a decorated member, run `read(path="file", sel="?")` and check for a sibling `chunk#ID` directly above your target.
**Python notes:** Python docstrings are body lines, not head lines. A `~` body write on a function that has a docstring deletes the docstring unless you include the docstring in `content`. Python enum members and nested functions/closures are often opaque inside their parent chunk and may not appear as addressable child chunks; rewrite the parent container body. Python decorated class/function `^` writes and Python `^` deletes are rejected because indentation-sensitive bodies can become attached to the wrong block while still parsing.
**Note on non-code formats:** for prose and data formats (markdown, YAML, JSON, frontmatter), unsupported `^` and `~` suffixes warn and fall back to whole-chunk editing. Always replace the entire chunk and include any delimiter syntax (fence backticks, `---` frontmatter markers, list markers, table rows, headings) in your `content` — omitting them deletes them. For markdown sections (`sect_*`), prefer unsuffixed whole-chunk replace because `^`/`~` on prose sections can replace the heading and child content too; if you only need the heading, target the heading child chunk shown in `sel="?"`. Fenced code blocks with a declared language are parsed again and can expose inner chunks such as `code_py#ID.fn_gre#ID`; target those inner chunks when available. Markdown root writes preserve fenced code indentation verbatim. Recognized pipe tables expose `row_N` children for row-level edits; table cells and list items are not independently addressable, so rewrite the whole list/table chunk for those structural changes. Appending a table-row-shaped string (`| value |`) to a table chunk inserts it before the trailing blank-line separator so it remains part of the table. Otherwise read with `raw` first and preserve the exact whitespace inside fences. To insert content after a markdown section heading, use `after` on the heading chunk (`sect_*.chunk` or `sect_*.chunk_1`) — not `before`/`prepend` on the section itself, which lands physically before the heading and gets absorbed by the preceding section on reparse.
</regions>
<ops>
Each edit entry has `path` (`file:selector`) plus **exactly one** operation field — `write`, `insert`, or `delete`. Never set more than one on the same entry. `write:null`, `write:""`, and bare `{path}` entries are rejected; they do not delete.
|fields|path (selector part)|effect|
|---|---|---|
|`write: "content"`|`file:chunk#ID`, `file:chunk#ID~`, or `file:chunk#ID^`|write complete new content to the region|
|`delete: true`|`file:chunk#ID`|delete the chunk explicitly|
|`insert: {loc, body}`|`file:chunk` or `file:chunk~`|insert before/after the chunk (`loc`: `"prepend"` or `"append"`)|
</ops>
<examples>
Given this `read` output for `counter.rs`:
```
| counter.rs·62L·rust·#anth
|
@imp#erhe
1 |use std::fmt;
|
@struct_Counte#onat
3^|/// A simple counter that tracks a value and its history.
4^|#[derive(Debug, Clone)]
5^|pub struct Counter {
-@struct_Counte.field_value#enth
6 | /// The current value.
7 | value: i32,
-@struct_Counte.field_max#seti
8 | /// Maximum allowed value.
9 | max: i32,
10 |}
|
@impl_Counte#reha
12^|impl Counter {
-@impl_Counte.fn_new#ndas
13^| /// Creates a new counter starting at zero.
14^| pub fn new(max: i32) -> Self {
15 | Self { value: 0, max }
16 | }
17 |
-@impl_Counte.fn_increm#ouer
18^| /// Increments the counter by one, clamping at max.
19^| pub fn increment(&mut self) {
20 | if self.value < self.max {
21 | self.value += 1;
22 | }
23 | }
24 |
-@impl_Counte.fn_decrem#arve
25^| /// Decrements the counter by one, clamping at zero.
26^| pub fn decrement(&mut self) {
27 | if self.value > 0 {
28 | self.value -= 1;
29 | }
30 | }
31 |
-@impl_Counte.fn_get#arco
32^| /// Returns the current value.
33^| pub fn get(&self) -> i32 {
34 | self.value
35 | }
36 |}
|
@impl_Displa#meha
38^|impl fmt::Display for Counter {
-@impl_Displa.fn_fmt#deri
39^| fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
40 | write!(f, "Counter({}/{})", self.value, self.max)
41 | }
42 |}
```
Lines marked `^` between the line number and `|` are **head** lines (doc comments, attributes, signature). Lines without `^` are **body** lines. `~` replaces body lines only; `^` replaces head lines only.
# Put body (`~` — the common case)
`{ "path": "counter.rs:impl_Counte.fn_increm#ouer~", "write": "self.value = (self.value + 1).min(self.max);\n" }`
Only body changes; doc comment, signature, and closing `}` are preserved.
# Write whole chunk (rewrite signature + doc + body)
`{ "path": "counter.rs:impl_Counte.fn_increm#ouer", "write": "/// Increments by the given step, clamping at max.\npub fn increment(&mut self, step: i32) {\n\tself.value = (self.value + step).min(self.max);\n}\n" }`
Everything is rewritten. Omitting the doc comment or signature deletes them.
# Write head (`^` — attributes, doc comments, signature)
`{ "path": "counter.rs:impl_Counte.fn_get#arco^", "write": "/// Returns the current counter value.\n#[inline]\npub fn get(&self) → i32 {\n" }`
Head changes (all `^` lines + opening brace); body untouched.
# Insert before a chunk (`prepend`)
`{ "path": "counter.rs:impl_Counte.fn_get", "insert": { "loc": "prepend", "body": "/// Resets the counter to zero.\npub fn reset(&mut self) {\n\tself.value = 0;\n}\n\n" } }`
# Insert after a chunk (`append`)
`{ "path": "counter.rs:struct_Counte", "insert": { "loc": "append", "body": "\nimpl Default for Counter {\n\tfn default() → Self {\n\t\tSelf { value: 0, max: 100 }\n\t}\n}\n" } }`
# Insert at start of container body (`~` + `prepend`)
`{ "path": "counter.rs:impl_Counte~", "insert": { "loc": "prepend", "body": "/// Creates a counter starting at the given value.\npub fn with_value(value: i32, max: i32) → Self {\n\tSelf { value: value.min(max), max }\n}\n\n" } }`
Lands at the top of the impl body, before existing methods.
# Insert at end of container body (`~` + `append`)
`{ "path": "counter.rs:impl_Counte~", "insert": { "loc": "append", "body": "\n/// Returns true if the counter is at its maximum.\npub fn is_maxed(&self) → bool {\n\tself.value ≥ self.max\n}\n" } }`
Lands at the end of the impl body, before the closing `}`.
# Delete a chunk
`{ "path": "counter.rs:impl_Counte.fn_decrem#arve", "delete": true }`
Removes the method including its doc comment and signature.
</examples>
@@ -14,14 +14,11 @@ Searches files using powerful regex matching.
- Text output is line-number-prefixed
{{/if}}
{{/if}}
{{#if IS_CHUNK_MODE}}
- Text output is chunk-path-prefixed: `path:sel>123|content`
{{/if}}
</output>
<critical>
- You **MUST** use the built-in Grep tool for any content search. Do **NOT** shell out to `grep`, `rg`, `ripgrep`, `ag`, `ack`, `git grep`, `awk`, `sed`-for-search, or any other CLI search via Bash — even for a single match, even "just to check quickly", even piped through other commands.
- Bash `grep`/`rg` returns raw text without chunk paths, loses `.gitignore` semantics, bypasses result limits, and wastes tokens. The Grep tool is faster, structured, and already wired into the workspace — there is no scenario where Bash search is preferable.
- Bash `grep`/`rg` loses `.gitignore` semantics, bypasses result limits, and wastes tokens. The Grep tool is faster, structured, and already wired into the workspace — there is no scenario where Bash search is preferable.
- If you catch yourself typing `grep`, `rg`, or `| grep` in a Bash command, stop and re-issue the search through the Grep tool instead.
- If the search is open-ended, requiring multiple rounds, you **MUST** use the Task tool with the explore subagent instead of chaining Grep calls yourself.
</critical>
@@ -4,4 +4,4 @@ You **MUST** use the `poll` tool (in a loop, if necessary) instead of manually r
If the timeout elapses before any job changes state, it returns the current snapshot (still-running jobs and any already-completed deliveries) without erroring — call `poll` again to keep waiting.
You **MUST NOT** poll the same job repeatedly without evidence of progress. Between calls, inspect `read jobs://<id>` to confirm new output or activity. If a job is stalled, has hung, or is producing nothing useful, cancel it via `cancel_job` and try a different approach instead of waiting indefinitely.
You **MUST NOT** poll the same job repeatedly without evidence of progress. Between calls, inspect `read jobs://<id>` to confirm new output or activity. If a job is stalled, has hung, or is producing nothing useful, cancel it via `cancel_job` and try a different approach instead of waiting indefinitely.
@@ -1,73 +0,0 @@
Reads files using syntax-aware chunks. Also inspects directories, archives, SQLite databases, images, documents (PDF/DOCX/PPTX/XLSX/RTF/EPUB/ipynb), **and URLs**.
<instruction>
The chunk-aware `read` variant returns AST-scoped chunks with current checksum IDs for structural editing, and otherwise behaves like `open` for non-code content.
- You **MUST** parallelize calls when exploring related files
- For URLs, `read` fetches the page and returns clean extracted text/markdown by default (reader-mode). It handles HTML pages, GitHub issues/PRs, Stack Overflow, Wikipedia, Reddit, NPM, arXiv, RSS/Atom, JSON endpoints, PDFs, etc. You **SHOULD** reach for `read` — not a browser/puppeteer tool — for fetching and inspecting web content.
## Parameters
- `path` — file path or URL; may include `:selector` suffix (required)
- `sel` — optional selector for chunks, line ranges, listing, or raw mode
- `timeout` — seconds, for URLs only
## Selectors
|`sel` value|Behavior|
|---|---|
|*(omitted)*|Read full file as chunks (up to {{DEFAULT_LIMIT}} lines)|
|`class_Foo`|Read a specific chunk|
|`class_Foo.fn_bar#thth~`|Read a chunk region (body `~` / head `^`) by ID|
|`?`|List all chunk paths with IDs|
|`L50`|Read from line 50 onward (shorthand for L50 to EOF)|
|`L50-L120`|Read lines 50 through 120|
|`L20-L20`|Read exactly one line|
|`raw`|Raw content without transformations (for URLs: untouched HTML)|
Max {{DEFAULT_MAX_LINES}} lines per call.
# Chunks
Each anchor `@full.chunk.path#thth` (with `-` prefixes for nesting depth) in the output identifies a chunk. Use `full.chunk.path#thth` as-is to read truncated chunks.
If you need a canonical target list, run `read(path="file", sel="?")`. That listing shows chunk paths with IDs and is the safest structural discovery mode. Summary lines in this listing are orientation hints; follow a selector with `read(path="file", sel="chunk#ID")` or use `raw` when you need exact source.
Line numbers in the gutter are absolute file line numbers.
{{#if chunkAutoIndent}}
Chunk reads normalize leading indentation so copied content round-trips cleanly into chunk edits.
{{else}}
Chunk reads preserve literal leading tabs/spaces from the file. When editing, keep the same whitespace characters you see here.
{{/if}}
`raw` shows the file's literal whitespace. Structured chunk views may normalize or display indentation for edit round-tripping, so use `raw` when exact tabs/spaces matter, especially inside markdown fenced code blocks.
IDs change after every edit. Use the new IDs from the edit response or refresh with `sel="?"` before the next `write`/`delete`. `insert` selectors may omit IDs, but still prefer fresh paths after structural edits.
Parser boundaries vary by language: TypeScript/JavaScript decorators and JSDoc above decorated methods may appear as sibling `chunk#ID` entries, Python decorators are part of the function/class head, Python docstrings are body lines, and Python enum members or nested closures may remain opaque inside their parent chunk. Decorated Python `^` writes and Python `^` deletes are rejected for safety.
Markdown sections, lists, and tables are structural chunks. Recognized pipe tables expose `row_N` children for row-level edits; list items and table cells are not independently addressable. Fenced code blocks with a declared language are parsed again when possible, so functions inside a markdown fence can appear as addressable nested chunks.
Chunk trees: JS, TS, TSX, Python, Rust, Go. Others use blank-line fallback.
# Inspection
Extracts text from PDF, Word, PowerPoint, Excel, RTF, EPUB, and Jupyter notebook files. Can inspect images.
# Directories & Archives
Directories and archive roots return a list of entries. Supports `.tar`, `.tar.gz`, `.tgz`, `.zip`. Use `archive.ext:path/inside/archive` to read contents.
# SQLite Databases
When used against a SQLite database (`.sqlite`, `.sqlite3`, `.db`, `.db3`), returns structured database content.
- `file.db` — list tables with row counts
- `file.db:table` — table schema + sample rows
- `file.db:table:key` — single row by primary key
- `file.db:table?limit=50&offset=100` — paginated rows
- `file.db:table?where=status='active'&order=created:desc` — filtered rows
- `file.db?q=SELECT …` — read-only SELECT query
# URLs
Extracts content from web pages, GitHub issues/PRs, Stack Overflow, Wikipedia, Reddit, NPM, arXiv, RSS/Atom feeds, JSON endpoints, PDFs at URLs, and similar text-based resources. Returns clean reader-mode text/markdown — no browser required. Use `sel="raw"` for untouched HTML; `timeout` to override the default request timeout. You **SHOULD** prefer `read` over a browser/puppeteer tool for fetching URL content; only use a browser when the page requires JS execution, authentication, or interactive actions (clicks, forms, scrolling).
</instruction>
<critical>
- You **MUST** `read` before editing — never invent chunk names or IDs.
- Chunk names are truncated (e.g., `handleRequest` becomes `fn_handleRequ`). Always copy chunk paths from `read` or `?` output — never construct them from source identifiers.
- You **MUST** use `read` (never bash `cat`/`head`/`tail`/`less`/`more`/`ls`/`tar`/`unzip`/`curl`/`wget`) for all file, directory, archive, and URL reads.
- You **MUST NOT** reach for a browser/puppeteer tool to fetch static web content — `read` handles HTML, PDFs, JSON, feeds, and docs directly. Reserve browser tools for JS-heavy pages or interactive flows.
- You **MUST** always include the `path` parameter; never call with `{}`.
- For specific line ranges, use `sel`: `read(path="file", sel="L50-L150")` — not `cat -n file | sed`.
- You **MAY** use `sel` with URL reads; the tool paginates cached fetched output.
</critical>
@@ -18,7 +18,7 @@ The `read` tool is multi-purpose and more capable than it looks — inspects fil
|`L50`|Read from line 50 onward (shorthand for L50 to EOF)|
|`L50-L120`|Read lines 50 through 120|
|`L20-L20`|Read exactly one line|
|`raw`|Skip line-numbering / hashline / chunking; return file content as plain text. For URLs: untouched HTML.|
|`raw`|Skip line-numbering / hashline; return file content as plain text. For URLs: untouched HTML.|
Max {{DEFAULT_MAX_LINES}} lines per call.
@@ -45,7 +45,7 @@ If `done`, `rm`, or `drop` omits both `task` and `phase`, it applies to all task
## Phase Anatomy
- `name`: Short, human-readable noun phrase (1-3 words). Capitalize naturally.
- Always prefix with a roman-numeral ordinal (`I.`, `II.`, `III.`, `IV.`, ...) to convey ordering — e.g. `I. Foundation`, `II. Auth`, `III. Routing`. Single-phase plans use `I.` too.
- Always prefix with a roman-numeral ordinal (`I.`, `II.`, `III.`, `IV.`, …) to convey ordering — e.g. `I. Foundation`, `II. Auth`, `III. Routing`. Single-phase plans use `I.` too.
- You **MUST NOT** use snake_case, `Phase1_*`, arabic numerals (`1.`), or letter prefixes (`A.`) — they render as ugly identifiers.
## Rules
@@ -66,7 +66,7 @@ Create a todo list when:
<examples>
# Initial setup (multi-phase)
`{"ops":[{"op":"replace","phases":[{"name":"I. Foundation","tasks":[{"content":"Scaffold crate"},{"content":"Wire workspace"}]},{"name":"II. Auth","tasks":[{"content":"Port credential store"},{"content":"Wire OAuth providers"}]},{"name":"III. Verification","tasks":[{"content":"Run cargo test"}]}]}]}`
# Initial setup (single phase " still prefixed)
# Initial setup (single phase — still prefixed)
`{"ops":[{"op":"replace","phases":[{"name":"I. Implementation","tasks":[{"content":"Apply fix"},{"content":"Run tests"}]}]}]}`
# Complete one task
`{"ops":[{"op":"done","task":"task-2"}]}`
@@ -1,12 +1,10 @@
import { invalidateFsScanCache } from "@oh-my-pi/pi-natives";
import { invalidateChunkCache } from "../edit/modes/chunk";
/**
* Invalidate shared filesystem scan caches after a content write/update.
*/
export function invalidateFsScanAfterWrite(path: string): void {
invalidateFsScanCache(path);
invalidateChunkCache(path);
}
/**
@@ -14,7 +12,6 @@ export function invalidateFsScanAfterWrite(path: string): void {
*/
export function invalidateFsScanAfterDelete(path: string): void {
invalidateFsScanCache(path);
invalidateChunkCache(path);
}
/**
@@ -25,9 +22,7 @@ export function invalidateFsScanAfterDelete(path: string): void {
*/
export function invalidateFsScanAfterRename(oldPath: string, newPath: string): void {
invalidateFsScanCache(oldPath);
invalidateChunkCache(oldPath);
if (newPath !== oldPath) {
invalidateFsScanCache(newPath);
invalidateChunkCache(newPath);
}
}
+1 -122
View File
@@ -6,9 +6,8 @@ import type { Component } from "@oh-my-pi/pi-tui";
import { Text } from "@oh-my-pi/pi-tui";
import { prompt, untilAborted } from "@oh-my-pi/pi-utils";
import { type Static, Type } from "@sinclair/typebox";
import { type ChunkedGrepMatch, describeChunkedGrepMatch } from "../edit/modes/chunk";
import type { RenderResultOptions } from "../extensibility/custom-tools/types";
import { getLanguageFromPath, type Theme } from "../modes/theme/theme";
import { type Theme } from "../modes/theme/theme";
import grepDescription from "../prompts/tools/grep.md" with { type: "text" };
import { DEFAULT_MAX_COLUMN, type TruncationResult, truncateHead } from "../session/streaming-output";
import { Ellipsis, Hasher, type RenderCache, renderStatusLine, renderTreeList, truncateToWidth } from "../tui";
@@ -83,7 +82,6 @@ export class GrepTool implements AgentTool<typeof grepSchema, GrepToolDetails> {
this.description = prompt.render(grepDescription, {
IS_HASHLINE_MODE: displayMode.hashLines,
IS_LINE_NUMBER_MODE: !displayMode.hashLines && displayMode.lineNumbers,
IS_CHUNK_MODE: displayMode.chunked,
});
}
@@ -98,7 +96,6 @@ export class GrepTool implements AgentTool<typeof grepSchema, GrepToolDetails> {
return untilAborted(signal, async () => {
const normalizedPattern = pattern.trim();
const chunkMode = resolveEditMode(this.session) === "chunk";
if (!normalizedPattern) {
throw new ToolError("Pattern must not be empty");
}
@@ -297,124 +294,6 @@ export class GrepTool implements AgentTool<typeof grepSchema, GrepToolDetails> {
}
matchesByFile.get(relativePath)!.push(match);
}
if (chunkMode) {
const annotatedMatches = await Promise.all(
selectedMatches.map(match => {
const relativePath = match.path.startsWith("/") ? match.path.slice(1) : match.path;
const absoluteFilePath = isDirectory ? path.join(searchPath, relativePath) : searchPath;
return describeChunkedGrepMatch({
filePath: absoluteFilePath,
lineNumber: match.lineNumber,
line: match.line,
cwd: this.session.cwd,
language: getLanguageFromPath(absoluteFilePath),
});
}),
);
const chunkMatchesByFile = new Map<string, ChunkedGrepMatch[]>();
for (const match of annotatedMatches) {
recordFile(match.displayPath);
if (!chunkMatchesByFile.has(match.displayPath)) {
chunkMatchesByFile.set(match.displayPath, []);
}
chunkMatchesByFile.get(match.displayPath)!.push(match);
}
const renderChunkedMatchesForFile = (relativePath: string): string[] => {
const renderedLines: string[] = [];
const fileMatches = chunkMatchesByFile.get(relativePath) ?? [];
if (fileMatches.length === 0) {
return renderedLines;
}
const matchesByChunk = new Map<string, ChunkedGrepMatch[]>();
for (const match of fileMatches) {
const chunkKey = match.chunkPath ?? "";
if (!matchesByChunk.has(chunkKey)) {
matchesByChunk.set(chunkKey, []);
}
matchesByChunk.get(chunkKey)!.push(match);
}
for (const [chunkPath, chunkMatches] of matchesByChunk) {
if (chunkPath) {
const chunkChecksum = chunkMatches[0]?.chunkChecksum;
const dashes = "-".repeat(chunkPath.split(".").length - 1);
const anchor = chunkChecksum
? `${dashes}@${chunkPath}#${chunkChecksum}`
: `${dashes}@${chunkPath}`;
renderedLines.push(anchor);
}
for (const match of chunkMatches) {
renderedLines.push(` ${match.lineNumber}|${match.line}`);
fileMatchCounts.set(relativePath, (fileMatchCounts.get(relativePath) ?? 0) + 1);
}
}
return renderedLines;
};
if (isDirectory) {
const filesByDirectory = new Map<string, string[]>();
for (const relativePath of fileList) {
const directory = path.dirname(relativePath).replace(/\\/g, "/");
if (!filesByDirectory.has(directory)) {
filesByDirectory.set(directory, []);
}
filesByDirectory.get(directory)!.push(relativePath);
}
for (const [directory, directoryFiles] of filesByDirectory) {
if (directory === ".") {
for (const relativePath of directoryFiles) {
const renderedLines = renderChunkedMatchesForFile(relativePath);
if (renderedLines.length === 0) continue;
if (outputLines.length > 0) {
outputLines.push("");
}
outputLines.push(`# ${path.basename(relativePath)}`);
outputLines.push(...renderedLines);
}
continue;
}
const renderedFiles = directoryFiles
.map(relativePath => ({ relativePath, lines: renderChunkedMatchesForFile(relativePath) }))
.filter(file => file.lines.length > 0);
if (renderedFiles.length === 0) continue;
if (outputLines.length > 0) {
outputLines.push("");
}
outputLines.push(`# ${directory}`);
for (const { relativePath, lines } of renderedFiles) {
outputLines.push(`## └─ ${path.basename(relativePath)}`);
outputLines.push(...lines);
}
}
} else {
for (const relativePath of fileList) {
outputLines.push(...renderChunkedMatchesForFile(relativePath));
}
}
if (matchLimitReached || result.limitReached) {
outputLines.push("", limitMessage);
}
const rawOutput = outputLines.join("\n");
const truncation = truncateHead(rawOutput, { maxLines: Number.MAX_SAFE_INTEGER });
const truncated = Boolean(matchLimitReached || result.limitReached || truncation.truncated);
const details: GrepToolDetails = {
scopePath,
matchCount: selectedMatches.length,
fileCount: fileList.length,
files: fileList,
fileMatches: fileList.map(path => ({
path,
count: fileMatchCounts.get(path) ?? 0,
})),
truncated,
matchLimitReached: matchLimitReached ? effectiveLimit : undefined,
resultLimitReached: result.limitReached ? internalLimit : undefined,
};
if (truncation.truncated) details.truncation = truncation;
const resultBuilder = toolResult(details).text(truncation.content);
if (truncation.truncated) {
resultBuilder.truncation(truncation, { direction: "head" });
}
return resultBuilder.done();
}
const displayLines: string[] = [];
const renderMatchesForFile = (relativePath: string): { model: string[]; display: string[] } => {
const modelOut: string[] = [];
+1 -1
View File
@@ -25,7 +25,7 @@ const WAIT_DURATION_MS: Record<string, number> = {
};
function parseWaitDurationMs(value: string | undefined): number {
return (value && WAIT_DURATION_MS[value]) ?? WAIT_DURATION_MS["30s"];
return (value ? WAIT_DURATION_MS[value] : undefined) ?? WAIT_DURATION_MS["30s"];
}
interface PollResult {
+13 -112
View File
@@ -9,20 +9,11 @@ import { Text } from "@oh-my-pi/pi-tui";
import { getRemoteDir, prompt, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils";
import { type Static, Type } from "@sinclair/typebox";
import { formatHashLines } from "../edit/line-hash";
import {
type ChunkReadTarget,
formatChunkedRead,
parseChunkReadPath,
parseChunkSelector,
resolveAnchorStyle,
resolveChunkAutoIndent,
} from "../edit/modes/chunk";
import type { RenderResultOptions } from "../extensibility/custom-tools/types";
import { parseInternalUrl } from "../internal-urls/parse";
import type { InternalUrl } from "../internal-urls/types";
import { getLanguageFromPath, type Theme } from "../modes/theme/theme";
import readDescription from "../prompts/tools/read.md" with { type: "text" };
import readChunkDescription from "../prompts/tools/read-chunk.md" with { type: "text" };
import type { ToolSession } from "../sdk";
import {
DEFAULT_MAX_BYTES,
@@ -71,12 +62,6 @@ import {
import { ToolAbortError, ToolError, throwIfAborted } from "./tool-errors";
import { toolResult } from "./tool-result";
const PROSE_LANGUAGES = new Set(["markdown", "text", "log", "asciidoc", "restructuredtext"]);
function isProseLanguage(language: string | undefined): boolean {
return language !== undefined && PROSE_LANGUAGES.has(language);
}
// Document types converted to markdown via markit.
const CONVERTIBLE_EXTENSIONS = new Set([".pdf", ".doc", ".docx", ".ppt", ".pptx", ".xls", ".xlsx", ".rtf", ".epub"]);
@@ -370,7 +355,6 @@ export interface ReadToolDetails {
isDirectory?: boolean;
resolvedPath?: string;
suffixResolution?: { from: string; to: string };
chunk?: ChunkReadTarget;
url?: string;
finalUrl?: string;
contentType?: string;
@@ -389,16 +373,14 @@ type ReadParams = ReadToolInput;
type ParsedSelector =
| { kind: "none" }
| { kind: "raw" }
| { kind: "lines"; startLine: number; endLine: number | undefined }
| { kind: "chunk"; selector: string };
| { kind: "lines"; startLine: number; endLine: number | undefined };
const LINE_RANGE_RE = /^L(\d+)(?:-L?(\d+))?$/i;
function parseSel(sel: string | undefined): ParsedSelector {
if (!sel || sel.length === 0) return { kind: "none" };
const normalizedSelector = parseChunkSelector(sel).selector ?? sel;
if (normalizedSelector === "raw") return { kind: "raw" };
const lineMatch = LINE_RANGE_RE.exec(normalizedSelector);
if (sel === "raw") return { kind: "raw" };
const lineMatch = LINE_RANGE_RE.exec(sel);
if (lineMatch) {
const rawStart = Number.parseInt(lineMatch[1]!, 10);
if (rawStart < 1) {
@@ -410,7 +392,7 @@ function parseSel(sel: string | undefined): ParsedSelector {
}
return { kind: "lines", startLine: rawStart, endLine: rawEnd };
}
return { kind: "chunk", selector: normalizedSelector };
throw new ToolError(`Invalid sel '${sel}'. Use 'raw' or a line range like 'L50' or 'L50-L120'.`);
}
/** Convert a line-range selector to the offset/limit pair used by internal pagination. */
@@ -477,18 +459,12 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
Math.min(session.settings.get("read.defaultLimit") ?? DEFAULT_MAX_LINES, DEFAULT_MAX_LINES),
);
this.#inspectImageEnabled = session.settings.get("inspect_image.enabled");
this.description =
resolveEditMode(session) === "chunk"
? prompt.render(readChunkDescription, {
anchorStyle: resolveAnchorStyle(session.settings),
chunkAutoIndent: resolveChunkAutoIndent(),
})
: prompt.render(readDescription, {
DEFAULT_LIMIT: String(this.#defaultLimit),
DEFAULT_MAX_LINES: String(DEFAULT_MAX_LINES),
IS_HASHLINE_MODE: displayMode.hashLines,
IS_LINE_NUMBER_MODE: !displayMode.hashLines && displayMode.lineNumbers,
});
this.description = prompt.render(readDescription, {
DEFAULT_LIMIT: String(this.#defaultLimit),
DEFAULT_MAX_LINES: String(DEFAULT_MAX_LINES),
IS_HASHLINE_MODE: displayMode.hashLines,
IS_LINE_NUMBER_MODE: !displayMode.hashLines && displayMode.lineNumbers,
});
}
async #resolveArchiveReadPath(readPath: string, signal?: AbortSignal): Promise<ResolvedArchiveReadPath | null> {
@@ -930,7 +906,6 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
readPath = expandPath(readPath);
}
const displayMode = resolveFileDisplayMode(this.session);
const chunkMode = resolveEditMode(this.session) === "chunk";
// Handle internal URLs (agent://, artifact://, memory://, skill://, rule://, local://, mcp://)
const internalRouter = this.session.internalRouter;
@@ -964,13 +939,8 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
return executeReadUrl(this.session, { path: parsedUrlTarget.path, timeout, raw: parsedUrlTarget.raw }, signal);
}
const parsedReadPath = chunkMode ? parseChunkReadPath(readPath) : { filePath: readPath };
const localReadPath = parsedReadPath.filePath;
const pathSelectorParsed = chunkMode ? parseSel(parsedReadPath.selector) : { kind: "none" as const };
const pathChunkSelector = pathSelectorParsed.kind === "chunk" ? pathSelectorParsed.selector : undefined;
const selectorInput = sel ?? parsedReadPath.selector;
const rawSelectorInput = sel ?? parsedReadPath.selector;
const parsed = parseSel(selectorInput);
const localReadPath = readPath;
const parsed = parseSel(sel);
const archivePath = await this.#resolveArchiveReadPath(localReadPath, signal);
if (archivePath) {
@@ -1032,52 +1002,8 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
const ext = path.extname(absolutePath).toLowerCase();
const hasEditTool = this.session.hasEditTool ?? true;
const language = getLanguageFromPath(absolutePath);
const skipChunksForExplore = !hasEditTool && !this.session.settings.get("read.explorechunks");
const skipChunksForProse = isProseLanguage(language) && !this.session.settings.get("read.prosechunks");
const shouldConvertWithMarkit =
CONVERTIBLE_EXTENSIONS.has(ext) || (ext === ".ipynb" && (parsed.kind === "raw" || !chunkMode));
if (chunkMode && parsed.kind !== "raw" && !skipChunksForExplore && !skipChunksForProse) {
const absoluteLineRange =
pathChunkSelector && parsed.kind === "lines"
? { startLine: parsed.startLine, endLine: parsed.endLine }
: undefined;
// sel= wins over path:chunk when both are provided (explicit param > embedded path).
const effectiveSelector = sel ? selectorInput : (pathChunkSelector ?? selectorInput);
const rawEffectiveSelector = sel ? selectorInput : (rawSelectorInput ?? effectiveSelector);
const chunkReadPath =
parsed.kind === "chunk" || (pathChunkSelector && !sel)
? rawEffectiveSelector
? `${localReadPath}:${rawEffectiveSelector}`
: localReadPath
: parsed.kind === "lines"
? parsed.endLine !== undefined
? `${localReadPath}:L${parsed.startLine}-L${parsed.endLine}`
: `${localReadPath}:L${parsed.startLine}`
: localReadPath;
const chunkResult = await formatChunkedRead({
filePath: absolutePath,
readPath: chunkReadPath,
cwd: this.session.cwd,
language,
omitChecksum: !hasEditTool,
anchorStyle: resolveAnchorStyle(this.session.settings),
absoluteLineRange,
});
let text = chunkResult.text;
if (suffixResolution) {
text = prependSuffixResolutionNotice(text, suffixResolution);
}
return toolResult<ReadToolDetails>({
resolvedPath: absolutePath,
suffixResolution,
chunk: chunkResult.chunk,
})
.text(text)
.sourcePath(absolutePath)
.done();
}
CONVERTIBLE_EXTENSIONS.has(ext) || (ext === ".ipynb" && parsed.kind === "raw");
// Read the file based on type
let content: Array<TextContent | ImageContent>;
let details: ReadToolDetails = {};
@@ -1160,31 +1086,6 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
content = [{ type: "text", text: `[Cannot read ${ext} file: conversion failed]` }];
}
} else {
// Chunk mode: dispatch to chunk tree unless raw or line range requested
if (chunkMode && parsed.kind !== "raw" && parsed.kind !== "lines") {
const chunkSel = parsed.kind === "chunk" ? parsed.selector : undefined;
const chunkResult = await formatChunkedRead({
filePath: absolutePath,
readPath: chunkSel ? `${localReadPath}:${chunkSel}` : localReadPath,
cwd: this.session.cwd,
language: getLanguageFromPath(absolutePath),
omitChecksum: !(this.session.hasEditTool ?? true),
anchorStyle: resolveAnchorStyle(this.session.settings),
});
let text = chunkResult.text;
if (suffixResolution) {
text = prependSuffixResolutionNotice(text, suffixResolution);
}
return toolResult<ReadToolDetails>({
resolvedPath: absolutePath,
suffixResolution,
chunk: chunkResult.chunk,
})
.text(text)
.sourcePath(absolutePath)
.done();
}
// Raw text or line-range mode
const { offset, limit } = selToOffsetLimit(parsed);
const startLine = offset ? Math.max(0, offset - 1) : 0;
+1 -2
View File
@@ -1,13 +1,12 @@
import { $env, $flag } from "@oh-my-pi/pi-utils";
export type EditMode = "replace" | "patch" | "hashline" | "chunk" | "vim" | "apply_patch" | "atom";
export type EditMode = "replace" | "patch" | "hashline" | "vim" | "apply_patch" | "atom";
export const DEFAULT_EDIT_MODE: EditMode = "hashline";
const EDIT_MODE_IDS = {
apply_patch: "apply_patch",
atom: "atom",
chunk: "chunk",
hashline: "hashline",
patch: "patch",
replace: "replace",
@@ -7,7 +7,6 @@ import { resolveEditMode } from "./edit-mode";
export interface FileDisplayMode {
lineNumbers: boolean;
hashLines: boolean;
chunked: boolean;
}
/** Session-like object providing settings and tool availability for display mode resolution. */
@@ -33,10 +32,8 @@ export function resolveFileDisplayMode(session: FileDisplayModeSession, options?
const usesHashLineAnchors = editMode === "hashline" || editMode === "atom";
const raw = options?.raw === true;
const hashLines = !raw && hasEditTool && usesHashLineAnchors && settings.get("readHashLines") !== false;
const chunked = !raw && hasEditTool && editMode === "chunk";
return {
hashLines,
lineNumbers: !raw && (hashLines || settings.get("readLineNumbers") === true),
chunked,
};
}
@@ -1,33 +0,0 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import * as os from "node:os";
import * as path from "node:path";
import { runReadCommand } from "@oh-my-pi/pi-coding-agent/cli/read-cli";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types";
describe("runReadCommand URL handling", () => {
afterEach(() => {
vi.restoreAllMocks();
});
it("delegates URL inputs through the read tool pipeline", async () => {
const cwd = path.join(os.tmpdir(), "read-cli-url-test");
const settings = Settings.isolated({ "fetch.enabled": true });
const pageUrl = "https://example.com/cli-read";
const consoleLogSpy = vi.spyOn(console, "log").mockImplementation(() => {});
vi.spyOn(Settings, "init").mockResolvedValue(settings);
vi.spyOn(scrapers, "loadPage").mockResolvedValue({
ok: true,
status: 200,
contentType: "text/plain",
finalUrl: pageUrl,
content: "CLI URL content",
});
const cwdSpy = vi.spyOn(process, "cwd").mockReturnValue(cwd);
await runReadCommand({ path: pageUrl });
expect(cwdSpy).toHaveBeenCalled();
expect(consoleLogSpy).toHaveBeenCalledWith(expect.stringContaining("CLI URL content"));
});
});
@@ -4,14 +4,10 @@ import * as os from "node:os";
import * as path from "node:path";
import {
adjustIndentation,
computeChunkDiff,
computeEditDiff,
computeHashlineDiff,
DEFAULT_FUZZY_THRESHOLD,
findMatch,
loadChunkSource,
parseChunkEditPath,
parseChunkReadPath,
} from "@oh-my-pi/pi-coding-agent/edit";
describe("findMatch", () => {
@@ -162,127 +158,6 @@ describe("findMatch", () => {
});
});
describe("computeChunkDiff", () => {
let tmpDir: string;
beforeEach(async () => {
tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "compute-chunk-"));
});
afterEach(async () => {
await fs.rm(tmpDir, { recursive: true, force: true });
});
test("returns { error } when chunk selector cannot resolve", async () => {
const file = path.join(tmpDir, "c.ts");
await fs.writeFile(file, "export const x = 1;\n");
const result = await computeChunkDiff(
{
path: "c.ts:fn_does_not_exist#ABCD",
edits: [
{
path: "c.ts:fn_does_not_exist#ABCD",
write: "console.log('replaced')\n",
},
],
},
tmpDir,
);
expect("error" in result).toBe(true);
});
test("returns { error } when path is empty", async () => {
const result = await computeChunkDiff({ path: "", edits: [{ path: "", write: "x\n" }] }, tmpDir);
expect("error" in result).toBe(true);
});
test("rejects write:null instead of previewing a delete", async () => {
const file = path.join(tmpDir, "null-delete.ts");
await fs.writeFile(file, "export const x = 1;\n");
const result = await computeChunkDiff(
{
path: "null-delete.ts",
edits: [{ path: "null-delete.ts", write: null }],
},
tmpDir,
);
expect("error" in result).toBe(true);
if ("error" in result) {
expect(result.error).toContain("write:null no longer deletes chunks");
}
});
test("rejects bare chunk edit entries instead of treating them as deletes", async () => {
const file = path.join(tmpDir, "bare.ts");
await fs.writeFile(file, "export const x = 1;\n");
const result = await computeChunkDiff(
{
path: "bare.ts",
edits: [{ path: "bare.ts" }],
},
tmpDir,
);
expect("error" in result).toBe(true);
if ("error" in result) {
expect(result.error).toContain("no operation specified");
}
});
test("rejects write empty string instead of previewing a destructive empty replacement", async () => {
const file = path.join(tmpDir, "empty-write.ts");
await fs.writeFile(file, "export const x = 1;\n");
const result = await computeChunkDiff(
{
path: "empty-write.ts",
edits: [{ path: "empty-write.ts", write: "" }],
},
tmpDir,
);
expect("error" in result).toBe(true);
if ("error" in result) {
expect(result.error).toContain('write:"" is a destructive empty replacement');
}
});
test("aborts when signal fires before compute completes", async () => {
const controller = new AbortController();
controller.abort();
const result = await computeChunkDiff(
{
path: "d.ts",
edits: [{ path: "d.ts", write: "foo\n" }],
},
tmpDir,
{ signal: controller.signal },
);
expect("error" in result).toBe(true);
});
test("computes diff for a root chunk replacement with valid checksum", async () => {
const file = path.join(tmpDir, "e.ts");
await fs.writeFile(file, "export const x = 1;\n");
// Read the file once via loadChunkSource so the test does not depend on
// knowing the internal chunk checksum scheme.
const loaded = await loadChunkSource({ cwd: tmpDir, path: "e.ts" });
expect(loaded.exists).toBe(true);
expect(loaded.rawContent).toContain("export const x");
});
});
describe("chunk path parsing", () => {
test("splits local plan URLs with chunk selectors after the URL path", () => {
expect(parseChunkEditPath("local://PLAN.md:sct_0_T#SRJJ")).toEqual({
filePath: "local://PLAN.md",
selector: "sct_0_T#SRJJ",
});
expect(parseChunkReadPath("local://PLAN.md:sct_6_R.sct_6_u#MZKS")).toEqual({
filePath: "local://PLAN.md",
selector: "sct_6_R.sct_6_u#MZKS",
});
});
test("does not treat the local URL scheme colon as a chunk selector separator", () => {
expect(parseChunkEditPath("local://PLAN.md")).toEqual({ filePath: "local://PLAN.md" });
});
});
describe("adjustIndentation", () => {
test("adds indentation when actualText is more indented than oldText", () => {
@@ -30,45 +30,6 @@ describe("dropIncompleteLastEdit", () => {
});
});
describe("chunk extractCompleteEdits", () => {
const strategy = EDIT_MODE_STRATEGIES.chunk;
test("passes through a single complete entry", () => {
const args = {
edits: [{ path: "a.ts", write: "foo" }],
__partialJson: '{"edits":[{"path":"a.ts","write":"foo"}]}',
};
const out = strategy.extractCompleteEdits(args, args.__partialJson) as typeof args;
expect(out.edits).toHaveLength(1);
});
test("drops trailing entry when partial JSON has open-brace after last close", () => {
const args = {
edits: [{ path: "a.ts", write: "foo" }, { path: "b.ts" }],
__partialJson: '{"edits":[{"path":"a.ts","write":"foo"},{"path":"b.ts"',
};
const out = strategy.extractCompleteEdits(args, args.__partialJson) as typeof args;
expect(out.edits).toHaveLength(1);
expect(out.edits[0].path).toBe("a.ts");
});
test("drops trailing entry when partial JSON ends in ':nu' (write: null guard)", () => {
const args = {
edits: [
{ path: "a.ts", write: "foo" },
{ path: "b.ts", write: null },
],
// simulates partial-json coercing the in-flight `nu` to `null`
__partialJson: '{"edits":[{"path":"a.ts","write":"foo"},{"path":"b.ts","write":nu',
};
const out = strategy.extractCompleteEdits(args, args.__partialJson) as typeof args;
// Last entry should be dropped because its `}` hasn't arrived yet, so
// incomplete null-write errors are suppressed while streaming.
expect(out.edits).toHaveLength(1);
expect(out.edits[0].path).toBe("a.ts");
});
});
describe("apply_patch extractCompleteEdits", () => {
const strategy = EDIT_MODE_STRATEGIES.apply_patch;
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Removed
- Removed the `chunk` napi module (`ChunkState`, chunk schema, chunk rendering, chunk edit) and dropped `generate_chunk_schema()` from the build script
## [14.3.0] - 2026-04-25
### Added
-311
View File
@@ -1,69 +1,5 @@
/* auto-generated by NAPI-RS */
/* eslint-disable */
/**
* Parsed file as a chunk tree: query nodes, render views, format grep hits,
* and apply edits.
*/
export declare class ChunkState {
/**
* Build chunk state by parsing `source` with the given `language` id (e.g.
* `typescript`).
*/
static parse(source: string, language: string): ChunkState
/** Normalized language identifier used for the tree-sitter parse. */
get language(): string
/** Full source text for this file. */
get source(): string
/** Stable checksum for the entire file contents. */
get checksum(): string
/** Line count of the source buffer. */
get lineCount(): number
/** Count of tree-sitter error nodes seen while building the tree. */
get parseErrors(): number
/** True when a fallback classifier produced the tree. */
get fallback(): boolean
/** Selector path string for the synthetic root (often empty). */
get rootPath(): string
/** Top-level child chunk paths under the root. */
get rootChildren(): Array<string>
/** Total number of chunk nodes. */
get chunkCount(): number
/** True when the parsed file contains unresolved merge conflicts. */
hasConflicts(): boolean
/** Count of unresolved merge conflicts represented in the chunk tree. */
conflictCount(): number
/** Summary for the root chunk, if it exists. */
root(): ChunkInfo | null
/** Look up [`ChunkInfo`] for a chunk selector path. */
chunk(chunkPath: string): ChunkInfo | null
/** Every chunk node as a [`ChunkInfo`] list. */
chunks(): Array<ChunkInfo>
/**
* Direct children of `chunkPath` (use empty or omit for root); errors if
* the path is missing.
*/
children(chunkPath?: string | undefined | null): Array<ChunkInfo>
/** Chunk selector path that contains 1-based source line `line`, if any. */
lineToContainingChunkPath(line: number): string | null
/** Render a chunk subtree or listing as UTF-8 text for tools. */
render(params: RenderParams): string
/**
* Parse `readPath` (selector, line scope, etc.) and return rendered text or
* errors.
*/
renderRead(params: ReadRenderParams): ReadResult
/**
* Prefix a grep line with `display_path` and the chunk path for
* `line_number`, when known.
*/
formatGrepLine(displayPath: string, lineNumber: number, line: string): string
/**
* Apply batch edits, re-parse, write files, and return updated state and
* messaging.
*/
applyEdits(params: EditParams): EditResult
}
/**
* Long-lived macOS appearance observer.
*
@@ -347,95 +283,6 @@ export interface AstReplaceResult {
parseErrors?: Array<string>
}
/**
* How chunk anchors are formatted in rendered output (name and checksum
* visibility).
*/
export declare enum ChunkAnchorStyle {
/** `[.name#crc]` style anchor. */
Full = 'full',
/** `[.kind#crc]` style anchor (kind is the name prefix before `_`). */
Kind = 'kind',
/** `[#crc]` style anchor. */
Bare = 'bare',
/** `[.name]` without checksum. */
FullOmit = 'full-omit',
/** `[.kind]` without checksum. */
KindOmit = 'kind-omit',
/** Minimal anchor without name or checksum. */
None = 'none'
}
/** Structural edit to apply relative to a chunk anchor. */
export declare enum ChunkEditOp {
/** Put new content into the targeted region. */
Put = 'put',
/** Find and replace a literal substring within the targeted region. */
Replace = 'replace',
/** Remove the targeted region. */
Delete = 'delete',
/** Insert `content` before the targeted region span. */
Before = 'before',
/** Insert `content` after the targeted region span. */
After = 'after',
/** Insert `content` at the start inside the targeted region. */
Prepend = 'prepend',
/** Insert `content` at the end inside the targeted region. */
Append = 'append'
}
/** How a chunk participates in a focus-scoped render pass. */
export declare enum ChunkFocusMode {
/** Emit full content and recurse normally. */
Expanded = 'expanded',
/** Emit just the opening anchor; do not recurse or emit body. */
Collapsed = 'collapsed',
/**
* Emit opening + closing anchors; recurse into focused children only.
* Interior gap lines between children are suppressed.
*/
Container = 'container'
}
/** Summary of a single chunk node for tool output and navigation. */
export interface ChunkInfo {
/** Chunk selector path within the tree. */
path: string
/** Bare chunk identifier (without kind prefix), if available. */
identifier?: string
/** Stable checksum anchor for this chunk. */
checksum: string
/** 1-based start line in the source file (inclusive). */
startLine: number
/** 1-based end line in the source file (inclusive). */
endLine: number
/** Whether this node is a leaf (no child chunks). */
leaf: boolean
}
/** Result of resolving a chunk read request against the tree. */
export declare enum ChunkReadStatus {
/** Selector matched a chunk and content was produced. */
Ok = 'ok',
/** No chunk matched the requested selector. */
NotFound = 'not_found',
/** Chunk matched but does not support the requested region. */
UnsupportedRegion = 'unsupported_region'
}
/** Outcome of resolving which chunk was read for a `renderRead`-style request. */
export interface ChunkReadTarget {
/** Whether the selector matched. */
status: ChunkReadStatus
/** Sanitized selector string that was applied. */
selector: string
}
export declare enum ChunkRegion {
Head = '^',
Body = '~'
}
/** Clipboard image payload encoded as PNG bytes. */
export interface ClipboardImage {
/** PNG-encoded image bytes. */
@@ -469,78 +316,6 @@ export declare function copyToClipboard(text: string): void
*/
export declare function detectMacOSAppearance(): MacOSAppearance | null
/**
* One edit in a batch; targets a chunk via `sel`/`crc` (with params-level
* defaults).
*/
export interface EditOperation {
/** Edit kind (replace, delete, insert relative to anchor). */
op: ChunkEditOp
/**
* Chunk selector path; falls back to `EditParams.defaultSelector` when
* omitted.
*/
sel?: string
/**
* Optional checksum anchor; falls back to `EditParams.defaultCrc` when
* omitted.
*/
crc?: string
/** Region to target. When omitted, targets the full chunk. */
region?: ChunkRegion
/** Replacement or inserted text (meaning depends on `op`). */
content?: string
/**
* For `replace` op: literal substring to find inside the target chunk.
* Must match exactly once.
*/
find?: string
}
/** Arguments for applying a batch of chunk edits to a file. */
export interface EditParams {
/** Edits to apply in order. */
operations: Array<EditOperation>
/**
* When true, normalize indentation for response rendering and inserted
* content. When false, preserve literal tabs/spaces.
*/
normalizeIndent?: boolean
/** Default chunk selector when an `EditOperation` omits `sel`. */
defaultSelector?: string
/** Default checksum when an `EditOperation` omits `crc`. */
defaultCrc?: string
/** Anchor formatting for rendered response text. */
anchorStyle?: ChunkAnchorStyle
/** Working directory used to resolve `filePath` and display paths. */
cwd: string
/** Path to the source file to edit (often relative to `cwd`). */
filePath: string
}
/**
* Result of applying edits: new parse state plus before/after source and
* messaging.
*/
export interface EditResult {
/** Chunk tree state after applying edits and re-parsing. */
state: ChunkState
/** Full file text before edits. */
diffBefore: string
/** Full file text after edits. */
diffAfter: string
/** Rendered summary for tooling (hunks, anchors), driven by `anchorStyle`. */
responseText: string
/** Whether the on-disk source changed. */
changed: boolean
/** Whether the updated source re-parsed without fatal issues. */
parseValid: boolean
/** Absolute or normalized paths that were written or touched. */
touchedPaths: Array<string>
/** Non-fatal issues (e.g. selector warnings) collected during apply. */
warnings: Array<string>
}
/** Ellipsis strategy for [`truncate_to_width`]. */
export declare enum Ellipsis {
/** Use a single Unicode ellipsis character ("…"). */
@@ -601,18 +376,6 @@ export declare enum FileType {
Symlink = 3
}
/** Path + focus mode pair for the N-API boundary (`HashMap` doesn't cross FFI). */
export interface FocusedPath {
path: string
mode: ChunkFocusMode
}
/**
* Format one chunk anchor string for a node at `depth` using `style` and
* optional checksum omission.
*/
export declare function formatAnchor(name: string, checksum: string, style: ChunkAnchorStyle, omitChecksum?: boolean | undefined | null): string
/** Fuzzy file path search for autocomplete. */
export declare function fuzzyFind(options: FuzzyFindOptions): Promise<FuzzyFindResult>
@@ -1157,69 +920,6 @@ export interface PtyStartOptions {
*/
export declare function readImageFromClipboard(): Promise<ClipboardImage | undefined | null>
/**
* Options for `ChunkState.renderRead`: selector path, display path, and
* optional line scoping.
*/
export interface ReadRenderParams {
/** Read selector (`sel=...` path, line range, or empty for whole tree). */
readPath: string
/** Path shown in titles and error messages (often the file path). */
displayPath: string
/** Optional language label for the rendered block. */
languageTag?: string
/** Hide checksums in rendered anchors. */
omitChecksum: boolean
/** Anchor formatting style. */
anchorStyle?: ChunkAnchorStyle
/** Optional absolute file line range to intersect with the resolved chunk. */
absoluteLineRange?: VisibleLineRange
/** Replace tabs in embedded previews. */
tabReplacement?: string
/** When true, normalize displayed indentation to canonical tabs. */
normalizeIndent?: boolean
}
/** Rendered chunk text plus optional resolution metadata for the read request. */
export interface ReadResult {
/** Rendered UTF-8 text (chunk tree, notice, or error message). */
text: string
/** When a selector was used, whether it matched and which selector applied. */
chunk?: ChunkReadTarget
}
/**
* Options for `ChunkState.render`: which subtree to show and how anchors
* appear.
*/
export interface RenderParams {
/** Path of the chunk to render; `None` uses the tree root. */
chunkPath?: string
/** Title line shown above the tree (often the file path). */
title: string
/** Optional language label for the header block. */
languageTag?: string
/** Restrict output to an inclusive line range of the file. */
visibleRange?: VisibleLineRange
/** When true, list only direct children instead of a full subtree. */
renderChildrenOnly: boolean
/** Hide checksums in anchors when true. */
omitChecksum: boolean
/** Anchor formatting style for chunk headers. */
anchorStyle?: ChunkAnchorStyle
/** Include a one-line preview for leaf chunks. */
showLeafPreview: boolean
/** Replace tab characters in displayed previews (e.g. two spaces). */
tabReplacement?: string
/** When true, normalize displayed indentation to canonical tabs. */
normalizeIndent?: boolean
/**
* When set, restrict rendering to these chunks with their specified focus
* modes. Everything not in this list is skipped.
*/
focusedPaths?: Array<FocusedPath>
}
/** Sampling filter for resize operations. */
export declare enum SamplingFilter {
/** Nearest-neighbor sampling (fast, low quality). */
@@ -1395,17 +1095,6 @@ export declare function supportsLanguage(lang: string): boolean
*/
export declare function truncateToWidth(text: string, maxWidth: number, ellipsisKind: Ellipsis | undefined | null, pad: boolean | undefined | null, tabWidth: number): string
/**
* Inclusive 1-based line range within a source file (used for scoped chunk
* rendering).
*/
export interface VisibleLineRange {
/** First line to include. */
startLine: number
/** Last line to include. */
endLine: number
}
/**
* Calculate visible width of text, excluding ANSI escape sequences.
*
-31
View File
@@ -233,37 +233,6 @@ module.exports.AstMatchStrictness = {
Signature: 'signature',
Template: 'template',
};
module.exports.ChunkAnchorStyle = {
Full: 'full',
Kind: 'kind',
Bare: 'bare',
FullOmit: 'full-omit',
KindOmit: 'kind-omit',
None: 'none',
};
module.exports.ChunkEditOp = {
Put: 'put',
Replace: 'replace',
Delete: 'delete',
Before: 'before',
After: 'after',
Prepend: 'prepend',
Append: 'append',
};
module.exports.ChunkFocusMode = {
Expanded: 'expanded',
Collapsed: 'collapsed',
Container: 'container',
};
module.exports.ChunkReadStatus = {
Ok: 'ok',
NotFound: 'not_found',
UnsupportedRegion: 'unsupported_region',
};
module.exports.ChunkRegion = {
Head: '^',
Body: '~',
};
module.exports.Ellipsis = {
Unicode: 0,
Ascii: 1,
@@ -98,7 +98,7 @@ Options:
--tasks <ids> Comma-separated task IDs to run (default: all)
--max-tasks <n> Max tasks to sample (default: 80, 0 = all)
--fixtures <path> Fixtures directory or .tar.gz archive (default: built-in)
--edit-variant <v> Edit variant: any string (e.g. replace, patch, hashline, chunk, vim, atom, apply_patch), or auto (default: auto)
--edit-variant <v> Edit variant: any string (e.g. replace, patch, hashline, vim, atom, apply_patch), or auto (default: auto)
--edit-fuzzy <bool> Fuzzy matching: true, false, auto (default: auto)
--edit-fuzzy-threshold <n> Fuzzy threshold 0-1 or auto (default: auto)
--auto-format Auto-format output files after verify (debug only)
@@ -136,7 +136,7 @@ export function generateReport(result: BenchmarkResult): string {
if (typeof summary.mutationIntentMatchRate === "number") {
lines.push(`| Mutation Intent Match Rate | ${formatPercent(summary.mutationIntentMatchRate)} |`);
}
if (config.editVariant === "patch" || config.editVariant === "hashline" || config.editVariant === "chunk") {
if (config.editVariant === "patch" || config.editVariant === "hashline") {
lines.push(`| Patch Failure Rate | ${formatRate(totalEditFailures, totalEditAttempts)} |`);
}
lines.push(`| Tasks All Passing | ${summary.tasksWithAllPassing} |`);
@@ -188,24 +188,6 @@ export function generateReport(result: BenchmarkResult): string {
}
}
if (summary.chunkEditSubtypes) {
const order = ["append", "prepend", "replace", "delete"] as const;
const total = order.reduce((sum, key) => sum + (summary.chunkEditSubtypes?.[key] ?? 0), 0);
if (total > 0) {
lines.push("### Chunk Edit Subtypes");
lines.push("");
lines.push("| Operation | Count | % |");
lines.push("|-----------|-------|---|");
for (const key of order) {
const count = summary.chunkEditSubtypes[key] ?? 0;
const pct = formatPercent(count / total);
lines.push(`| ${key} | ${count} | ${pct} |`);
}
lines.push(`| **Total** | **${total}** | 100% |`);
lines.push("");
}
}
lines.push("## Task Results");
lines.push("");
lines.push("| Task | File | Success | Edit Hit | R/E/W | Tokens (In/Out) | Time | Indent |");
@@ -192,8 +192,6 @@ function getEditPathFromArgs(args: unknown): string | null {
const HASHLINE_SUBTYPES = ["set", "set_range", "insert"] as const;
const CHUNK_OP_SUBTYPES = ["append", "prepend", "replace", "delete"] as const;
const BENCHMARK_TOOL_NAMES = ["read", "edit", "vim", "write", "apply_patch"] as const;
const EDIT_TOOL_NAMES = ["edit", "vim", "apply_patch"] as const;
@@ -205,21 +203,6 @@ function isMutationTool(toolName: unknown): boolean {
return isEditTool(toolName) || toolName === "write";
}
function countChunkEditSubtypes(args: unknown): Record<string, number> {
const counts: Record<string, number> = Object.fromEntries(CHUNK_OP_SUBTYPES.map(k => [k, 0]));
if (!args || typeof args !== "object") return counts;
const operations = (args as { operations?: unknown[] }).operations;
if (!Array.isArray(operations)) return counts;
for (const operation of operations) {
if (!operation || typeof operation !== "object") continue;
const op = (operation as { op?: string }).op;
if (typeof op === "string" && op in counts) {
counts[op]++;
}
}
return counts;
}
function countHashlineEditSubtypes(args: unknown): Record<string, number> {
const counts: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0]));
if (!args || typeof args !== "object") return counts;
@@ -771,8 +754,6 @@ export interface TaskRunResult {
editAutocorrectCount: number;
/** Hashline edit subtype counts (replaceLine, replaceLines, etc.) — only when editVariant is hashline */
hashlineEditSubtypes?: Record<string, number>;
/** Chunk edit subtype counts — only when editVariant is chunk */
chunkEditSubtypes?: Record<string, number>;
mutationIntentMatched?: boolean;
mutationIntentReason?: string;
timeoutTelemetry?: PromptAttemptTelemetry;
@@ -838,8 +819,6 @@ export interface BenchmarkSummary {
mutationIntentMatchRate?: number;
/** Hashline edit subtype totals — only when editVariant is hashline */
hashlineEditSubtypes?: Record<string, number>;
/** Chunk edit subtype totals — only when editVariant is chunk */
chunkEditSubtypes?: Record<string, number>;
}
export interface BenchmarkResult {
@@ -931,7 +910,6 @@ async function runSingleTask(
totalInputChars: 0,
};
const hashlineSubtypes: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0]));
const chunkSubtypes: Record<string, number> = Object.fromEntries(CHUNK_OP_SUBTYPES.map(k => [k, 0]));
const logFile = path.join(TMP, `run-${task.id}-${runIndex}.jsonl`);
const logEvent = async (event: unknown) => {
@@ -1154,12 +1132,6 @@ async function runSingleTask(
hashlineSubtypes[key] += counts[key];
}
}
if (config.editVariant === "chunk" && args) {
const counts = countChunkEditSubtypes(args);
for (const key of CHUNK_OP_SUBTYPES) {
chunkSubtypes[key] += counts[key];
}
}
if (e.isError) {
toolStats.editFailures++;
const error = await appendNoChangeMutationHint(
@@ -1286,7 +1258,6 @@ async function runSingleTask(
editWarnings,
editAutocorrectCount,
hashlineEditSubtypes: config.editVariant === "hashline" ? hashlineSubtypes : undefined,
chunkEditSubtypes: config.editVariant === "chunk" ? chunkSubtypes : undefined,
mutationIntentMatched: mutationIntentValidation?.matched,
mutationIntentReason: mutationIntentValidation?.reason,
timeoutTelemetry,
@@ -1335,7 +1306,6 @@ async function _runRpcBenchmarkRun(
totalInputChars: 0,
};
const hashlineSubtypes: Record<string, number> = Object.fromEntries(HASHLINE_SUBTYPES.map(k => [k, 0]));
const chunkSubtypes: Record<string, number> = Object.fromEntries(CHUNK_OP_SUBTYPES.map(k => [k, 0]));
const logFile = path.join(sessionDir, `run-${task.id}-${runIndex}.jsonl`);
const logEvent = async (event: unknown) => {
@@ -1483,12 +1453,6 @@ async function _runRpcBenchmarkRun(
hashlineSubtypes[key] += counts[key];
}
}
if (config.editVariant === "chunk" && args) {
const counts = countChunkEditSubtypes(args);
for (const key of CHUNK_OP_SUBTYPES) {
chunkSubtypes[key] += counts[key];
}
}
if (e.isError) {
toolStats.editFailures++;
const toolError = await appendNoChangeMutationHint(
@@ -1609,7 +1573,6 @@ async function _runRpcBenchmarkRun(
editWarnings,
editAutocorrectCount,
hashlineEditSubtypes: config.editVariant === "hashline" ? hashlineSubtypes : undefined,
chunkEditSubtypes: config.editVariant === "chunk" ? chunkSubtypes : undefined,
mutationIntentMatched: mutationIntentValidation?.matched,
mutationIntentReason: mutationIntentValidation?.reason,
timeoutTelemetry,
@@ -2161,15 +2124,6 @@ export async function runBenchmark(
)
: undefined;
const chunkEditSubtypes: Record<string, number> | undefined =
config.editVariant === "chunk"
? Object.fromEntries(
CHUNK_OP_SUBTYPES.map(key => [
key,
allRuns.reduce((sum, r) => sum + (r.chunkEditSubtypes?.[key] ?? 0), 0),
]),
)
: undefined;
const denom = effectiveRuns || 1;
const summary: BenchmarkSummary = {
@@ -2212,7 +2166,6 @@ export async function runBenchmark(
transportFailureRuns,
mutationIntentMatchRate,
hashlineEditSubtypes,
chunkEditSubtypes,
};
return {
+2 -7
View File
@@ -2,11 +2,10 @@
"""
Edit benchmark: tests the edit tool across models with a simple edit task.
Select the edit variant via the PI_EDIT_VARIANT env var (e.g. `chunk`, `vim`,
Select the edit variant via the PI_EDIT_VARIANT env var (e.g. `vim`,
`hashline`, `replace`, `patch`, `apply_patch`) or `--variant`.
Examples:
PI_EDIT_VARIANT=chunk scripts/edit-benchmark.py
PI_EDIT_VARIANT=vim scripts/edit-benchmark.py
scripts/edit-benchmark.py --variant hashline
"""
@@ -54,11 +53,7 @@ def build_spec(variant: str) -> BenchmarkSpec:
f"```rust\n"
f"{EXPECTED_CONTENT}```\n"
)
retry = (
'Use `read(path="test.rs")` to refresh chunk selectors if needed, then try again using the edit tool.'
if variant == "chunk"
else f"Please try again using the edit tool {mode_phrase}."
)
retry = f"Please try again using the edit tool {mode_phrase}."
return BenchmarkSpec(
description=f"Benchmark edit tool in {variant} mode across models with simple edit tasks.",
workspace_prefix=f"{variant}-benchmark",
+1 -1
View File
@@ -720,7 +720,7 @@ MARKDOWN_FIXTURE = (
| Surface | Expected stress |
| --- | --- |
| `main.ts` | Structural chunk addressing |
| `main.ts` | Type/interface and class member edits |
| `main.rs` | Enum and impl member edits |
| `main.py` | Indentation-sensitive blocks |
| `main.md` | Prose and block-level text edits |