refactor(chunk): distributed AST classification to language modules

- Distributed language-specific AST classification logic from defaults module to individual language modules for 15+ languages.
- Simplified defaults module to minimal catch-all fallbacks, removing 341 lines of match-based classification logic.
- Consolidated chunk edit operations by removing splice operation and unifying with line-scoped replace parameters.
- Simplified chunk rendering output format by reducing thresholds and changing line count display from 'lines' to 'ln' suffix.
- Removed PI_CHUNK_SPLICES environment variable and chunkSplicesEnabled conditional logic throughout codebase.
This commit is contained in:
can1357
2026-04-06 20:52:07 +02:00
parent 79e2f8b2b1
commit 5e106e22db
28 changed files with 2409 additions and 1003 deletions
@@ -45,6 +45,37 @@ impl LangClassifier for ShellBuildClassifier {
Some(make_named_chunk(node, format!("define_{name}"), source, None))
},
"conditional" => Some(positional_candidate(node, "if", source)),
// Bash commands and pipelines
"command" | "pipeline" => Some(group_candidate(node, "stmts", source)),
// Bash control flow
"if_statement" => Some(positional_candidate(node, "if", source)),
"case_statement" => Some(positional_candidate(node, "switch", source)),
"while_statement" | "for_statement" => Some(positional_candidate(node, "loop", source)),
// Bash function definition
"function_definition" => Some(named_candidate(
node,
"fn",
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// Diff nodes
"hunks" => Some(group_candidate(node, "hunks", source)),
"file_change" => Some(named_candidate(node, "file", source, None)),
_ => None,
}
}
fn classify_class<'t>(&self, _node: Node<'t>, _source: &str) -> Option<RawChunkCandidate<'t>> {
None
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"if_statement" => Some(positional_candidate(node, "if", source)),
"case_statement" => Some(positional_candidate(node, "switch", source)),
"while_statement" | "for_statement" => Some(positional_candidate(node, "loop", source)),
"command" | "pipeline" => Some(group_candidate(node, "stmts", source)),
"subshell" => Some(positional_candidate(node, "block", source)),
_ => None,
}
}
+368 -7
View File
@@ -1,12 +1,373 @@
//! Language-specific chunk classifiers for C, C++, and Objective-C.
//!
//! These languages are well-served by the default classification rules.
//! `translation_unit` and `compilation_unit` root wrappers are already
//! recognized by the shared `is_root_wrapper_kind` helper, so no
//! overrides are needed here.
use super::classify::LangClassifier;
use tree_sitter::Node;
use super::{classify::LangClassifier, common::*, defaults::classify_var_decl};
pub struct CCppClassifier;
impl LangClassifier for CCppClassifier {}
// ── C/C++ declarator name extraction ────────────────────────────────────
/// Extract the function name from a C/C++ `function_definition` or
/// `function_declaration` node by traversing into the `declarator` chain.
fn extract_c_function_name(node: Node<'_>, source: &str) -> Option<String> {
let decl = node.child_by_field_name("declarator")?;
extract_c_declarator_name(decl, source)
}
/// Recursively resolve a C/C++ declarator to its leaf identifier.
/// Handles `function_declarator`, `pointer_declarator`, `reference_declarator`,
/// `qualified_identifier`, `destructor_name`, `template_function`, etc.
fn extract_c_declarator_name(node: Node<'_>, source: &str) -> Option<String> {
match node.kind() {
"identifier" | "field_identifier" | "type_identifier" => {
sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()))
},
"destructor_name" => {
// ~ClassName
sanitize_identifier(node_text(source, node.start_byte(), node.end_byte()))
},
"qualified_identifier" | "scoped_identifier" => {
// e.g. Entity::update — extract the "name" field or last identifier
node
.child_by_field_name("name")
.and_then(|n| extract_c_declarator_name(n, source))
.or_else(|| {
named_children(node)
.into_iter()
.rev()
.find(|c| {
matches!(
c.kind(),
"identifier" | "destructor_name" | "template_function" | "field_identifier"
)
})
.and_then(|c| extract_c_declarator_name(c, source))
})
},
"template_function" => {
// template_function has a "name" field or direct identifier child
node
.child_by_field_name("name")
.and_then(|n| sanitize_identifier(node_text(source, n.start_byte(), n.end_byte())))
.or_else(|| {
named_children(node)
.into_iter()
.find(|c| c.kind() == "identifier")
.and_then(|c| {
sanitize_identifier(node_text(source, c.start_byte(), c.end_byte()))
})
})
},
_ => {
// function_declarator, pointer_declarator, reference_declarator, etc.
// recurse into the "declarator" field
node
.child_by_field_name("declarator")
.and_then(|inner| extract_c_declarator_name(inner, source))
.or_else(|| {
// fallback: look for direct identifier-like child
named_children(node)
.into_iter()
.find(|c| {
matches!(
c.kind(),
"identifier"
| "field_identifier"
| "qualified_identifier"
| "scoped_identifier"
| "destructor_name"
| "template_function"
)
})
.and_then(|c| extract_c_declarator_name(c, source))
})
},
}
}
/// Extract the field name from a C/C++ `field_declaration` node.
/// The name sits in the `declarator` field which may be a plain
/// `field_identifier`, or a `function_declarator` / `pointer_declarator` etc.
fn extract_c_field_name(node: Node<'_>, source: &str) -> Option<String> {
let decl = node.child_by_field_name("declarator")?;
extract_c_declarator_name(decl, source)
}
/// Build a prefixed name for a C/C++ function node using declarator traversal.
fn c_prefixed_fn_name(prefix: &str, node: Node<'_>, source: &str) -> String {
let identifier =
extract_c_function_name(node, source).unwrap_or_else(|| "anonymous".to_string());
format!("{prefix}_{identifier}")
}
impl LangClassifier for CCppClassifier {
fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Imports ──
"include_directive" | "preproc_include" | "using_directive" | "using_statement"
| "import_declaration" | "module_import" => Some(group_candidate(node, "imports", source)),
// ── Functions ──
"function_definition" | "function_declaration" => Some(make_named_chunk(
node,
c_prefixed_fn_name("fn", node, source),
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
"constructor_definition" => Some(make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Templates (unwrap to find the inner declaration) ──
"template_declaration" => {
// Find the inner function_definition / class_specifier / etc.
let inner = named_children(node).into_iter().find(|c| {
matches!(
c.kind(),
"function_definition"
| "function_declaration"
| "class_specifier"
| "struct_specifier"
| "type_alias_declaration"
)
});
match inner {
Some(inner) => {
let mut candidate = self.classify_root(inner, source)?;
// Expand range to include the template<...> prefix
candidate.range_start_byte = node.start_byte();
candidate.range_start_line = node.start_position().row + 1;
candidate.checksum_start_byte = node.start_byte();
Some(candidate)
},
None => Some(named_candidate(
node,
"template",
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
}
},
// ── Containers ──
"class_specifier" | "class_declaration" | "class_interface" | "class_implementation" => {
Some(container_candidate(node, "class", source, recurse_class(node)))
},
"struct_specifier" | "struct_declaration" => {
Some(container_candidate(node, "struct", source, recurse_class(node)))
},
"enum_specifier" | "enum_declaration" => {
Some(container_candidate(node, "enum", source, recurse_enum(node)))
},
"namespace_definition" => {
Some(container_candidate(node, "mod", source, recurse_class(node)))
},
"union_declaration" => {
Some(container_candidate(node, "union", source, recurse_class(node)))
},
// ── Types ──
"type_alias_declaration" | "user_defined_type_definition" => {
Some(named_candidate(node, "type", source, recurse_class(node)))
},
// ── Variables / assignments ──
"variable_declaration" => Some(classify_var_decl(node, source)),
"assignment_statement" | "property_declaration" => {
Some(group_candidate(node, "decls", source))
},
// ── Macros ──
"macro_definition" => Some(named_candidate(
node,
"macro",
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Control flow (top-level scripts) ──
"if_statement" | "switch_statement" | "for_statement" | "while_statement"
| "do_statement" | "try_block" => Some(classify_function_c(node, source)),
// ── Statements ──
"expression_statement" => Some(group_candidate(node, "stmts", source)),
_ => None,
}
}
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Methods ──
"function_definition" | "function_declaration" | "method_declaration" => {
let name = extract_c_function_name(node, source)
.or_else(|| extract_identifier(node, source))
.unwrap_or_else(|| "anonymous".to_string());
if name == "constructor" {
Some(make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
))
} else {
Some(make_named_chunk(
node,
format!("fn_{name}"),
source,
recurse_body(node, ChunkContext::FunctionBody),
))
}
},
// ── Constructors ──
"constructor_definition" | "constructor_declaration" => Some(make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Fields ──
"field_declaration" => Some(match extract_c_field_name(node, source) {
Some(name) => make_named_chunk(node, format!("field_{name}"), source, None),
None => group_candidate(node, "fields", source),
}),
// ── Enum variants ──
"enum_constant" => Some(match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("variant_{name}"), source, None),
None => group_candidate(node, "variants", source),
}),
// ── Nested containers ──
"class_specifier" | "class_declaration" | "class_interface" | "class_implementation" => {
Some(container_candidate(node, "class", source, recurse_class(node)))
},
"struct_specifier" | "struct_declaration" => {
Some(container_candidate(node, "struct", source, recurse_class(node)))
},
"enum_specifier" | "enum_declaration" => {
Some(container_candidate(node, "enum", source, recurse_enum(node)))
},
"union_declaration" => {
Some(container_candidate(node, "union", source, recurse_class(node)))
},
"namespace_definition" => {
Some(container_candidate(node, "mod", source, recurse_class(node)))
},
// ── Templates (class body) ──
"template_declaration" => {
let inner = named_children(node).into_iter().find(|c| {
matches!(
c.kind(),
"function_definition"
| "function_declaration"
| "class_specifier"
| "struct_specifier"
| "type_alias_declaration"
)
});
match inner {
Some(inner) => {
let mut candidate = self.classify_class(inner, source)?;
candidate.range_start_byte = node.start_byte();
candidate.range_start_line = node.start_position().row + 1;
candidate.checksum_start_byte = node.start_byte();
Some(candidate)
},
None => Some(named_candidate(
node,
"template",
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
}
},
// ── Types ──
"type_alias_declaration" => Some(named_candidate(node, "type", source, None)),
_ => None,
}
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(classify_function_c(node, source))
}
}
fn classify_function_c<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
match node.kind() {
"if_statement" => {
make_candidate(node, "if".to_string(), NameStyle::Named, None, fn_recurse(), false, source)
},
"switch_statement" => make_candidate(
node,
"switch".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"try_block" | "catch_clause" | "finally_clause" => make_candidate(
node,
"try".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"for_statement" => make_candidate(
node,
"for".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"while_statement" => make_candidate(
node,
"while".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"do_statement" => make_candidate(
node,
"block".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"variable_declaration" => {
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
if span > 1 {
if let Some(name) = extract_single_declarator_name(node, source) {
make_named_chunk(node, format!("var_{name}"), source, None)
} else {
group_candidate(node, "variable", source)
}
} else {
group_candidate(node, "variable", source)
}
},
_ => {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
},
}
}
+249 -7
View File
@@ -1,12 +1,254 @@
//! Language-specific chunk classifiers for C# and Java.
//!
//! Both languages are well-served by the default classification rules:
//! class/interface/enum/record declarations, method/constructor declarations,
//! using/import directives, namespace/package declarations, and property/event
//! declarations all fall through to defaults.
use super::classify::LangClassifier;
use tree_sitter::Node;
use super::{classify::LangClassifier, common::*, defaults::classify_var_decl};
pub struct CSharpJavaClassifier;
impl LangClassifier for CSharpJavaClassifier {}
impl LangClassifier for CSharpJavaClassifier {
fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Imports ──
"import_declaration"
| "using_directive"
| "package_declaration"
| "namespace_statement" => Some(group_candidate(node, "imports", source)),
// ── Functions ──
"method_declaration" => Some(named_candidate(
node,
"meth",
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
"function_declaration" | "function_definition" => Some(named_candidate(
node,
"fn",
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Constructors ──
"constructor_declaration" => Some(make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Containers ──
"class_declaration" => {
Some(container_candidate(node, "class", source, recurse_class(node)))
},
"interface_declaration" => {
Some(container_candidate(node, "iface", source, recurse_interface(node)))
},
"enum_declaration" => Some(container_candidate(node, "enum", source, recurse_enum(node))),
"namespace_declaration" | "file_scoped_namespace_declaration" => {
Some(container_candidate(node, "mod", source, recurse_class(node)))
},
"struct_declaration" | "record_declaration" => {
Some(container_candidate(node, "struct", source, recurse_class(node)))
},
// ── Types ──
"type_alias_declaration" => {
Some(named_candidate(node, "type", source, recurse_class(node)))
},
// ── Variables / assignments ──
"variable_declaration" | "lexical_declaration" => Some(classify_var_decl(node, source)),
"property_declaration" | "state_variable_declaration" => {
Some(group_candidate(node, "decls", source))
},
// ── Control flow (top-level scripts) ──
"if_statement" | "switch_statement" | "switch_expression" | "for_statement"
| "foreach_statement" | "while_statement" | "do_statement" | "try_statement" => {
Some(classify_function_csharp_java(node, source))
},
// ── Statements ──
"expression_statement" => Some(group_candidate(node, "stmts", source)),
_ => None,
}
}
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Container declarations (inside namespace/class bodies) ──
"class_declaration" => {
Some(container_candidate(node, "class", source, recurse_class(node)))
},
"interface_declaration" => {
Some(container_candidate(node, "iface", source, recurse_interface(node)))
},
"enum_declaration" => Some(container_candidate(node, "enum", source, recurse_enum(node))),
"struct_declaration" | "record_declaration" => {
Some(container_candidate(node, "struct", source, recurse_class(node)))
},
"namespace_declaration" | "file_scoped_namespace_declaration" => {
Some(container_candidate(node, "mod", source, recurse_class(node)))
},
// ── Methods ──
"method_declaration" | "function_declaration" | "function_definition" => {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
if name == "constructor" {
Some(make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
))
} else {
Some(make_named_chunk(
node,
format!("fn_{name}"),
source,
recurse_body(node, ChunkContext::FunctionBody),
))
}
},
// ── Constructors ──
"constructor_declaration" | "secondary_constructor" => Some(make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Fields ──
"field_declaration"
| "property_declaration"
| "constant_declaration"
| "event_field_declaration" => Some(match extract_field_name(node, source) {
Some(name) => make_named_chunk(node, format!("field_{name}"), source, None),
None => group_candidate(node, "fields", source),
}),
// ── Enum members ──
"enum_member_declaration" | "enum_constant" | "enum_entry" => {
Some(match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("variant_{name}"), source, None),
None => group_candidate(node, "variants", source),
})
},
// ── Static blocks ──
"class_static_block" => {
Some(make_named_chunk(node, "static_init".to_string(), source, None))
},
_ => None,
}
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(classify_function_csharp_java(node, source))
}
}
/// Extract the variable name from a field/constant declaration.
///
/// Java `field_declaration` has the structure:
/// `field_declaration` { modifiers, type: `type_identifier`, declarator:
/// `variable_declarator` { name: identifier } }
///
/// `extract_identifier` would find `type_identifier` first, so we look into
/// `variable_declarator` children for the actual variable name.
fn extract_field_name(node: Node<'_>, source: &str) -> Option<String> {
for child in named_children(node) {
if child.kind() == "variable_declarator" {
return extract_identifier(child, source);
}
}
extract_identifier(node, source)
}
fn classify_function_csharp_java<'tree>(
node: Node<'tree>,
source: &str,
) -> RawChunkCandidate<'tree> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
match node.kind() {
"if_statement" => {
make_candidate(node, "if".to_string(), NameStyle::Named, None, fn_recurse(), false, source)
},
"switch_statement" | "switch_expression" => make_candidate(
node,
"switch".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"try_statement" | "catch_clause" | "finally_clause" => make_candidate(
node,
"try".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"for_statement" => make_candidate(
node,
"for".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"foreach_statement" => make_candidate(
node,
"for".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"while_statement" => make_candidate(
node,
"while".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"do_statement" => make_candidate(
node,
"block".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"variable_declaration" | "lexical_declaration" => {
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
if span > 1 {
if let Some(name) = extract_single_declarator_name(node, source) {
make_named_chunk(node, format!("var_{name}"), source, None)
} else {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
}
} else {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
}
},
_ => {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
},
}
}
+10
View File
@@ -64,6 +64,8 @@ fn classify_css_node<'t>(node: Node<'t>, source: &str) -> Option<RawChunkCandida
source,
Some(recurse_self(node, ChunkContext::ClassBody)),
)),
// Top-level or nested property declarations.
"declaration" => Some(group_candidate(node, "fields", source)),
_ => None,
}
}
@@ -77,6 +79,14 @@ impl LangClassifier for CssClassifier {
classify_css_node(node, source)
}
fn classify_function<'t>(
&self,
_node: Node<'t>,
_source: &str,
) -> Option<RawChunkCandidate<'t>> {
None
}
fn is_root_wrapper(&self, kind: &str) -> bool {
kind == "stylesheet"
}
@@ -14,6 +14,14 @@ impl LangClassifier for DataFormatsClassifier {
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
classify_data_node(node, source, false)
}
fn classify_function<'t>(
&self,
_node: Node<'t>,
_source: &str,
) -> Option<RawChunkCandidate<'t>> {
None
}
}
fn classify_data_node<'t>(
@@ -88,13 +96,15 @@ fn classify_data_node<'t>(
}
}
/// Extract key from a JSON `pair` node: unquote and sanitize.
/// Extract key from a `pair` node (JSON or TOML).
/// JSON pairs have a `"key"` field; TOML pairs have no field names, so we fall
/// back to looking for the first `bare_key`, `quoted_key`, or `dotted_key`
/// child.
fn extract_pair_key(node: Node<'_>, source: &str) -> Option<String> {
node.child_by_field_name("key").and_then(|key| {
sanitize_identifier(
unquote_text(node_text(source, key.start_byte(), key.end_byte())).as_str(),
)
})
let key = node
.child_by_field_name("key")
.or_else(|| child_by_kind(node, &["bare_key", "quoted_key", "dotted_key"]))?;
sanitize_identifier(unquote_text(node_text(source, key.start_byte(), key.end_byte())).as_str())
}
/// Extract key from a YAML `block_mapping_pair` or `flow_pair` node.
+19
View File
@@ -89,6 +89,8 @@ fn call_name(node: Node<'_>, source: &str) -> Option<String> {
// The first named child after the target is typically `arguments`.
// For `def run(x)`, arguments contains a `call` node whose target is `run`.
// For `defmodule App`, arguments contains an `alias` node with text `App`.
// For `def run(x) when is_integer(x)`, arguments contains a `binary_operator`
// with the call on the left and the guard on the right.
// Extract the meaningful name, not the full text with parameters.
named_children(node).into_iter().skip(1).find_map(|child| {
if child.kind() == "do_block" {
@@ -100,6 +102,16 @@ fn call_name(node: Node<'_>, source: &str) -> Option<String> {
if arg.kind() == "call" {
// `def run(x)` → arguments has call(target=run), extract target name
call_target(arg, source).and_then(|t| sanitize_identifier(&t))
} else if arg.kind() == "binary_operator" {
// `def run(x) when guard` → binary_operator(left=call, right=guard)
// Extract name from the left side (the actual function call).
arg.child_by_field_name("left").and_then(|left| {
if left.kind() == "call" {
call_target(left, source).and_then(|t| sanitize_identifier(&t))
} else {
sanitize_identifier(node_text(source, left.start_byte(), left.end_byte()))
}
})
} else {
// `defmodule App` → arguments has alias("App")
sanitize_identifier(node_text(source, arg.start_byte(), arg.end_byte()))
@@ -131,4 +143,11 @@ impl LangClassifier for ElixirClassifier {
_ => None,
}
}
fn is_trivia(&self, kind: &str) -> bool {
// `@doc`, `@spec`, `@impl`, `@type`, `@moduledoc`, etc. are all
// `unary_operator` nodes in the Elixir grammar (operator `@`).
// Treat them as trivia so they get absorbed into the next chunk.
kind == "unary_operator"
}
}
+165 -4
View File
@@ -14,7 +14,135 @@ pub struct GoClassifier;
impl LangClassifier for GoClassifier {
fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Imports / package ──
"import_declaration" | "package_clause" => Some(group_candidate(node, "imports", source)),
// ── Variables ──
"const_declaration" | "var_declaration" | "short_var_declaration" => {
Some(match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("var_{name}"), source, None),
None => group_candidate(node, "decls", source),
})
},
// ── Functions ──
"function_declaration" => Some(named_candidate(
node,
"fn",
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
"method_declaration" => Some(named_candidate(
node,
"fn",
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Containers ──
"type_declaration" => Some(classify_type_decl(node, source)),
// ── Control flow (top-level scripts) ──
"if_statement"
| "switch_statement"
| "expression_switch_statement"
| "type_switch_statement"
| "select_statement"
| "for_statement" => Some(classify_function_go(node, source)),
// ── Statements ──
"expression_statement" | "go_statement" | "defer_statement" | "send_statement" => {
Some(group_candidate(node, "stmts", source))
},
_ => None,
}
}
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Methods ──
"method_spec" => Some(named_candidate(node, "meth", source, None)),
// ── Fields ──
"field_declaration" | "embedded_field" => Some(match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("field_{name}"), source, None),
None => group_candidate(node, "fields", source),
}),
// ── Field / method lists ──
"field_declaration_list" => Some(group_candidate(node, "fields", source)),
"method_spec_list" => Some(group_candidate(node, "methods", source)),
_ => None,
}
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Control flow ──
"if_statement" => Some(make_candidate(
node,
"if".to_string(),
NameStyle::Named,
None,
recurse_body(node, ChunkContext::FunctionBody),
false,
source,
)),
"switch_statement" | "expression_switch_statement" | "type_switch_statement" => {
Some(make_candidate(
node,
"switch".to_string(),
NameStyle::Named,
None,
recurse_body(node, ChunkContext::FunctionBody),
false,
source,
))
},
"select_statement" => Some(make_candidate(
node,
"switch".to_string(),
NameStyle::Named,
None,
recurse_body(node, ChunkContext::FunctionBody),
false,
source,
)),
// ── Loops ──
"for_statement" => Some(make_candidate(
node,
"for".to_string(),
NameStyle::Named,
None,
recurse_body(node, ChunkContext::FunctionBody),
false,
source,
)),
// ── Blocks ──
"go_statement" | "defer_statement" | "send_statement" => {
Some(group_candidate(node, "stmts", source))
},
// ── Variables ──
"short_var_declaration" | "var_declaration" | "const_declaration" => {
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
Some(if span > 1 {
if let Some(name) = extract_identifier(node, source) {
make_named_chunk(node, format!("var_{name}"), source, None)
} else {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
}
} else {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
})
},
_ => None,
}
}
@@ -30,6 +158,39 @@ impl LangClassifier for GoClassifier {
}
}
/// Classify Go function-level nodes (reused for top-level control flow
/// delegation).
fn classify_function_go<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
match node.kind() {
"if_statement" => {
make_candidate(node, "if".to_string(), NameStyle::Named, None, fn_recurse(), false, source)
},
"switch_statement"
| "expression_switch_statement"
| "type_switch_statement"
| "select_statement" => make_candidate(
node,
"switch".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"for_statement" => make_candidate(
node,
"for".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
_ => group_candidate(node, "stmts", source),
}
}
/// Classify Go `type_declaration` nodes.
///
/// A single `type_spec` with a struct/interface body becomes a container;
@@ -77,7 +238,7 @@ fn reparent_receiver_methods(
root_children: &mut Vec<String>,
source: &str,
) {
// Build map: type name → chunk path for root-level type chunks.
// Build map: type name -> chunk path for root-level type chunks.
let type_paths: HashMap<String, String> = chunks
.iter()
.filter(|c| c.parent_path.as_deref() == Some("") && c.path.starts_with("type_"))
@@ -180,7 +341,7 @@ fn reparent_new_type_constructors(
sort_chunk_children_by_position(chunks);
}
/// `fn_NewServer` + type `Server` → `Some("Server")`; `fn_Start` → None.
/// `fn_NewServer` + type `Server` -> `Some("Server")`; `fn_Start` -> None.
fn constructor_suffix_after_new(fn_path: &str) -> Option<String> {
let name = fn_path.strip_prefix("fn_")?;
let tail = name.strip_prefix("New")?;
@@ -192,8 +353,8 @@ fn constructor_suffix_after_new(fn_path: &str) -> Option<String> {
/// Extract the receiver type name from a Go method's header.
///
/// `func (s *Server) Start()` → `Some("Server")`
/// `func (s Server) Stop()` → `Some("Server")`
/// `func (s *Server) Start()` -> `Some("Server")`
/// `func (s Server) Stop()` -> `Some("Server")`
fn extract_receiver_type_name(chunk: &ChunkNode, source: &str) -> Option<String> {
let header = normalized_header(source, chunk.start_byte as usize, chunk.end_byte as usize);
let receiver = header
@@ -1,7 +1,127 @@
//! Language-specific chunk classifiers for Haskell and Scala.
use super::classify::LangClassifier;
use tree_sitter::Node;
use super::{classify::LangClassifier, common::*};
pub struct HaskellScalaClassifier;
impl LangClassifier for HaskellScalaClassifier {}
impl LangClassifier for HaskellScalaClassifier {
fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
// ── Imports / packages ──
"import_declaration" => group_candidate(node, "imports", source),
"package_declaration" => group_candidate(node, "imports", source),
// ── Haskell module ──
"module" => container_candidate(node, "mod", source, recurse_class(node)),
// ── Functions ──
"function_declaration" => {
named_candidate(node, "fn", source, recurse_body(node, ChunkContext::FunctionBody))
},
"function_definition" => {
named_candidate(node, "fn", source, recurse_body(node, ChunkContext::FunctionBody))
},
// ── Containers (Scala) ──
"class_definition" => container_candidate(node, "class", source, recurse_class(node)),
"object_definition" => container_candidate(node, "mod", source, recurse_class(node)),
"trait_definition" => container_candidate(node, "iface", source, recurse_interface(node)),
// ── Types ──
"type_alias_declaration" | "type_item" => {
named_candidate(node, "type", source, recurse_class(node))
},
// ── Variables / assignments ──
"variable_declaration" | "assignment" => group_candidate(node, "decls", source),
// ── Statements ──
"expression_statement" => group_candidate(node, "stmts", source),
_ => return None,
})
}
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
// ── Methods ──
"function_declaration" | "function_definition" | "method_definition" => {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
if name == "constructor" {
make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
)
} else {
make_named_chunk(
node,
format!("fn_{name}"),
source,
recurse_body(node, ChunkContext::FunctionBody),
)
}
},
// ── Fields ──
"variable_declaration" | "property_declaration" => {
match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("field_{name}"), source, None),
None => group_candidate(node, "fields", source),
}
},
_ => return None,
})
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
Some(match node.kind() {
// ── Control flow ──
"if_statement" => make_candidate(
node,
"if".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"match_expression" => make_candidate(
node,
"match".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"for_expression" | "while_expression" => make_candidate(
node,
"loop".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
// ── Blocks ──
"block_expression" => make_candidate(
node,
"block".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
_ => return None,
})
}
}
@@ -56,4 +56,12 @@ impl LangClassifier for HtmlXmlClassifier {
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
classify_element(node, source)
}
fn classify_function<'t>(
&self,
_node: Node<'t>,
_source: &str,
) -> Option<RawChunkCandidate<'t>> {
None
}
}
+160 -2
View File
@@ -9,21 +9,85 @@ pub struct JsTsClassifier;
impl LangClassifier for JsTsClassifier {
fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Exports / decorators ──
"export_statement" => Some(classify_export_statement(node, source)),
"decorated_definition" => Some(classify_decorated(node, source)),
// ── Imports ──
"import_statement" | "import_declaration" => {
Some(group_candidate(node, "imports", source))
},
// ── Variables ──
"lexical_declaration" | "variable_declaration" => {
// Must handle here to ensure promote_assigned_expression runs.
// The shared defaults classify_var_decl should do this, but we
// need direct control for JS/TS patterns.
Some(classify_var_decl_js(node, source))
},
// ── Functions ──
"function_declaration" => Some(named_candidate(
node,
"fn",
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Containers ──
"class_declaration" => {
Some(container_candidate(node, "class", source, recurse_class(node)))
},
"interface_declaration" => {
Some(container_candidate(node, "iface", source, recurse_interface(node)))
},
"enum_declaration" => Some(container_candidate(node, "enum", source, recurse_enum(node))),
"internal_module" => Some(container_candidate(node, "mod", source, recurse_class(node))),
// ── Types ──
"type_alias_declaration" => Some(named_candidate(node, "type", source, None)),
// ── Control flow at top level ──
"if_statement" | "switch_statement" | "switch_expression" | "try_statement"
| "for_statement" | "for_in_statement" | "for_of_statement" | "while_statement"
| "do_statement" | "with_statement" => Some(classify_function_js(node, source)),
// ── Statements ──
"expression_statement" => {
// Unwrap `expression_statement` wrapping an `internal_module` (namespace).
let inner = named_children(node)
.into_iter()
.find(|c| c.kind() == "internal_module");
if let Some(ns) = inner {
Some(container_candidate(ns, "mod", source, recurse_class(ns)))
} else {
Some(group_candidate(node, "stmts", source))
}
},
_ => None,
}
}
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
"method_definition" => {
// ── Exports / decorators (re-exported members) ──
"export_statement" => Some(classify_export_statement(node, source)),
"decorated_definition" => Some(classify_decorated(node, source)),
// ── Variables ──
"lexical_declaration" | "variable_declaration" => Some(classify_var_decl_js(node, source)),
// ── Constructor ──
"constructor" => Some(make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
)),
// ── Methods ──
"method_definition" | "method_signature" | "abstract_method_signature" => {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
if name == "constructor" {
Some(make_named_chunk(
@@ -33,12 +97,92 @@ impl LangClassifier for JsTsClassifier {
recurse_body(node, ChunkContext::FunctionBody),
))
} else {
None // let defaults handle normal methods
Some(make_named_chunk(
node,
format!("fn_{name}"),
source,
recurse_body(node, ChunkContext::FunctionBody),
))
}
},
// ── Fields ──
"public_field_definition"
| "field_definition"
| "property_definition"
| "property_signature"
| "property_declaration"
| "abstract_class_field" => match extract_identifier(node, source) {
Some(name) => Some(make_named_chunk(node, format!("field_{name}"), source, None)),
None => Some(group_candidate(node, "fields", source)),
},
// ── Enum members ──
"enum_assignment" | "enum_member_declaration" => match extract_identifier(node, source) {
Some(name) => Some(make_named_chunk(node, format!("variant_{name}"), source, None)),
None => Some(group_candidate(node, "variants", source)),
},
// ── Static blocks ──
"class_static_block" => {
Some(make_named_chunk(node, "static_init".to_string(), source, None))
},
// ── Types ──
"type_alias_declaration" => Some(named_candidate(node, "type", source, None)),
_ => None,
}
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(classify_function_js(node, source))
}
}
// ── Function body classification (JS/TS) ────────────────────────────────
/// Classify nodes inside a function body for JS/TS.
fn classify_function_js<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
match node.kind() {
// ── Control flow ──
"if_statement" => make_named_chunk(node, "if".to_string(), source, fn_recurse()),
"switch_statement" | "switch_expression" => {
make_named_chunk(node, "switch".to_string(), source, fn_recurse())
},
"try_statement" => make_named_chunk(node, "try".to_string(), source, fn_recurse()),
// ── Loops ──
"for_statement" => make_named_chunk(node, "for".to_string(), source, fn_recurse()),
"for_in_statement" => make_named_chunk(node, "for_in".to_string(), source, fn_recurse()),
"for_of_statement" => make_named_chunk(node, "for_of".to_string(), source, fn_recurse()),
"while_statement" => make_named_chunk(node, "while".to_string(), source, fn_recurse()),
"do_statement" => make_named_chunk(node, "block".to_string(), source, fn_recurse()),
// ── Blocks ──
"with_statement" => make_named_chunk(node, "block".to_string(), source, fn_recurse()),
// ── Variables ──
"lexical_declaration" | "variable_declaration" => {
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
if span > 1 {
if let Some(name) = extract_single_declarator_name(node, source) {
make_named_chunk(node, format!("var_{name}"), source, None)
} else {
group_candidate(node, &sanitize_node_kind(node.kind()), source)
}
} else {
group_candidate(node, &sanitize_node_kind(node.kind()), source)
}
},
// ── Fallback ──
_ => {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
},
}
}
// ── Variable declaration (JS/TS) ────────────────────────────────────────
@@ -158,6 +302,20 @@ fn classify_export_statement<'t>(node: Node<'t>, source: &str) -> RawChunkCandid
)
}
},
"internal_module" => {
let recurse = recurse_class(child);
if is_default {
make_container_chunk_from(node, child, "default_export".to_string(), source, recurse)
} else {
make_container_chunk_from(
node,
child,
prefixed_name("mod", child, source),
source,
recurse,
)
}
},
"lexical_declaration" | "variable_declaration" => {
if is_default {
make_named_chunk_from(node, child, "default_export".to_string(), source, None)
@@ -77,6 +77,14 @@ impl LangClassifier for MarkupClassifier {
_ => None,
}
}
fn classify_function<'t>(
&self,
_node: Node<'t>,
_source: &str,
) -> Option<RawChunkCandidate<'t>> {
None
}
}
/// Extract heading text from a Markdown `section` node's `atx_heading` or
+390 -2
View File
@@ -1,8 +1,396 @@
//! Chunk classifiers for languages well-served by defaults:
//! Kotlin, Swift, PHP, Solidity, Julia, Odin, Verilog, Zig, Regex, Diff.
//!
//! This is the catch-all classifier: it handles every node kind that any of the
//! miscellaneous languages produce so that nothing silently falls through.
use super::classify::LangClassifier;
use tree_sitter::Node;
use super::{classify::LangClassifier, common::*, defaults::classify_var_decl};
pub struct MiscClassifier;
impl LangClassifier for MiscClassifier {}
impl LangClassifier for MiscClassifier {
fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let fn_recurse = || {
recurse_body(node, ChunkContext::FunctionBody)
.or_else(|| recurse_into(node, ChunkContext::FunctionBody, &["body"], &["block"]))
};
Some(match node.kind() {
// ── Imports / package headers ──
"import_statement"
| "import_declaration"
| "using_directive"
| "using_statement"
| "namespace_use_declaration"
| "namespace_statement"
| "import_list"
| "import_header"
| "package_header"
| "package_declaration" => group_candidate(node, "imports", source),
// ── Variables / assignments ──
"lexical_declaration" | "variable_declaration" => classify_var_decl(node, source),
"const_declaration" | "var_declaration" => match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("var_{name}"), source, None),
None => group_candidate(node, "decls", source),
},
"assignment" | "property_declaration" | "state_variable_declaration" => {
group_candidate(node, "decls", source)
},
// ── Statements ──
"expression_statement" | "global_statement" | "command" | "pipeline" | "function_call" => {
group_candidate(node, "stmts", source)
},
// ── Functions ──
"function_declaration"
| "function_definition"
| "procedure_declaration"
| "overloaded_procedure_declaration"
| "test_declaration" => named_candidate(node, "fn", source, fn_recurse()),
"method_declaration" => {
named_candidate(node, "meth", source, recurse_body(node, ChunkContext::FunctionBody))
},
"constructor_definition"
| "constructor_declaration"
| "secondary_constructor"
| "init_declaration"
| "fallback_receive_definition" => make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
),
// ── Containers ──
"class_declaration" | "class_definition" => {
container_candidate(node, "class", source, recurse_class(node))
},
"interface_declaration" | "protocol_declaration" => {
container_candidate(node, "iface", source, recurse_interface(node))
},
"struct_declaration" | "object_declaration" => {
container_candidate(node, "struct", source, recurse_class(node))
},
"enum_declaration" | "enum_definition" => {
container_candidate(node, "enum", source, recurse_enum(node))
},
"trait_definition" | "class" => {
container_candidate(node, "trait", source, recurse_class(node))
},
"contract_declaration" | "library_declaration" | "trait_declaration" => {
container_candidate(node, "contract", source, recurse_class(node))
},
"namespace_declaration" | "module_definition" | "extension_definition" => {
container_candidate(node, "mod", source, recurse_class(node))
},
// ── Types / aliases ──
"type_alias_declaration" | "const_type_declaration" | "opaque_declaration" => {
named_candidate(node, "type", source, recurse_class(node))
},
// ── Macros ──
"macro_definition" | "modifier_definition" => {
named_candidate(node, "macro", source, recurse_body(node, ChunkContext::FunctionBody))
},
// ── Systems (Verilog etc.) ──
"covergroup_declaration" | "checker_declaration" => {
container_candidate(node, "group", source, recurse_class(node))
},
"module_declaration" => container_candidate(node, "mod", source, recurse_class(node)),
"union_declaration" => container_candidate(node, "union", source, recurse_class(node)),
// ── Control flow at top level → delegate to function-level ──
"if_statement"
| "unless"
| "guard_statement"
| "switch_statement"
| "switch_expression"
| "case_statement"
| "expression_switch_statement"
| "type_switch_statement"
| "select_statement"
| "try_statement"
| "try_block"
| "for_statement"
| "for_in_statement"
| "for_of_statement"
| "foreach_statement"
| "while_statement"
| "do_statement"
| "with_statement" => return self.classify_function(node, source),
_ => return None,
})
}
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
// ── Constructors ──
"constructor"
| "constructor_declaration"
| "secondary_constructor"
| "init_declaration" => make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
),
// ── Methods ──
"method_definition"
| "method_signature"
| "abstract_method_signature"
| "method_declaration"
| "function_declaration"
| "function_definition"
| "function_item"
| "procedure_declaration"
| "protocol_function_declaration"
| "method"
| "singleton_method" => {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
if name == "constructor" {
make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
)
} else {
make_named_chunk(
node,
format!("fn_{name}"),
source,
recurse_body(node, ChunkContext::FunctionBody),
)
}
},
// ── Fields (named properties) ──
"public_field_definition"
| "field_definition"
| "property_definition"
| "property_signature"
| "property_declaration"
| "protocol_property_declaration"
| "abstract_class_field"
| "const_declaration"
| "constant_declaration"
| "event_field_declaration" => match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("field_{name}"), source, None),
None => group_candidate(node, "fields", source),
},
// ── Enum variants ──
"enum_assignment"
| "enum_member_declaration"
| "enum_constant"
| "enum_entry"
| "enum_variant" => match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("variant_{name}"), source, None),
None => group_candidate(node, "variants", source),
},
// ── Other fields ──
"field_declaration" | "embedded_field" | "container_field" | "binding" => {
match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("field_{name}"), source, None),
None => group_candidate(node, "fields", source),
}
},
// ── Method specs ──
"method_spec" => named_candidate(node, "meth", source, None),
// ── Field / method lists ──
"field_declaration_list" => group_candidate(node, "fields", source),
"method_spec_list" => group_candidate(node, "methods", source),
// ── Static initializer ──
"class_static_block" => make_named_chunk(node, "static_init".to_string(), source, None),
// ── Decorated definitions ──
"decorated_definition" => {
let inner = named_children(node)
.into_iter()
.find(|c| c.kind() == "function_definition");
if let Some(child) = inner {
let name =
extract_identifier(child, source).unwrap_or_else(|| "anonymous".to_string());
make_named_chunk(node, format!("fn_{name}"), source, {
let context = ChunkContext::FunctionBody;
recurse_into(child, context, &["body"], &["block"])
})
} else {
return None;
}
},
// ── Grouped field-like entries ──
"assignment"
| "expression_statement"
| "attribute"
| "pair"
| "block_mapping_pair"
| "flow_pair" => group_candidate(node, "fields", source),
// ── Types inside classes ──
"type_item" | "type_alias_declaration" | "type_alias" => {
named_candidate(node, "type", source, None)
},
// ── Const / macro inside classes ──
"const_item" | "macro_invocation" => group_candidate(node, "fields", source),
_ => return None,
})
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
Some(match node.kind() {
// ── Control flow: conditionals ──
"if_statement" | "unless" | "guard_statement" => make_candidate(
node,
"if".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
// ── Control flow: switches ──
"switch_statement"
| "switch_expression"
| "case_statement"
| "case_match"
| "expression_switch_statement"
| "type_switch_statement"
| "select_statement"
| "receive_statement"
| "yul_switch_statement" => make_candidate(
node,
"switch".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
// ── Control flow: try/catch ──
"try_statement" | "try_block" | "catch_clause" | "finally_clause"
| "assembly_statement" => make_candidate(
node,
"try".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
// ── Loops: for variants (with Python-like check) ──
"for_statement" | "for_in_statement" | "for_of_statement" => {
let name = if looks_like_python_statement(node, source) {
"loop".to_string()
} else {
sanitize_node_kind(node.kind())
};
make_candidate(node, name, NameStyle::Named, None, fn_recurse(), false, source)
},
// ── Loops: while ──
"while_statement" => {
let name = if looks_like_python_statement(node, source) {
"loop"
} else {
"while"
};
make_candidate(
node,
name.to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
)
},
// ── Blocks ──
"do_statement" | "with_statement" | "do_block" | "subshell" | "async_block"
| "unsafe_block" | "const_block" | "block_expression" => make_candidate(
node,
"block".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
// ── Loops: foreach ──
"foreach_statement" => make_candidate(
node,
"for".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
// ── Statements ──
"defer_statement" | "go_statement" | "send_statement" => {
group_candidate(node, "stmts", source)
},
// ── Positional candidates ──
"elif_clause" => positional_candidate(node, "elif", source),
"except_clause" => positional_candidate(node, "except", source),
"when_statement" => positional_candidate(node, "when", source),
"match_expression" | "match_block" => positional_candidate(node, "match", source),
// ── Loops / misc expressions ──
"loop_expression"
| "while_expression"
| "for_expression"
| "errdefer_statement"
| "comptime_statement"
| "nosuspend_statement"
| "suspend_statement"
| "yul_if_statement"
| "yul_for_statement" => positional_candidate(node, "loop", source),
// ── Variable declarations ──
"lexical_declaration"
| "variable_declaration"
| "const_declaration"
| "var_declaration"
| "short_var_declaration"
| "let_declaration" => {
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
if span > 1 {
if let Some(name) = extract_single_declarator_name(node, source) {
make_named_chunk(node, format!("var_{name}"), source, None)
} else {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
}
} else {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
}
},
_ => return None,
})
}
}
@@ -80,6 +80,23 @@ impl LangClassifier for NixHclClassifier {
Some(group_candidate(node, "hunks", source))
}
},
// Nix expressions
"function_expression" | "let_expression" => {
Some(named_candidate(node, "expr", source, recurse_value_container(node)))
},
// Nix inherit
"inherit" => Some(group_candidate(node, "imports", source)),
// Variable/assignment declarations
"variable_declaration" | "assignment" => Some(group_candidate(node, "decls", source)),
// HCL top-level block types
"provider" | "resource" | "data" | "locals" | "variable" | "output" | "module" => {
Some(container_candidate(
node,
sanitize_node_kind(node.kind()).as_str(),
source,
recurse_into(node, ChunkContext::ClassBody, &[], &["body"]),
))
},
_ => None,
}
}
@@ -109,4 +126,13 @@ impl LangClassifier for NixHclClassifier {
_ => None,
}
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// Nix control flow
"if_expression" => Some(positional_candidate(node, "if", source)),
"let_expression" => Some(positional_candidate(node, "block", source)),
_ => None,
}
}
}
+178
View File
@@ -9,10 +9,188 @@ pub struct PythonClassifier;
impl LangClassifier for PythonClassifier {
fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Imports ──
"import_statement" | "import_from_statement" => {
Some(group_candidate(node, "imports", source))
},
// ── Variables / assignments ──
"assignment" => Some(group_candidate(node, "decls", source)),
// ── Functions ──
"function_definition" => Some(make_named_chunk(
node,
prefixed_name("fn", node, source),
source,
recurse_into(node, ChunkContext::FunctionBody, &["body"], &["block"]),
)),
// ── Containers ──
"class_definition" => Some(make_container_chunk(
node,
prefixed_name("class", node, source),
source,
recurse_into(node, ChunkContext::ClassBody, &["body"], &["block"]),
)),
// ── Control flow (top-level scripts) ──
"if_statement" | "for_statement" | "while_statement" | "try_statement"
| "with_statement" => Some(classify_function_python(node, source)),
// ── Statements ──
"expression_statement" | "global_statement" => {
Some(group_candidate(node, "stmts", source))
},
// ── Decorated ──
"decorated_definition" => Some(classify_decorated(node, source)),
_ => None,
}
}
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Methods ──
"function_definition" => {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
let chunk_name = if name == "__init__" || name == "__new__" {
"constructor".to_string()
} else {
format!("fn_{name}")
};
Some(make_named_chunk(
node,
chunk_name,
source,
recurse_into(node, ChunkContext::FunctionBody, &["body"], &["block"]),
))
},
// ── Decorated methods ──
"decorated_definition" => {
let inner = named_children(node)
.into_iter()
.find(|c| c.kind() == "function_definition");
if let Some(child) = inner {
let name =
extract_identifier(child, source).unwrap_or_else(|| "anonymous".to_string());
let chunk_name = if name == "__init__" || name == "__new__" {
"constructor".to_string()
} else {
format!("fn_{name}")
};
Some(make_named_chunk(
node,
chunk_name,
source,
recurse_into(child, ChunkContext::FunctionBody, &["body"], &["block"]),
))
} else {
Some(infer_named_candidate(node, source))
}
},
// ── Fields ──
"expression_statement" | "assignment" => Some(group_candidate(node, "fields", source)),
// ── Type aliases ──
"type_alias_statement" => Some(named_candidate(node, "type", source, None)),
_ => None,
}
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// ── Control flow ──
"if_statement" => Some(make_candidate(
node,
"if".to_string(),
NameStyle::Named,
None,
recurse_body(node, ChunkContext::FunctionBody),
false,
source,
)),
"for_statement" | "while_statement" => Some(make_candidate(
node,
"loop".to_string(),
NameStyle::Named,
None,
recurse_body(node, ChunkContext::FunctionBody),
false,
source,
)),
"try_statement" => Some(make_candidate(
node,
"try".to_string(),
NameStyle::Named,
None,
recurse_body(node, ChunkContext::FunctionBody),
false,
source,
)),
"with_statement" => Some(make_candidate(
node,
"block".to_string(),
NameStyle::Named,
None,
recurse_body(node, ChunkContext::FunctionBody),
false,
source,
)),
// ── Positional ──
"elif_clause" => Some(positional_candidate(node, "elif", source)),
"except_clause" => Some(positional_candidate(node, "except", source)),
"match_statement" => Some(positional_candidate(node, "match", source)),
// ── Variables ──
"expression_statement" | "assignment" => Some(group_candidate(node, "stmts", source)),
_ => None,
}
}
}
/// Classify Python function-level nodes (reused for top-level control flow
/// delegation).
fn classify_function_python<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
match node.kind() {
"if_statement" => {
make_candidate(node, "if".to_string(), NameStyle::Named, None, fn_recurse(), false, source)
},
"for_statement" | "while_statement" => make_candidate(
node,
"loop".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"try_statement" => make_candidate(
node,
"try".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"with_statement" => make_candidate(
node,
"block".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
_ => group_candidate(node, "stmts", source),
}
}
fn classify_decorated<'t>(node: Node<'t>, source: &str) -> RawChunkCandidate<'t> {
+129 -7
View File
@@ -1,12 +1,134 @@
//! Language-specific chunk classifiers for Ruby and Lua.
//!
//! Both languages are well-served by the default classification logic:
//! - Ruby: `module`, `class`, `method`/`singleton_method` all match default
//! patterns
//! - Lua: `function_declaration` matches the default function pattern
use super::classify::LangClassifier;
use tree_sitter::Node;
use super::{classify::LangClassifier, common::*};
pub struct RubyLuaClassifier;
impl LangClassifier for RubyLuaClassifier {}
impl LangClassifier for RubyLuaClassifier {
fn preserve_root_wrapper(&self, kind: &str) -> bool {
kind == "module"
}
fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
// ── Imports ──
"command" | "call" => {
// Ruby `require`/`require_relative` appear as call/command nodes.
// Check if the target is an import keyword.
let target = extract_identifier(node, source);
match target.as_deref() {
Some("require" | "require_relative" | "load" | "autoload") => {
group_candidate(node, "imports", source)
},
_ => group_candidate(node, "stmts", source),
}
},
// ── Functions ──
"function_definition" => {
named_candidate(node, "fn", source, recurse_body(node, ChunkContext::FunctionBody))
},
"method" | "singleton_method" => {
named_candidate(node, "fn", source, recurse_body(node, ChunkContext::FunctionBody))
},
// ── Containers ──
"class" => container_candidate(node, "class", source, recurse_class(node)),
"module" => container_candidate(node, "mod", source, recurse_class(node)),
// ── Control flow (top-level scripts) ──
"if_statement" | "unless" | "while_statement" | "for_statement" => {
return Some(
self
.classify_function(node, source)
.unwrap_or_else(|| group_candidate(node, "stmts", source)),
);
},
// ── Assignments ──
"assignment" => group_candidate(node, "decls", source),
// ── Statements ──
"expression_statement" | "function_call" => group_candidate(node, "stmts", source),
_ => return None,
})
}
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
// ── Methods ──
"method" | "singleton_method" => {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
if name == "initialize" {
make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
)
} else {
make_named_chunk(
node,
format!("fn_{name}"),
source,
recurse_body(node, ChunkContext::FunctionBody),
)
}
},
// ── Nested containers ──
"class" => container_candidate(node, "class", source, recurse_class(node)),
"module" => container_candidate(node, "mod", source, recurse_class(node)),
// ── Fields / constants ──
"assignment" => group_candidate(node, "fields", source),
// ── Calls (include, attr_reader, etc.) and bare identifiers (private) ──
"call" | "command" | "identifier" => group_candidate(node, "stmts", source),
_ => return None,
})
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
Some(match node.kind() {
// ── Control flow ──
"if_statement" | "unless" => make_candidate(
node,
"if".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"case_statement" | "case_match" => make_candidate(
node,
"switch".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"while_statement" | "for_statement" => make_candidate(
node,
"loop".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
// ── Variables ──
"assignment" => group_candidate(node, "stmts", source),
_ => return None,
})
}
}
+135 -5
View File
@@ -8,18 +8,148 @@ pub struct RustClassifier;
impl LangClassifier for RustClassifier {
fn classify_root<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
Some(match node.kind() {
// ── Imports ──
"use_declaration" | "extern_crate_declaration" => group_candidate(node, "imports", source),
// ── Functions ──
"function_item" | "function_definition" => named_candidate(
node,
"fn",
source,
recurse_body(node, ChunkContext::FunctionBody)
.or_else(|| recurse_into(node, ChunkContext::FunctionBody, &["body"], &["block"])),
),
// ── Containers ──
"struct_item" => container_candidate(node, "struct", source, recurse_class(node)),
"enum_item" => container_candidate(node, "enum", source, recurse_enum(node)),
"trait_item" => container_candidate(node, "trait", source, recurse_class(node)),
"mod_item" | "foreign_block" => {
container_candidate(node, "mod", source, recurse_class(node))
},
"impl_item" => {
let name = extract_impl_name(node, source).unwrap_or_else(|| "anonymous".to_string());
Some(make_container_chunk(
make_container_chunk(
node,
format!("impl_{name}"),
source,
recurse_into(node, ChunkContext::ClassBody, &["body"], &["declaration_list"]),
))
)
},
_ => None,
}
// ── Types ──
"type_item" => named_candidate(node, "type", source, recurse_class(node)),
// ── Macros ──
"macro_definition" | "macro_rule" => {
named_candidate(node, "macro", source, recurse_body(node, ChunkContext::FunctionBody))
},
// ── Statics / consts ──
"static_item" | "const_item" => group_candidate(node, "decls", source),
// ── Attributes ──
"inner_attribute_item" => group_candidate(node, "attrs", source),
// ── Variables ──
"let_declaration" => match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("var_{name}"), source, None),
None => group_candidate(node, "decls", source),
},
// ── Expression statements ──
"expression_statement" => group_candidate(node, "stmts", source),
_ => return None,
})
}
fn classify_class<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
Some(match node.kind() {
// ── Methods ──
"function_item" | "function_definition" => {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
make_named_chunk(
node,
format!("fn_{name}"),
source,
recurse_body(node, ChunkContext::FunctionBody),
)
},
// ── Types ──
"type_item" | "type_alias" => named_candidate(node, "type", source, None),
// ── Fields ──
"field_declaration" => match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("field_{name}"), source, None),
None => group_candidate(node, "fields", source),
},
// ── Enum variants ──
"enum_variant" => match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("variant_{name}"), source, None),
None => group_candidate(node, "variants", source),
},
// ── Consts / macros in class body ──
"const_item" | "macro_invocation" => group_candidate(node, "fields", source),
// ── Attributes (absorbed by the framework, but handle explicitly) ──
"attribute_item" => return None, // absorbed by is_absorbable_attr
_ => return None,
})
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
Some(match node.kind() {
// ── Control flow ──
"if_expression" => make_candidate(
node,
"if".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"match_expression" => positional_candidate(node, "match", source),
"loop_expression" | "while_expression" | "for_expression" => {
positional_candidate(node, "loop", source)
},
// ── Blocks ──
"unsafe_block" | "async_block" | "const_block" | "block_expression" => make_candidate(
node,
"block".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
// ── Variables ──
"let_declaration" => {
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
if span > 1 {
match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("var_{name}"), source, None),
None => group_candidate(node, "let", source),
}
} else {
group_candidate(node, "let", source)
}
},
// ── Expression statements ──
"expression_statement" => group_candidate(node, "stmts", source),
_ => return None,
})
}
}
+14 -1
View File
@@ -12,7 +12,7 @@ use super::{
common::{
ChunkContext, RawChunkCandidate, RecurseSpec, child_by_kind, extract_identifier,
group_candidate, make_container_chunk, make_container_chunk_from, make_named_chunk,
recurse_self, sanitize_identifier,
positional_candidate, recurse_self, sanitize_identifier,
},
types::ChunkNode,
};
@@ -78,6 +78,19 @@ impl LangClassifier for TlaplusClassifier {
}
}
fn classify_function<'t>(&self, node: Node<'t>, source: &str) -> Option<RawChunkCandidate<'t>> {
match node.kind() {
// PlusCal control flow
"pcal_if" => Some(positional_candidate(node, "if", source)),
"pcal_while" => Some(positional_candidate(node, "loop", source)),
"pcal_either" => Some(positional_candidate(node, "either", source)),
"pcal_with" => Some(positional_candidate(node, "with", source)),
// PlusCal assignments
"pcal_assign" => Some(group_candidate(node, "stmts", source)),
_ => None,
}
}
fn preserve_trivia(&self, kind: &str) -> bool {
kind == "block_comment"
}
+11 -352
View File
@@ -1,377 +1,36 @@
//! Default (shared) classification logic.
//!
//! These functions handle node kinds that are common across many languages.
//! Per-language classifiers are tried first; these defaults are the fallback.
//! These are minimal catch-all fallbacks for node kinds not handled by any
//! per-language classifier. The real classification lives in the `ast_*`
//! modules; these defaults only fire for truly unrecognized node kinds.
use tree_sitter::Node;
use super::common::*;
// ── Root-level defaults ──────────────────────────────────────────────────
// ── Root-level default ──────────────────────────────────────────────────
pub fn classify_root_default<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
match node.kind() {
// ── Imports / package headers ──
"import_statement"
| "import_from_statement"
| "use_declaration"
| "import_declaration"
| "using_directive"
| "namespace_use_declaration"
| "module_import"
| "include_directive"
| "import_list"
| "import_header"
| "package_header"
| "package_declaration"
| "using_statement"
| "namespace_statement"
| "preproc_include"
| "extern_crate_declaration"
| "yaml_directive"
| "tag_directive"
| "reserved_directive" => group_candidate(node, "imports", source),
// ── Variables / assignments ──
"lexical_declaration" | "variable_declaration" => classify_var_decl(node, source),
"const_declaration" | "var_declaration" | "short_var_declaration" => {
match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("var_{name}"), source, None),
None => group_candidate(node, "decls", source),
}
},
"assignment"
| "assignment_statement"
| "property_declaration"
| "state_variable_declaration" => group_candidate(node, "decls", source),
"expression_statement" | "global_statement" | "command" | "pipeline" | "function_call" => {
group_candidate(node, "stmts", source)
},
// ── Control flow (top-level scripts) ──
"if_statement"
| "unless"
| "guard_statement"
| "switch_statement"
| "switch_expression"
| "case_statement"
| "expression_switch_statement"
| "type_switch_statement"
| "select_statement"
| "try_statement"
| "try_block"
| "for_statement"
| "for_in_statement"
| "for_of_statement"
| "foreach_statement"
| "while_statement"
| "do_statement"
| "with_statement" => classify_function_default(node, source),
// ── Containers / namespaces / modules ──
"class_declaration"
| "class_definition"
| "class_specifier"
| "class_interface"
| "class_implementation" => container_candidate(node, "class", source, recurse_class(node)),
"interface_declaration" | "protocol_declaration" | "trait_definition" => {
container_candidate(node, "iface", source, recurse_interface(node))
},
"namespace_declaration"
| "file_scoped_namespace_declaration"
| "namespace_definition"
| "namespace_definition_name"
| "module_definition"
| "module"
| "mod_item"
| "package_clause"
| "object_definition"
| "extension_definition"
| "foreign_block" => container_candidate(node, "mod", source, recurse_class(node)),
"struct_item" | "struct_specifier" | "struct_declaration" | "record_declaration"
| "object_declaration" => container_candidate(node, "struct", source, recurse_class(node)),
"enum_declaration" | "enum_item" | "enum_specifier" | "enum_definition" => {
container_candidate(node, "enum", source, recurse_enum(node))
},
"trait_item" | "class" | "deftype" | "defrecord" => {
container_candidate(node, "trait", source, recurse_class(node))
},
"contract_declaration" | "library_declaration" | "trait_declaration" => {
container_candidate(node, "contract", source, recurse_class(node))
},
// ── Functions / methods / macros ──
"function_declaration"
| "function_definition"
| "function_item"
| "procedure_declaration"
| "overloaded_procedure_declaration"
| "function_definition_header"
| "test_declaration" => named_candidate(
node,
"fn",
source,
recurse_body(node, ChunkContext::FunctionBody)
.or_else(|| recurse_body(node, ChunkContext::FunctionBody))
.or_else(|| {
let context = ChunkContext::FunctionBody;
recurse_into(node, context, &["body"], &["block"])
}),
),
"method_declaration" => {
named_candidate(node, "meth", source, recurse_body(node, ChunkContext::FunctionBody))
},
"constructor_definition"
| "constructor_declaration"
| "secondary_constructor"
| "init_declaration"
| "fallback_receive_definition" => make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
),
"macro_definition" | "macro_rule" | "modifier_definition" => {
named_candidate(node, "macro", source, recurse_body(node, ChunkContext::FunctionBody))
},
// ── Types / aliases ──
"type_alias_declaration"
| "type_item"
| "type_alias"
| "user_defined_type_definition"
| "const_type_declaration"
| "opaque_declaration" => named_candidate(node, "type", source, recurse_class(node)),
// ── Systems extras ──
"static_item" => group_candidate(node, "decls", source),
"union_declaration" => container_candidate(node, "union", source, recurse_class(node)),
"covergroup_declaration" | "checker_declaration" => {
container_candidate(node, "group", source, recurse_class(node))
},
"module_declaration" => container_candidate(node, "mod", source, recurse_class(node)),
"inner_attribute_item" => group_candidate(node, "attrs", source),
_ => infer_named_candidate(node, source),
}
infer_named_candidate(node, source)
}
// ── Class-level defaults ─────────────────────────────────────────────────
// ── Class-level default ─────────────────────────────────────────────────
pub fn classify_class_default<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
match node.kind() {
"constructor" | "constructor_declaration" | "secondary_constructor" | "init_declaration" => {
make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
)
},
"method_definition"
| "method_signature"
| "abstract_method_signature"
| "method_declaration"
| "function_declaration"
| "function_definition"
| "function_item"
| "procedure_declaration"
| "protocol_function_declaration"
| "method"
| "singleton_method" => {
let name = extract_identifier(node, source).unwrap_or_else(|| "anonymous".to_string());
if name == "constructor" {
make_named_chunk(
node,
"constructor".to_string(),
source,
recurse_body(node, ChunkContext::FunctionBody),
)
} else {
make_named_chunk(
node,
format!("fn_{name}"),
source,
recurse_body(node, ChunkContext::FunctionBody),
)
}
},
"public_field_definition"
| "field_definition"
| "property_definition"
| "property_signature"
| "property_declaration"
| "protocol_property_declaration"
| "abstract_class_field"
| "const_declaration"
| "constant_declaration"
| "event_field_declaration" => match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("field_{name}"), source, None),
None => group_candidate(node, "fields", source),
},
"enum_assignment"
| "enum_member_declaration"
| "enum_constant"
| "enum_entry"
| "enum_variant" => match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("variant_{name}"), source, None),
None => group_candidate(node, "variants", source),
},
"field_declaration" | "embedded_field" | "container_field" | "binding" => {
match extract_identifier(node, source) {
Some(name) => make_named_chunk(node, format!("field_{name}"), source, None),
None => group_candidate(node, "fields", source),
}
},
"method_spec" => named_candidate(node, "meth", source, None),
"field_declaration_list" => group_candidate(node, "fields", source),
"method_spec_list" => group_candidate(node, "methods", source),
"class_static_block" => make_named_chunk(node, "static_init".to_string(), source, None),
"decorated_definition" => {
let inner = named_children(node)
.into_iter()
.find(|c| c.kind() == "function_definition");
if let Some(child) = inner {
let name = extract_identifier(child, source).unwrap_or_else(|| "anonymous".to_string());
make_named_chunk(node, format!("fn_{name}"), source, {
let context = ChunkContext::FunctionBody;
recurse_into(child, context, &["body"], &["block"])
})
} else {
infer_named_candidate(node, source)
}
},
"assignment"
| "expression_statement"
| "attribute"
| "pair"
| "block_mapping_pair"
| "flow_pair" => group_candidate(node, "fields", source),
"type_item" | "type_alias_declaration" | "type_alias" => {
named_candidate(node, "type", source, None)
},
"const_item" => group_candidate(node, "fields", source),
"macro_invocation" => group_candidate(node, "fields", source),
_ => infer_named_candidate(node, source),
}
infer_named_candidate(node, source)
}
// ── Function-level defaults ──────────────────────────────────────────────
// ── Function-level default ──────────────────────────────────────────────
pub fn classify_function_default<'tree>(
node: Node<'tree>,
source: &str,
) -> RawChunkCandidate<'tree> {
let fn_recurse = || recurse_body(node, ChunkContext::FunctionBody);
match node.kind() {
"if_statement" | "unless" | "guard_statement" => {
make_candidate(node, "if".to_string(), NameStyle::Named, None, fn_recurse(), false, source)
},
"switch_statement"
| "switch_expression"
| "case_statement"
| "case_match"
| "expression_switch_statement"
| "type_switch_statement"
| "select_statement"
| "receive_statement"
| "yul_switch_statement" => make_candidate(
node,
"switch".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"try_statement" | "try_block" | "catch_clause" | "finally_clause" | "assembly_statement" => {
make_candidate(
node,
"try".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
)
},
"for_statement" | "for_in_statement" | "for_of_statement" => {
let name = if looks_like_python_statement(node, source) {
"loop".to_string()
} else {
sanitize_node_kind(node.kind())
};
make_candidate(node, name, NameStyle::Named, None, fn_recurse(), false, source)
},
"while_statement" => {
let name = if looks_like_python_statement(node, source) {
"loop"
} else {
"while"
};
make_candidate(node, name.to_string(), NameStyle::Named, None, fn_recurse(), false, source)
},
"do_statement" | "with_statement" | "do_block" | "subshell" | "async_block"
| "unsafe_block" | "const_block" | "block_expression" => make_candidate(
node,
"block".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"foreach_statement" => make_candidate(
node,
"for".to_string(),
NameStyle::Named,
None,
fn_recurse(),
false,
source,
),
"defer_statement" | "go_statement" | "send_statement" => {
group_candidate(node, "stmts", source)
},
"elif_clause" => positional_candidate(node, "elif", source),
"except_clause" => positional_candidate(node, "except", source),
"when_statement" => positional_candidate(node, "when", source),
"match_expression" | "match_block" => positional_candidate(node, "match", source),
"loop_expression"
| "while_expression"
| "for_expression"
| "errdefer_statement"
| "comptime_statement"
| "nosuspend_statement"
| "suspend_statement"
| "yul_if_statement"
| "yul_for_statement" => positional_candidate(node, "loop", source),
"lexical_declaration"
| "variable_declaration"
| "const_declaration"
| "var_declaration"
| "short_var_declaration"
| "let_declaration" => {
let span = line_span(node.start_position().row + 1, node.end_position().row + 1);
if span > 1 {
if let Some(name) = extract_single_declarator_name(node, source) {
make_named_chunk(node, format!("var_{name}"), source, None)
} else {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
}
} else {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
}
},
_ => {
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
},
}
let kind_name = sanitize_node_kind(node.kind());
group_candidate(node, &kind_name, source)
}
// ── Variable declaration classification (shared) ─────────────────────────
// ── Variable declaration classification (shared) ────────────────────────
pub fn classify_var_decl<'tree>(node: Node<'tree>, source: &str) -> RawChunkCandidate<'tree> {
if let Some(candidate) = promote_assigned_expression(node, node, source) {
+85
View File
@@ -1733,6 +1733,91 @@ struct Config {
);
}
#[test]
fn ruby_class_methods_chunked() {
let source = r#"module PaymentProcessing
class Money
include Comparable
attr_reader :amount, :currency
def initialize(amount, currency = :usd)
@amount = amount
@currency = currency
end
def self.zero(currency = :usd)
new(0, currency)
end
def to_s
"$#{amount}"
end
private
def validate!
raise "Invalid" if amount < 0
end
end
end
"#;
let tree = build_chunk_tree(source, "ruby").expect("tree should build");
assert_eq!(tree.root_children, vec!["mod_PaymentProcessing"]);
let module = tree
.chunks
.iter()
.find(|c| c.path == "mod_PaymentProcessing")
.expect("mod_PaymentProcessing");
assert_eq!(module.kind, "branch");
assert!(
module
.children
.iter()
.any(|c| c == "mod_PaymentProcessing.class_Money"),
"expected class_Money inside module, got {:?}",
module.children
);
let class = tree
.chunks
.iter()
.find(|c| c.path == "mod_PaymentProcessing.class_Money")
.expect("class_Money");
assert_eq!(class.kind, "branch");
assert!(
class
.children
.iter()
.any(|c| c == "mod_PaymentProcessing.class_Money.constructor"),
"expected constructor in class children: {:?}",
class.children
);
assert!(
class
.children
.iter()
.any(|c| c == "mod_PaymentProcessing.class_Money.fn_zero"),
"expected fn_zero in class children: {:?}",
class.children
);
assert!(
class
.children
.iter()
.any(|c| c == "mod_PaymentProcessing.class_Money.fn_to_s"),
"expected fn_to_s in class children: {:?}",
class.children
);
assert!(
class
.children
.iter()
.any(|c| c == "mod_PaymentProcessing.class_Money.fn_validate"),
"expected fn_validate in class children: {:?}",
class.children
);
}
#[test]
fn keeps_mixed_enum_children_addressable() {
let source = r"enum Message {
+8 -20
View File
@@ -8,12 +8,12 @@ use crate::{
type ChunkLookup<'a> = HashMap<&'a str, &'a ChunkNode>;
env_uint! {
// Configured full display threshold.
static FULL_DISPLAY_THRESHOLD: usize = "PI_CHUNK_FULL_DISPLAY_THRESHOLD" or 80 => [1, usize::MAX];
// Configured preview head lines.
static PREVIEW_HEAD_LINES: usize = "PI_CHUNK_PREVIEW_HEAD_LINES" or 20 => [1, usize::MAX];
// Configured preview tail lines.
static PREVIEW_TAIL_LINES: usize = "PI_CHUNK_PREVIEW_TAIL_LINES" or 8 => [1, usize::MAX];
// Configured full display threshold.
static FULL_DISPLAY_THRESHOLD: usize = "PI_CHUNK_FULL_DISPLAY_THRESHOLD" or 40 => [1, usize::MAX];
// Configured preview head lines.
static PREVIEW_HEAD_LINES: usize = "PI_CHUNK_PREVIEW_HEAD_LINES" or 10 => [1, usize::MAX];
// Configured preview tail lines.
static PREVIEW_TAIL_LINES: usize = "PI_CHUNK_PREVIEW_TAIL_LINES" or 5 => [1, usize::MAX];
}
pub fn line_to_containing_chunk_path(tree: &ChunkTree, line: u32) -> Option<String> {
@@ -330,13 +330,12 @@ fn format_header_meta(
omit_checksum: bool,
) -> String {
let language = language_tag.unwrap_or("text");
let line_label = if line_count == 1 { "line" } else { "lines" };
let checksum_part = if omit_checksum {
String::new()
} else {
format!(" · #{checksum}")
};
format!("{title} · {line_count} {line_label} · {language}{checksum_part}")
format!("{title} · {line_count}ln · {language}{checksum_part}")
}
fn should_render_gap_line(
@@ -591,18 +590,7 @@ fn emit_leaf_body(ctx: &mut RenderCtx<'_>, _chunk: &ChunkNode, span: VisibleSpan
match entry {
LeafEntry::Line { abs_line, text } => push_code(ctx, abs_line, &text),
LeafEntry::Ellipsis { count, start_abs, end_abs } => {
push_meta(
ctx,
format!(
"⋮ {} {} ({}–{}) — sel=L{}-L{} to expand ⋮",
count,
if count == 1 { "line" } else { "lines" },
start_abs,
end_abs,
start_abs,
end_abs,
),
);
push_meta(ctx, format!("sel=L{start_abs}-L{end_abs} to expand ({count} lines)"));
},
}
}
+10
View File
@@ -14,6 +14,15 @@
### Changed
- Renamed chunk edit line-range parameters from `beg`/`end` to `line`/`end_line` for consistency with documentation
- Unified `splice` and `replace` operations into a single `replace` operation with optional `line`/`end_line` parameters for line-scoped edits
- Updated `replace` operation to support optional `line` and `end_line` parameters, with `end_line` defaulting to `line` when omitted for single-line replacements
- Removed `splice` operation from chunk edit schema; use `replace` with `line`/`end_line` parameters instead
- Simplified chunk edit prompt documentation to consolidate operation guidance and remove splice-specific examples
- Renamed chunk edit line-range parameters from `beg`/`end` to `line`/`end_line` for clarity and consistency with documentation
- Updated `replace` operation to support optional `line` and `end_line` parameters for line-scoped edits, with `end_line` defaulting to `line` when omitted for single-line replacements
- Simplified chunk edit prompt documentation to consolidate `replace` and `splice` operations into a single `replace` operation with optional line-range parameters
- Added backward compatibility mapping to accept legacy `beg`/`end` and `end_line` field names in chunk edit operations
- Changed chunk edit operation format from flat `{ "op": "splice", ... }` to keyed `{ "splice": { ... } }` format for improved clarity and schema validation
- Updated chunk edit schema to use discriminated union types with operation-specific field requirements (insert ops no longer require CRC, mutation ops require CRC)
- Improved chunk selector sanitization to strip filename prefixes and normalize checksums to uppercase, handling model output variations
@@ -33,6 +42,7 @@
### Removed
- Removed `PI_CHUNK_SPLICES` environment variable and `chunkSplicesEnabled()` function; splice operations are no longer conditionally available
- Autoresearch segment fingerprint hashing and `segmentFingerprint` on experiment state / `autoresearch.jsonl` config lines
### Fixed
+75 -173
View File
@@ -30,7 +30,7 @@ import patchDescription from "../prompts/tools/patch.md" with { type: "text" };
import replaceDescription from "../prompts/tools/replace.md" with { type: "text" };
import type { ToolSession } from "../tools";
import { checkAutoGeneratedFile, checkAutoGeneratedFileContent } from "../tools/auto-generated-guard";
import { applyChunkEdits, type ChunkEditOperation, chunkSplicesEnabled, parseChunkReadPath } from "../tools/chunk-tree";
import { applyChunkEdits, type ChunkEditOperation, parseChunkReadPath } from "../tools/chunk-tree";
import {
invalidateFsScanAfterDelete,
invalidateFsScanAfterRename,
@@ -104,66 +104,40 @@ const patchEditSchema = Type.Object({
diff: Type.Optional(Type.String({ description: "Diff hunks (update) or full content (create)" })),
});
const chunkEditSelProp = { sel: Type.Optional(Type.String({ description: "Chunk path selector" })) } as const;
const chunkEditCrcProp = { crc: Type.String({ description: "Chunk checksum for staleness validation" }) } as const;
const chunkEditContentProp = {
content: Type.Union([Type.String(), Type.Array(Type.String())], { description: "Inserted or replacement content" }),
const pSel = { sel: Type.Optional(Type.String({ description: "Chunk path selector" })) } as const;
const pCrc = { crc: Type.String({ description: "Chunk checksum for staleness validation" }) } as const;
const pContent = {
content: Type.Union([Type.String(), Type.Array(Type.String())], { description: "Replacement content" }),
} as const;
const chunkEditOptionalContentProp = {
content: Type.Optional(
Type.Union([Type.String(), Type.Array(Type.String())], {
description: "Replacement content (empty string to delete lines)",
}),
),
} as const;
const chunkEditBegEndProps = {
beg: Type.Integer({
description:
"Start line: absolute file line (1-indexed, same as read gutter). With end, inclusive when beg ≤ end; when beg = end + 1, zero-width insertion.",
}),
end: Type.Integer({
description:
"End line: absolute file line (1-indexed). Inclusive when beg ≤ end; for zero-width insertion use end = beg − 1.",
}),
} as const;
const chunkEditOptionalBegEndProps = {
beg: Type.Optional(
const pRangesOpt = {
line: Type.Optional(
Type.Integer({
description:
"Start line for line-scoped replace. When both beg and end are provided, replaces only those file lines (same as splice).",
"Start line for line-scoped replace (1-indexed file line from read gutter). Omit for whole-chunk replace.",
}),
),
end: Type.Optional(Type.Integer({ description: "End line for line-scoped replace." })),
end_line: Type.Optional(
Type.Integer({ description: "End line for line-scoped replace. Omit for single-line edit." }),
),
} as const;
// Insert ops (no CRC)
const appendChildOp = Type.Object({ append_child: Type.Object({ ...chunkEditSelProp, ...chunkEditContentProp }) });
const prependChildOp = Type.Object({ prepend_child: Type.Object({ ...chunkEditSelProp, ...chunkEditContentProp }) });
const appendSiblingOp = Type.Object({ append_sibling: Type.Object({ ...chunkEditSelProp, ...chunkEditContentProp }) });
const prependSiblingOp = Type.Object({
prepend_sibling: Type.Object({ ...chunkEditSelProp, ...chunkEditContentProp }),
});
const pInsertDetails = Type.Object({ ...pSel, ...pContent });
const appendChildOp = Type.Object({ append_child: pInsertDetails });
const prependChildOp = Type.Object({ prepend_child: pInsertDetails });
const appendSiblingOp = Type.Object({ append_sibling: pInsertDetails });
const prependSiblingOp = Type.Object({ prepend_sibling: pInsertDetails });
// Mutation ops (CRC required)
const pReplaceDetails = Type.Object({ ...pSel, ...pCrc, ...pContent, ...pRangesOpt });
const replaceOp = Type.Object({
replace: Type.Object({
...chunkEditSelProp,
...chunkEditCrcProp,
...chunkEditContentProp,
...chunkEditOptionalBegEndProps,
}),
});
const deleteOp = Type.Object({ delete: Type.Object({ ...chunkEditSelProp, ...chunkEditCrcProp }) });
const spliceOp = Type.Object({
splice: Type.Object({
...chunkEditSelProp,
...chunkEditCrcProp,
...chunkEditBegEndProps,
...chunkEditOptionalContentProp,
}),
replace: pReplaceDetails,
});
const chunkEditOperationSchemaWithoutSplice = Type.Union([
const pDeleteDetails = Type.Object({ ...pSel, ...pCrc });
const deleteOp = Type.Object({ delete: pDeleteDetails });
const chunkEditOperationSchema = Type.Union([
appendChildOp,
prependChildOp,
appendSiblingOp,
@@ -172,25 +146,11 @@ const chunkEditOperationSchemaWithoutSplice = Type.Union([
deleteOp,
]);
const chunkEditOperationSchemaWithSplice = Type.Union([
appendChildOp,
prependChildOp,
appendSiblingOp,
prependSiblingOp,
replaceOp,
deleteOp,
spliceOp,
]);
const chunkEditOperationSchema = chunkSplicesEnabled()
? chunkEditOperationSchemaWithSplice
: chunkEditOperationSchemaWithoutSplice;
const chunkEditParamsSchema = Type.Object({
path: Type.String({ description: "File path; may include :chunk_path to set a default selector" }),
crc: Type.Optional(Type.String({ description: "Default checksum for the path-level selector" })),
path: Type.String({ description: "File path; may include :chunk_path for default selector" }),
crc: Type.Optional(Type.String({ description: "Default checksum" })),
operations: Type.Array(chunkEditOperationSchema, {
description: "Chunk edit operations applied in order",
description: "Edit operations",
minItems: 1,
}),
});
@@ -208,7 +168,7 @@ export function hashlineParseText(edit: string[] | string | null): string[] {
return stripNewLinePrefixes(edit);
}
function normalizeChunkContentInput(content: string | string[] | undefined): string {
function flattenContent(content: string | string[] | undefined): string {
if (content === undefined) return "";
if (Array.isArray(content)) return content.join("\n");
return content;
@@ -421,44 +381,6 @@ function isChunkParams(params: ReplaceParams | PatchParams | HashlineParams | Ch
return "operations" in params;
}
const CHUNK_OP_KEYS = [
"append_child",
"prepend_child",
"append_sibling",
"prepend_sibling",
"replace",
"delete",
"splice",
] as const;
type ChunkOpKey = (typeof CHUNK_OP_KEYS)[number];
interface ParsedChunkOp {
op: ChunkOpKey;
sel?: string;
crc?: string;
content?: string | string[];
beg?: number;
end?: number;
}
/** Normalize keyed `{ splice: { ... } }` format to flat internal format. Also tolerates legacy `{ op: "splice", ... }`. */
function parseKeyedChunkOp(raw: Record<string, unknown>): ParsedChunkOp {
// Legacy flat format: { op: "splice", sel: "...", ... }
if (typeof raw.op === "string" && CHUNK_OP_KEYS.includes(raw.op as ChunkOpKey)) {
return raw as unknown as ParsedChunkOp;
}
// Keyed format: { splice: { sel: "...", ... } }
for (const key of CHUNK_OP_KEYS) {
if (key in raw && typeof raw[key] === "object" && raw[key] !== null) {
const inner = raw[key] as Record<string, unknown>;
return { op: key, ...inner } as unknown as ParsedChunkOp;
}
}
throw new Error(`Unknown chunk edit operation shape: ${JSON.stringify(Object.keys(raw))}`);
}
/**
* Edit tool implementation.
*
@@ -541,7 +463,7 @@ export class EditTool implements AgentTool<TInput> {
get description(): string {
switch (this.mode) {
case "chunk":
return renderPromptTemplate(chunkEditDescription, { chunkSplices: chunkSplicesEnabled() });
return renderPromptTemplate(chunkEditDescription);
case "patch":
return renderPromptTemplate(patchDescription);
case "hashline":
@@ -611,79 +533,59 @@ export class EditTool implements AgentTool<TInput> {
? parseChunkTree(normalizeToLF(stripBom(rawContent).text), getLanguageFromPath(resolvedPath) ?? "")
: undefined;
const normalizedOperations: ChunkEditOperation[] = operations.map(operation => {
const opEntry = parseKeyedChunkOp(operation as Record<string, unknown>);
const anchorSelector = opEntry.sel ?? defaultSelector;
const anchorCrc = opEntry.crc ?? (opEntry.sel === undefined ? defaultCrc : undefined);
const content = normalizeChunkContentInput(opEntry.content);
const normalizedOperations: ChunkEditOperation[] = [];
const requireChecksum = (op: string): void => {
if (anchorCrc) return;
if (anchorSelector && chunkTree) {
const resolved = resolveChunkPath(chunkTree, anchorSelector);
if (resolved) {
throw new Error(
`Checksum required for ${op} on "${anchorSelector}". ` +
`Re-read the chunk to get its checksum, then pass crc: "${resolved.checksum}" in the operation.`,
);
}
const assertChecksum = (op: string, crc: string | undefined, sel: string | undefined): string => {
if (crc) return crc.toUpperCase();
if (sel && chunkTree) {
const resolved = resolveChunkPath(chunkTree, sel);
if (resolved) {
throw new Error(
`Chunk not found: "${anchorSelector}". Re-read the file to see available chunk paths.`,
`Checksum required for ${op} on "${sel}". ` +
`Re-read the chunk to get its checksum, then pass crc: "${resolved.checksum}" in the operation.`,
);
}
throw new Error(
`Checksum required for ${op} on sel "${anchorSelector ?? ""}". ` +
`Re-read the file first, then provide crc: "XXXX" from the chunk header shown in the read output.`,
);
};
switch (opEntry.op) {
case "append_child":
case "prepend_child":
case "append_sibling":
case "prepend_sibling":
if (!content) {
throw new Error(`Content required for ${opEntry.op} on ${anchorSelector ?? "<root>"}`);
}
return { op: opEntry.op, sel: opEntry.sel, crc: opEntry.crc, content } as ChunkEditOperation;
case "replace":
requireChecksum("replace");
if (content.length === 0) {
return { op: "delete", sel: opEntry.sel, crc: anchorCrc } as ChunkEditOperation;
}
return {
op: "replace",
sel: opEntry.sel,
crc: anchorCrc,
content,
beg: opEntry.beg,
end: opEntry.end,
} as ChunkEditOperation;
case "delete":
requireChecksum("delete");
return { op: "delete", sel: opEntry.sel, crc: anchorCrc } as ChunkEditOperation;
case "splice":
if (!chunkSplicesEnabled()) {
throw new Error(
"Chunk splice operations are disabled (PI_CHUNK_SPLICES=0). Use replace, delete, or sibling/child insert operations instead.",
);
}
requireChecksum("splice");
if (opEntry.beg == null || opEntry.end == null) {
throw new Error(`beg and end required for splice on ${anchorSelector}`);
}
return {
op: "splice",
sel: opEntry.sel,
crc: anchorCrc,
beg: opEntry.beg,
end: opEntry.end,
content,
} as ChunkEditOperation;
default:
throw new Error(`Unknown chunk edit operation: ${opEntry.op}`);
throw new Error(`Chunk not found: "${sel}". Re-read the file to see available chunk paths.`);
}
});
throw new Error(
`Checksum required for ${op} on sel "${sel ?? ""}". ` +
`Re-read the file first, then provide crc: "XXXX" from the chunk header shown in the read output.`,
);
};
for (const operation of operations) {
for (const [op, details] of Object.entries(operation)) {
switch (op) {
case "append_child":
case "prepend_child":
case "append_sibling":
case "prepend_sibling": {
const { content, sel } = details as unknown as Static<typeof pInsertDetails>;
if (!content) {
throw new Error(`Content required for ${op} on ${sel ?? "<root>"}`);
}
normalizedOperations.push({ op, sel, content: flattenContent(content) });
break;
}
case "replace": {
const { content, sel, crc, line, end_line } = details as unknown as Static<typeof pReplaceDetails>;
normalizedOperations.push({
op: "replace",
sel,
crc: assertChecksum(op, crc, sel),
content: flattenContent(content),
line,
endLine: end_line,
});
break;
}
case "delete": {
const { sel, crc } = details as unknown as Static<typeof pDeleteDetails>;
normalizedOperations.push({ op: "delete", sel, crc: assertChecksum(op, crc, sel) });
break;
}
}
}
}
const chunkResult = applyChunkEdits({
source: rawContent,
@@ -1,80 +1,42 @@
Edits files by addressing syntax-aware chunks from `read` output.
Read the file first with `read(path="file.ts")`. Use chunk paths exactly as shown by the latest `read` result. {{#if chunkSplices}}For `replace`, `delete`, and `splice`, supply{{else}}For `replace` and `delete`, supply{{/if}} the chunk checksum separately as `crc`.
Read the file first with `read(path="file.ts")`. Use chunk paths exactly as shown by the latest `read` result. For `replace` and `delete`, supply the chunk checksum separately as `crc`.
Successful edit responses already include the updated root chunk rendering for that file. Do not immediately re-read the file just to refresh checksums unless the file may have changed externally.
Successful edit responses include the updated chunk tree with checksums. Do not re-read just to refresh checksums unless the file changed externally.
**Checksum scope:** Each chunk has its own CRC over that chunk's source span. Editing non-overlapping lines elsewhere in the file does not change unrelated chunks' checksums. There is no separate whole-file guard.
**Checksum scope:** Each chunk has its own CRC over its source span. Editing non-overlapping lines elsewhere does not change unrelated chunks' checksums.
<operations>
{{#if chunkSplices}}
**Choosing the right operation:**
- To fix a single line → `splice` with `beg`=`end`=that line number
- To fix a contiguous range of lines → `splice` with `beg`=first line, `end`=last line
- To rewrite an entire function/method from scratch → `replace` (without `beg`/`end`)
- `splice` is almost always the right choice for small, targeted fixes
- `replace` with `beg`/`end` also works for line-level fixes (same behavior as `splice`)
{{/if}}
- To fix a single line → `replace` with just `line`=that line number
- To fix a contiguous range → `replace` with `line`=first line, `end_line`=last line
- To rewrite an entire function/method → `replace` without `line`/`end_line`
- To insert new code → `replace` with `line`=insertion line, `end_line`=`line`-1 (zero-width: inserts without removing)
|operation|format|effect|
|---|---|---|
|`append_child`|`{ "append_child": { "sel": "…", "content": "…" } }`|insert as last child inside a container (before closing delimiter)|
|`prepend_child`|`{ "prepend_child": { "sel": "…", "content": "…" } }`|insert as first child inside a container (after opening delimiter)|
|`append_sibling`|`{ "append_sibling": { "sel": "…", "content": "…" } }`|insert after the selected chunk|
|`prepend_sibling`|`{ "prepend_sibling": { "sel": "…", "content": "…" } }`|insert before the selected chunk|
|`replace`|`{ "replace": { "sel": "…", "crc": "…", "content": "…" } }`|without `beg`/`end`: rewrite the **entire** chunk from first line to last line; supply full replacement content. With `beg`/`end`: replace only those **file lines** within the chunk (same as `splice` but semantically a replacement). `splice` is almost always the right choice for small, targeted fixes|
|`delete`|`{ "delete": { "sel": "…", "crc": "…" } }`|remove the entire chunk|
{{#if chunkSplices}}
|`splice`|`{ "splice": { "sel": "…", "crc": "…", "beg": N, "end": N } }`|when `beg` ≤ `end`, replace **file** lines `beg`–`end` (inclusive) within the selected chunk; empty `content` deletes them. When `beg` = `end` + 1, **zero-width splice**: insert in the gap after file line `end` and before file line `beg` (no lines removed)|
{{/if}}
- `sel` is the chunk path
- `crc` is the chunk checksum used for staleness validation on `replace`, `delete`{{#if chunkSplices}}, and `splice`{{/if}}; it guards the selected chunk, not unrelated file changes elsewhere in the file
{{#if chunkSplices}}
- `beg`/`end` are **absolute file line numbers** (same as the gutter in `read`). They must fall within the selected chunk's file line span (or form a valid zero-width gap on its boundary).
{{/if}}
- `path="file.ts:chunk_path"` sets a default `sel` for operations that omit it
- top-level `crc` sets the default checksum for the path-level selector
- **Auto-indent (flat):** The tool adds the target insertion column's base indent to every **non-empty** line of your `content`. It does not parse or understand block nesting; put any deeper relative indentation in the `content` yourself (extra leading spaces/tabs on inner lines).
- Chunk paths are always fully qualified: `class_Server.fn_start`, not bare `fn_start`
- Batch ops observe earlier edits in the same request. If op 1 changes a chunk's checksum, span, or path, op 2 must use the **post-op-1** checksum/span/path.
- `replace`/`delete` operate on the selected chunk range, including leading comments or attributes that the parser attached to that chunk.
|`replace`|`{ "replace": { "sel": "…", "crc": "…", "content": "…" } }`|without `line`/`end_line`: rewrite entire chunk. With `line`: replace that line. With `line`+`end_line`: replace that range|
|`delete`|`{ "delete": { "sel": "…", "crc": "…" } }`|remove entire chunk|
|`append_child`|`{ "append_child": { "sel": "…", "content": "…" } }`|insert as last child|
|`prepend_child`|`{ "prepend_child": { "sel": "…", "content": "…" } }`|insert as first child|
|`append_sibling`|`{ "append_sibling": { "sel": "…", "content": "…" } }`|insert after chunk|
|`prepend_sibling`|`{ "prepend_sibling": { "sel": "…", "content": "…" } }`|insert before chunk|
- `line`/`end_line` are **absolute file line numbers** from `read` gutter. `line` alone = single line. `line` + `end_line` = inclusive range. `line` with `end_line` = `line`-1 = zero-width insert.
- `path="file.ts:chunk_path"` sets default `sel`; top-level `crc` sets default checksum
- **Auto-indent:** Tool adds base indent to non-empty lines. Put deeper indentation in `content` yourself.
- Chunk paths are fully qualified: `class_Server.fn_start`, not bare `fn_start`
- Batch ops observe earlier edits. If op 1 changes checksum/span/path, op 2 must use post-op-1 values.
- `replace`/`delete` include leading comments/attributes attached to the chunk.
</operations>
{{#if chunkSplices}}
<splice>
**Zero-width `splice` (insert only, `beg` = `end` + 1), using file line numbers `S` = chunk's first line, `E` = chunk's last line:**
Use zero-width splice for separator gaps between neighboring chunks. Gap edits are independent from either chunk's checksum: inserting into the gap does not update the following chunk's `crc`, and later `replace`/`delete` operations only affect that chunk's own selected range.
|Goal|`beg`|`end`|
|---|---|---|
|Insert before file line L|L|L − 1|
|Insert after file line L|L + 1|L|
|Insert before the chunk's first line S|S|S − 1|
|Insert after the chunk's last line E|E + 1|E|
</splice>
{{/if}}
<examples>
All examples reference this `read` output (`beg`/`end` match the gutter):
All examples reference this `read` output:
```
│ server.ts · 40 lines · ts · #VSKB
│ server.ts · 40L · ts · #VSKB
│
1 │ import { log, warn } from "./logger";
│ <:imports#SPVY>
3 │ const MAX_RETRIES = 3;
│ <:var_MAX_RETRIES#ZHZM>
5 │ class Server {
│ <:class_Server#XKQZ>
6 │ private port: number;
│ <.fields#NMYB>
8 │ constructor(port: number) {
│ <.constructor#BNNH>
9 │ this.port = port;
10 │ }
12 │ start(): void {
│ <.fn_start#HTST>
@@ -106,17 +68,15 @@ All examples reference this `read` output (`beg`/`end` match the gutter):
```
</example>
{{#if chunkSplices}}
<example name="splice a line subrange">
<example name="replace a single line">
```
"path": "server.ts",
"operations": [
{
"splice": {
"replace": {
"sel": "class_Server.fn_start",
"crc": "HTST",
"beg": 13,
"end": 13,
"line": 13,
"content": " warn(\"booting on \" + this.port);"
}
}
@@ -124,23 +84,6 @@ All examples reference this `read` output (`beg`/`end` match the gutter):
```
</example>
<example name="insert a line inside a method (zero-width splice)">
```
"path": "server.ts",
"operations": [
{
"splice": {
"sel": "class_Server.fn_start",
"crc": "HTST",
"beg": 13,
"end": 12,
"content": "const startedAt = Date.now();"
}
}
]
```
</example>
{{/if}}
<example name="delete a chunk">
```
"path": "server.ts",
@@ -154,66 +97,16 @@ All examples reference this `read` output (`beg`/`end` match the gutter):
]
```
</example>
{{#if chunkSplices}}
<example name="batch multiple edits">
```
"path": "server.ts",
"operations": [
{
"splice": {
"sel": "class_Server.fn_start",
"crc": "HTST",
"beg": 13,
"end": 13,
"content": " warn(\"booting on \" + this.port);"
}
},
{
"delete": {
"sel": "class_Server.fn_tryBind",
"crc": "VNWR"
}
}
]
```
</example>
{{else}}
<example name="batch multiple edits">
```
"path": "server.ts",
"operations": [
{
"replace": {
"sel": "class_Server.fn_start",
"crc": "HTST",
"content": "start(): void {\n warn(\"booting on \" + this.port);\n for (let i = 0; i < MAX_RETRIES; i++) {\n this.tryBind();\n }\n}"
}
},
{
"delete": {
"sel": "class_Server.fn_tryBind",
"crc": "VNWR"
}
}
]
```
</example>
{{/if}}
</examples>
<critical>
- You **MUST** always include `path` in every edit call.
- You **MUST** read the latest chunk output before editing.
- You **MUST** provide `crc` for `replace`, `delete`{{#if chunkSplices}}, and `splice`{{/if}}.
- You **MUST** use the updated root chunk output from the edit response for follow-up edits in the same file when possible.
- You **MUST** use the smallest correct chunk; do not rewrite siblings unnecessarily.
- You **MUST NOT** invent chunk paths. Read the current listing first and copy the actual path names, including `fn_*` prefixes as shown. Chunk nesting uses `.`.
{{#if chunkSplices}}
- For `splice`, copy `beg`/`end` from the **file line numbers** in `read` output (the same values as elision `sel=L…` ranges).
- **Do NOT batch multiple `splice` operations on the same chunk** in one edit call. Each splice changes the chunk's checksum, and you cannot predict the new checksum. Instead: either combine the changes into a single splice with a wider `beg`/`end` range covering all affected lines, or make separate edit calls (one splice per call).
{{/if}}
- When restoring a deleted statement, check surrounding lines carefully — insert at the exact position relative to existing code. Do not duplicate adjacent lines.
- Use only chunk paths that appear in the `read` output. If you need deeper chunks, `read` the parent chunk first to discover its children.
- **MUST** include `path` in every edit call.
- **MUST** read latest chunk output before editing.
- **MUST** provide `crc` for `replace` and `delete`.
- **MUST** use updated chunk output from edit response for follow-up edits.
- **MUST** use smallest correct chunk; do not rewrite siblings unnecessarily.
- **MUST NOT** invent chunk paths. Copy from `read` output including `fn_*` prefixes. Nesting uses `.`.
- For line-scoped `replace`, use file line numbers from `read` gutter.
- Do NOT batch multiple line-scoped `replace` operations on the same chunk. Combine into a wider `line`/`end_line` range or use separate calls.
</critical>
</output>
@@ -13,25 +13,21 @@ Reads files using syntax-aware chunks.
|_(omitted)_|Render the file root chunk|
|`class_Foo`|Read a chunk by path|
|`class_Foo.fn_bar`|Read a nested chunk path|
|`L50` or `L50-L120`|Absolute **file** line range: show chunks that overlap those lines, or that contain a child under those lines (e.g. Go receiver methods under a type whose header is outside the range)|
|`L50` or `L50-L120`|Absolute file line range|
|`raw`|Read full raw file content (no chunk rendering)|
Each anchor line shows `<:name#CCCC>` or `<.name#CCCC>` — `#CCCC` is the edit checksum. Copy it when editing with `chunk-edit`.
Selectors can overlap. For example, Rust `attribute_N` chunks and the following `enum_` / `struct_` chunk may cover the same leading doc-comment or attribute lines. Use the attribute selector when you only want to edit the attached comments/attributes; use the type selector when you want to replace the declaration together with those attached lines.
If `path:chunk` and `sel` are both provided, `sel` wins. Missing chunk paths return `[Chunk not found]`.
If a `path:chunk` suffix and `sel` are both provided, `sel` wins unless `path` carries the chunk selector and `sel` is a line range (`L<n>` or `L<n>-L<m>`). In that case `L…` is still **absolute file lines**, clipped to that chunk; if the range does not overlap the chunk, the tool reports the chunk’s file line span and a suggested `sel=`. Missing chunk paths return `[Chunk not found]`.
Code rows use **absolute file line numbers** in the gutter. `chunk-edit` `line`/`end_line` use those same numbers.
The header `N lines` reports the actual file line count for a file-root read. For nested chunk selectors, it reports the lines this selector currently renders, not a raw parser field. That count can be larger than the chunk’s lexical header when the renderer groups related descendants under the parent (for example Go receiver methods shown beneath their receiver type).
Code rows use **absolute file line numbers** in the gutter. Middle elisions use `sel=L<start>-L<end>` with the same absolute indices. `chunk-edit` **splice** `beg`/`end` use those same **absolute file line numbers** (see `chunk-edit` tool docs).
Rendered gap lines are visual context, not checksum ownership. If you need to edit the separator between two chunks, use zero-width `splice` on the adjacent chunk boundary instead of `replace` on either neighboring chunk.
## Examples
`read(path="src/math.ts")`
```text
│ src/math.ts · 120 lines · ts · #A744
│ src/math.ts · 120ln · ts · #A744
│
5 │ export function sum(values: readonly number[]): number {
@@ -48,47 +44,14 @@ Rendered gap lines are visual context, not checksum ownership. If you need to ed
14 │ }
```
`read(path="src/math.ts", sel="class_Calculator")`
```text
│ src/math.ts:class_Calculator · 5 lines · ts · #5D36
│
10 │ export class Calculator {
│ <:Calculator#5D36>
11 │ multiply(left: number, right: number): number {
│ <.multiply#B592>
12 │ return left * right;
13 │ }
14 │ }
```
`read(path="src/math.ts", sel="L7-L12")`
```text
[Notice: chunk view scoped to requested lines L7-L12; non-overlapping lines omitted.]
│ src/math.ts · 120 lines · ts · #A744
│
```
`read(path="src/math.ts:class_Calculator.fn_square", sel="L11-L12")`
```text
│ src/math.ts:class_Calculator.fn_square · 3 lines · ts · #C9A8
│
11 │ square(value: number): number {
12 │ return this.multiply(value, value);
```
## Language Support
Chunk trees: JavaScript, TypeScript, TSX, Python, Rust, Go. Others use blank-line fallback.
</instruction>
<critical>
- You **MUST** use `read` instead of shell commands for file reading.
- You **MUST** copy the current checksum before editing a chunk with `chunk-edit`.
- You **MUST** not assume chunk names; always read the current output first.
- **MUST** use `read` instead of shell commands for file reading.
- **MUST** copy the current checksum before editing a chunk with `chunk-edit`.
- **MUST NOT** assume chunk names; always read the current output first.
</critical>
</output>
+64 -137
View File
@@ -81,9 +81,8 @@ export type ChunkEditOperation =
| { op: "prepend_child"; sel?: string; crc?: string; content: string }
| { op: "append_sibling"; sel?: string; crc?: string; content: string }
| { op: "prepend_sibling"; sel?: string; crc?: string; content: string }
| { op: "replace"; sel?: string; crc?: string; content: string; beg?: number; end?: number }
| { op: "delete"; sel?: string; crc?: string }
| { op: "splice"; sel?: string; crc?: string; beg: number; end: number; content: string };
| { op: "replace"; sel?: string; crc?: string; content: string; line?: number; endLine?: number }
| { op: "delete"; sel?: string; crc?: string };
export type ChunkEditResult = {
diffSourceBefore: string;
@@ -114,17 +113,6 @@ type ParsedChunkReadPath = {
// Helpers
// ---------------------------------------------------------------------------
/**
* `PI_CHUNK_SPLICES` — when `0`, chunk edit tool omits the `splice` operation from the schema
* and prompts. Default `1` (splices enabled). Other values throw at startup when the schema loads.
*/
export function chunkSplicesEnabled(): boolean {
const v = Bun.env.PI_CHUNK_SPLICES;
if (v === undefined || v === "" || v === "1") return true;
if (v === "0") return false;
throw new Error(`Invalid PI_CHUNK_SPLICES: expected "0" or "1" (default: 1), got ${JSON.stringify(v)}`);
}
/**
* `PI_CHUNI_VALIDATE` — when `0`, chunk edit tool skips the validation of the edited
*/
@@ -795,98 +783,64 @@ function validateBatchCrc(params: { chunk: ChunkNode; crc: string | undefined; r
validateCrc(chunk, crc);
}
function validateSpliceRange(anchor: ChunkNode, beg: number, end: number): void {
/**
* Validate that line/endLine fall within the anchor chunk's span.
* Supports three modes:
* - line <= endLine: replace lines line–endLine (inclusive)
* - line = endLine + 1: zero-width insert between endLine and line
* - single-line (endLine === line): replace just that line
*/
function validateLineRange(anchor: ChunkNode, line: number, endLine: number): void {
const cStart = anchor.startLine;
const cEnd = anchor.endLine;
const chunkName = operation_name(anchor);
if (beg < 1) {
if (line < 1) {
throw new Error(
`Splice beg ${beg} is invalid for ${chunkName}; beg and end are absolute file line numbers (1-indexed). This chunk spans file lines ${cStart}-${cEnd}.`,
`Line ${line} is invalid for ${chunkName}; line and end_line are absolute file line numbers (1-indexed). This chunk spans file lines ${cStart}-${cEnd}.`,
);
}
if (end < 0) {
throw new Error(`Invalid splice range L${beg}-L${end} for ${chunkName}: end cannot be negative.`);
if (endLine < 0) {
throw new Error(`Invalid line range L${line}-L${endLine} for ${chunkName}: end_line cannot be negative.`);
}
if (beg > end + 1) {
if (line > endLine + 1) {
throw new Error(
`Invalid splice range L${beg}-L${end} for ${chunkName}: use beg ≤ end to replace lines, or beg = end + 1 for a zero-width insertion in the gap after file line end (before file line beg).`,
`Invalid line range L${line}-L${endLine} for ${chunkName}: use line \u2264 end_line to replace lines, or line = end_line + 1 for zero-width insertion.`,
);
}
if (beg <= end) {
if (beg < cStart || end > cEnd) {
if (line <= endLine) {
if (line < cStart || endLine > cEnd) {
throw new Error(
`Splice range L${beg}-L${end} is outside ${chunkName} (chunk spans file lines ${cStart}-${cEnd}). Use absolute line numbers from read output.`,
`Line range L${line}-L${endLine} is outside ${chunkName} (chunk spans file lines ${cStart}-${cEnd}). Use absolute line numbers from read output.`,
);
}
return;
}
// beg === end + 1: zero-width insertion between file line `end` and file line `beg`
const beforeChunk = end === cStart - 1 && beg === cStart;
const insideGap = cStart <= end && end < cEnd && beg === end + 1;
const afterChunk = end === cEnd && beg === cEnd + 1;
// line === endLine + 1: zero-width insertion
const beforeChunk = endLine === cStart - 1 && line === cStart;
const insideGap = cStart <= endLine && endLine < cEnd && line === endLine + 1;
const afterChunk = endLine === cEnd && line === cEnd + 1;
if (beforeChunk || insideGap || afterChunk) {
return;
}
throw new Error(
`Invalid zero-width splice L${beg}-L${end} for ${chunkName} (chunk spans file lines ${cStart}-${cEnd}). ` +
`Use end = ${cStart - 1}, beg = ${cStart} to insert before the first chunk line; ` +
`end = k, beg = k + 1 with ${cStart} ≤ k < ${cEnd} between interior lines; ` +
`end = ${cEnd}, beg = ${cEnd + 1} to insert after the last chunk line.`,
`Invalid zero-width insert L${line}-L${endLine} for ${chunkName} (chunk spans file lines ${cStart}-${cEnd}). ` +
`Use end_line = ${cStart - 1}, line = ${cStart} to insert before the first chunk line; ` +
`end_line = k, line = k + 1 with ${cStart} \u2264 k < ${cEnd} between interior lines; ` +
`end_line = ${cEnd}, line = ${cEnd + 1} to insert after the last chunk line.`,
);
}
function isZeroWidthSplice(beg: number, end: number): boolean {
return beg === end + 1;
function isZeroWidthInsert(line: number, endLine: number): boolean {
return line === endLine + 1;
}
/** Absolute file line `absEnd` is the line after which the gap starts; before-chunk uses absEnd === cStart − 1. */
function zeroWidthSpliceSortKey(anchor: ChunkNode, absEnd: number, absBeg: number): number {
/** Sort key for zero-width inserts: higher endLine runs first (bottom-up) so line numbers stay stable. */
function zeroWidthInsertSortKey(anchor: ChunkNode, endLine: number, line: number): number {
const cStart = anchor.startLine;
if (absEnd === cStart - 1 && absBeg === cStart) {
if (endLine === cStart - 1 && line === cStart) {
return cStart;
}
return absEnd;
}
function zeroWidthInsertionOffset(state: MutableChunkState, anchor: ChunkNode, absEnd: number): number {
const offsets = lineOffsets(state.source);
if (absEnd === anchor.startLine - 1) {
return lineStartOffset(offsets, anchor.startLine, state.source);
}
return lineEndOffset(offsets, absEnd, state.source);
}
function zeroWidthInsertionIndent(state: MutableChunkState, anchor: ChunkNode, absBeg: number, absEnd: number): string {
const offsets = lineOffsets(state.source);
const sliceFileLine = (absLine: number): string => {
if (absLine < 1 || absLine > state.tree.lineCount) {
return "";
}
return state.source.slice(
lineStartOffset(offsets, absLine, state.source),
lineEndOffset(offsets, absLine, state.source),
);
};
if (absEnd === anchor.startLine - 1) {
return detectCommonIndent(sliceFileLine(anchor.startLine)).prefix;
}
const iEnd = detectCommonIndent(sliceFileLine(absEnd)).prefix;
if (absBeg > state.tree.lineCount) {
return iEnd;
}
const iBeg = detectCommonIndent(sliceFileLine(absBeg)).prefix;
return iEnd.length >= iBeg.length ? iEnd : iBeg;
}
function normalizeZeroWidthInsertionContent(state: MutableChunkState, offset: number, content: string): string {
const trimmed = content.replace(/^\n+/, "").replace(/\n+$/, "");
if (trimmed.length === 0) return content;
const prevChar = offset > 0 ? state.source[offset - 1] : "";
const nextChar = offset < state.source.length ? state.source[offset] : "";
const prefix = prevChar !== "" && prevChar !== "\n" ? "\n" : "";
const suffix = nextChar !== "" && !trimmed.endsWith("\n") ? "\n" : "";
return `${prefix}${trimmed}${suffix}`;
return endLine;
}
type ScheduledChunkEditOperation = {
@@ -1290,6 +1244,13 @@ function getInsertionPointForPosition(
}
}
/** True when the replace op targets specific lines (not a whole-chunk replace). */
function isLineScoped(
op: ChunkEditOperation,
): op is { op: "replace"; sel?: string; crc?: string; line: number; endLine?: number; content: string } {
return op.op === "replace" && op.line != null;
}
export function applyChunkEdits(params: {
source: string;
language?: string;
@@ -1318,48 +1279,50 @@ export function applyChunkEdits(params: {
for (const scheduled of scheduledOps) {
const operation = scheduled.operation;
if (operation.op === "splice") {
if (isLineScoped(operation)) {
const anchor = scheduled.initialChunk;
if (!anchor) {
throw new Error(`Chunk tree is missing an anchor for ${describeScheduledOperation(scheduled)}`);
}
validateSpliceRange(anchor, operation.beg, operation.end);
const absEnd = operation.endLine ?? operation.line;
validateLineRange(anchor, operation.line, absEnd);
}
}
const spliceSortKey = (scheduled: ScheduledChunkEditOperation): number => {
const lineScopedSortKey = (scheduled: ScheduledChunkEditOperation): number => {
const operation = scheduled.operation;
if (operation.op !== "splice") {
if (!isLineScoped(operation)) {
return 0;
}
const anchor = scheduled.initialChunk;
if (!anchor) {
return 0;
}
if (isZeroWidthSplice(operation.beg, operation.end)) {
return zeroWidthSpliceSortKey(anchor, operation.end, operation.beg);
const absEnd = operation.endLine ?? operation.line;
if (isZeroWidthInsert(operation.line, absEnd)) {
return zeroWidthInsertSortKey(anchor, absEnd, operation.line);
}
return operation.beg;
return operation.line;
};
const executionOps: ScheduledChunkEditOperation[] = [];
for (let index = 0; index < scheduledOps.length; index++) {
const scheduled = scheduledOps[index]!;
if (scheduled.operation.op !== "splice") {
if (!isLineScoped(scheduled.operation)) {
executionOps.push(scheduled);
continue;
}
const spliceBlock: ScheduledChunkEditOperation[] = [scheduled];
while (index + 1 < scheduledOps.length && scheduledOps[index + 1]!.operation.op === "splice") {
const lineScopedBlock: ScheduledChunkEditOperation[] = [scheduled];
while (index + 1 < scheduledOps.length && isLineScoped(scheduledOps[index + 1]!.operation)) {
index++;
spliceBlock.push(scheduledOps[index]!);
lineScopedBlock.push(scheduledOps[index]!);
}
spliceBlock.sort((a, b) => {
const lineDiff = spliceSortKey(b) - spliceSortKey(a);
lineScopedBlock.sort((a, b) => {
const lineDiff = lineScopedSortKey(b) - lineScopedSortKey(a);
if (lineDiff !== 0) return lineDiff;
return a.originalIndex - b.originalIndex;
});
executionOps.push(...spliceBlock);
executionOps.push(...lineScopedBlock);
}
const currentDefaultSelector = initialDefaultSelector;
@@ -1388,12 +1351,13 @@ export function applyChunkEdits(params: {
const { chunk: anchor, crc: resolvedCrc } = resolveAnchorWithCrc(state, anchorSelector, crc, warnings);
validateBatchCrc({ chunk: anchor, crc: resolvedCrc, required: requiresChecksum });
// When beg/end are provided, replace only those lines (line-scoped replace).
if (operation.beg != null && operation.end != null) {
validateSpliceRange(anchor, operation.beg, operation.end);
// When line is provided, replace only those lines (line-scoped replace).
// endLine defaults to line when omitted (single-line edit).
if (operation.line != null) {
const absEnd = operation.endLine ?? operation.line;
validateLineRange(anchor, operation.line, absEnd);
const offsets = lineOffsets(state.source);
const absBeg = operation.beg;
const absEnd = operation.end;
const absBeg = operation.line;
const rangeStart = lineStartOffset(offsets, absBeg, state.source);
const rangeEnd = lineEndOffset(offsets, absEnd, state.source);
const replacedRange = state.source.slice(rangeStart, rangeEnd);
@@ -1411,7 +1375,7 @@ export function applyChunkEdits(params: {
break;
}
// Whole-chunk replace (no beg/end).
// Whole-chunk replace (no line/endLine).
const targetIndent = (anchor.indentChar || "\t").repeat(anchor.indent);
let replacement = normalizeInsertedContent(operation.content, targetIndent);
if (replacement.length > 0 && !replacement.endsWith("\n") && anchor.endLine < state.tree.lineCount) {
@@ -1492,7 +1456,7 @@ export function applyChunkEdits(params: {
});
if (commentOnly && anchor.path === "" && anchor.children.includes("file_preamble")) {
throw new Error(
"Comment-only prepend_child on root is not allowed when the file has a file_preamble chunk. Use replace or splice on the file_preamble chunk instead.",
"Comment-only prepend_child on root is not allowed when the file has a file_preamble chunk. Use replace on the file_preamble chunk instead.",
);
}
if (commentOnly && anchor.children.length > 0) {
@@ -1506,43 +1470,6 @@ export function applyChunkEdits(params: {
if (operation.sel === undefined) clearDefaultCrc();
break;
}
case "splice": {
const anchorSelector = operation.sel ?? currentDefaultSelector;
const crc = operation.crc ?? (operation.sel === undefined ? currentDefaultCrc : undefined);
const requiresChecksum = operation.sel !== undefined || Boolean(currentDefaultCrc);
ensureBatchOperationTargetCurrent(state, scheduled, crc, touchedPaths);
const { chunk: anchor, crc: resolvedCrc } = resolveAnchorWithCrc(state, anchorSelector, crc, warnings);
validateBatchCrc({ chunk: anchor, crc: resolvedCrc, required: requiresChecksum });
validateSpliceRange(anchor, operation.beg, operation.end);
const offsets = lineOffsets(state.source);
if (isZeroWidthSplice(operation.beg, operation.end)) {
const offset = zeroWidthInsertionOffset(state, anchor, operation.end);
const targetIndent = zeroWidthInsertionIndent(state, anchor, operation.beg, operation.end);
let replacement = normalizeInsertedContent(operation.content, targetIndent);
replacement = normalizeZeroWidthInsertionContent(state, offset, replacement);
state.source = insertAtOffset(state.source, offset, replacement);
touchedPaths.push(anchor.path);
if (operation.sel === undefined) clearDefaultCrc();
break;
}
const absBeg = operation.beg;
const absEnd = operation.end;
const rangeStart = lineStartOffset(offsets, absBeg, state.source);
const rangeEnd = lineEndOffset(offsets, absEnd, state.source);
const replacedRange = state.source.slice(rangeStart, rangeEnd);
const targetIndent = detectCommonIndent(replacedRange).prefix;
let replacement = normalizeInsertedContent(operation.content, targetIndent);
if (replacement.length > 0 && !replacement.endsWith("\n") && absEnd < state.tree.lineCount) {
replacement += "\n";
}
state.source = replaceRangeByLines(state.source, absBeg, absEnd, replacement);
if (replacement.length === 0) {
state.source = cleanupBlankLineArtifactsAtOffset(state.source, rangeStart);
}
touchedPaths.push(anchor.path);
if (operation.sel === undefined) clearDefaultCrc();
break;
}
}
state = rebuildChunkState(state.source, normalizedLanguage);
+2 -8
View File
@@ -37,13 +37,7 @@ import {
import { convertFileWithMarkit } from "../utils/markit";
import { detectSupportedImageMimeTypeFromFile } from "../utils/mime";
import { type ArchiveReader, openArchive, parseArchivePathCandidates } from "./archive-reader";
import {
type ChunkReadTarget,
chunkSplicesEnabled,
formatChunkedRead,
parseChunkReadPath,
parseChunkSelector,
} from "./chunk-tree";
import { type ChunkReadTarget, formatChunkedRead, parseChunkReadPath, parseChunkSelector } from "./chunk-tree";
import {
executeReadUrl,
isReadableUrlPath,
@@ -457,7 +451,7 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
this.#inspectImageEnabled = session.settings.get("inspect_image.enabled");
this.description =
resolveEditMode(session) === "chunk"
? renderPromptTemplate(readChunkDescription, { chunkSplices: chunkSplicesEnabled() })
? renderPromptTemplate(readChunkDescription)
: renderPromptTemplate(readDescription, {
DEFAULT_LIMIT: String(this.#defaultLimit),
DEFAULT_MAX_LINES: String(DEFAULT_MAX_LINES),
@@ -425,7 +425,7 @@ describe("edit safety invariants", () => {
};
}
for (const operation of ["replace", "delete", "splice"] as const) {
for (const operation of ["replace", "delete", "line-scoped replace"] as const) {
test(`rejects stale checksum for ${operation} with current and provided checksums in the error`, () => {
const { source, staleChecksum, currentChecksum } = buildStaleRunFixture();
@@ -449,11 +449,11 @@ describe("edit safety invariants", () => {
return edit(
[
{
op: "splice",
op: "replace",
sel: runChunkPath,
crc: staleChecksum,
beg: 7,
end: 7,
line: 7,
endLine: 7,
content: '\t\tconsole.log("again");',
},
],
@@ -511,19 +511,19 @@ describe("edit safety invariants", () => {
expect(() =>
edit([
{
op: "splice",
op: "replace",
sel: runChunkPath,
crc: checksum,
beg: 6,
end: 6,
line: 6,
endLine: 6,
content: '\trun(task = "default"): void {',
},
{
op: "splice",
op: "replace",
sel: runChunkPath,
crc: checksum,
beg: 7,
end: 7,
line: 7,
endLine: 7,
content: "\t\tconsole.log(task);",
},
]),
@@ -535,30 +535,30 @@ describe("edit safety invariants", () => {
// Batch splices are applied bottom-up by absolute file line (higher line first).
const afterFirst = edit([
{
op: "splice",
op: "replace",
sel: runChunkPath,
crc: checksum,
beg: 7,
end: 7,
line: 7,
endLine: 7,
content: "\t\tconsole.log(task);",
},
]).diffSourceAfter;
const checksum2 = getChecksum(afterFirst, runChunkPath);
const result = edit([
{
op: "splice",
op: "replace",
sel: runChunkPath,
crc: checksum,
beg: 7,
end: 7,
line: 7,
endLine: 7,
content: "\t\tconsole.log(task);",
},
{
op: "splice",
op: "replace",
sel: runChunkPath,
crc: checksum2,
beg: 6,
end: 6,
line: 6,
endLine: 6,
content: '\trun(task = "default"): void {',
},
]);
@@ -589,11 +589,11 @@ describe("edit safety invariants", () => {
const before = getChecksum(testSource, "class_Worker.constructor");
const after = edit([
{
op: "splice",
op: "replace",
sel: runChunkPath,
crc: getChecksum(testSource, runChunkPath),
beg: 7,
end: 7,
line: 7,
endLine: 7,
content: '\t\tconsole.log("nearby");',
},
]).diffSourceAfter;
@@ -714,7 +714,7 @@ describe("formatChunkedRead", () => {
language: "typescript",
});
expect(result.text).toMatch(new RegExp(`with-tail\\.ts · ${totalLines} lines`));
expect(result.text).toMatch(new RegExp(`with-tail\\.ts · ${totalLines}ln`));
});
test("leaf read shows absolute file lines and raw source indentation", async () => {
@@ -955,7 +955,7 @@ describe("grouped Go receiver chunk headers", () => {
language: "go",
});
expect(result.text).toContain("server.go:type_Server · 6 lines");
expect(result.text).toContain("server.go:type_Server · 6ln");
expect(result.text).toContain(".fn_Start#");
expect(result.text).toContain(".fn_Stop#");
});
@@ -1010,10 +1010,10 @@ describe("zero-width splice (line insertion)", () => {
const ac = { sel: "class_Worker.fn_run", crc: getChecksum(testSource, "class_Worker.fn_run") };
const result = edit([
{
op: "splice",
op: "replace",
...ac,
beg: 7,
end: 6,
line: 7,
endLine: 6,
content: "if (!this.name) return;",
},
]);
@@ -1025,10 +1025,10 @@ describe("zero-width splice (line insertion)", () => {
const ac = { sel: "class_Worker.fn_run", crc: getChecksum(testSource, "class_Worker.fn_run") };
const result = edit([
{
op: "splice",
op: "replace",
...ac,
beg: 8,
end: 7,
line: 8,
endLine: 7,
content: "trackRun();",
},
]);
@@ -1041,10 +1041,10 @@ describe("zero-width splice (line insertion)", () => {
expect(() =>
edit([
{
op: "splice",
op: "replace",
...ac,
beg: 20,
end: 19,
line: 20,
endLine: 19,
content: "noop();",
},
]),
@@ -1056,11 +1056,11 @@ describe("zero-width splice (line insertion)", () => {
const inserted = edit(
[
{
op: "splice",
op: "replace",
sel: "class_Worker.fn_run",
crc: checksum,
beg: 5,
end: 4,
line: 5,
endLine: 4,
content: "// inserted",
},
],
@@ -1126,11 +1126,11 @@ describe("blank-line cleanup", () => {
const result = edit(
[
{
op: "splice",
op: "replace",
sel: "class_Worker.fn_restart",
crc: checksum,
beg: 6,
end: 6,
line: 6,
endLine: 6,
content: "",
},
],
@@ -1187,10 +1187,10 @@ describe("splice", () => {
// fn_run spans file lines 6-8; line 7 is console.log
const result = edit([
{
op: "splice",
op: "replace",
...ac,
beg: 7,
end: 7,
line: 7,
endLine: 7,
content: '\t\tconsole.log("updated");',
},
]);
@@ -1209,11 +1209,11 @@ describe("splice", () => {
filePath: "/tmp/widget.rs",
operations: [
{
op: "splice",
op: "replace",
sel: "impl_Widget.fn_old",
crc: checksum,
beg: 3,
end: 4,
line: 3,
endLine: 4,
content: " false\n}",
},
],
@@ -1227,11 +1227,11 @@ describe("splice", () => {
expect(() =>
edit([
{
op: "splice",
op: "replace",
sel: "class_Worker.fn_run",
crc: "ZZZZ",
beg: 7,
end: 7,
line: 7,
endLine: 7,
content: "replacement",
},
]),
@@ -1243,10 +1243,10 @@ describe("splice", () => {
expect(() =>
edit([
{
op: "splice",
op: "replace",
...ac,
beg: 4,
end: 2,
line: 4,
endLine: 2,
content: "replacement",
},
]),
@@ -1258,10 +1258,10 @@ describe("splice", () => {
expect(() =>
edit([
{
op: "splice",
op: "replace",
...ac,
beg: 0,
end: 1,
line: 0,
endLine: 1,
content: "replacement",
},
]),
@@ -1274,10 +1274,10 @@ describe("splice", () => {
expect(() =>
edit([
{
op: "splice",
op: "replace",
...ac,
beg: 1,
end: 5,
line: 1,
endLine: 5,
content: "replacement",
},
]),
@@ -241,11 +241,11 @@ describe("chunk mode tools", () => {
filePath,
operations: [
{
op: "splice",
op: "replace",
sel: chunkPath,
crc: checksum,
beg: 63,
end: 63,
line: 63,
endLine: 63,
content: "return err.message.toUpperCase() + total;",
},
],
@@ -257,17 +257,17 @@ describe("chunk mode tools", () => {
crc: checksum,
operations: [
{
splice: {
beg: 63,
end: 63,
replace: {
line: 63,
end_line: 63,
content: "return err.message.toUpperCase() + total;",
},
},
{
splice: {
replace: {
crc: checksum2,
beg: 3,
end: 3,
line: 3,
end_line: 3,
content: "let total = 1;",
},
},
@@ -338,11 +338,11 @@ describe("chunk mode tools", () => {
filePath,
operations: [
{
op: "splice",
op: "replace",
sel: chunkPath,
crc: checksum,
beg: 4,
end: 3,
line: 4,
endLine: 3,
content: "const end = Date.now();",
},
],
@@ -354,20 +354,20 @@ describe("chunk mode tools", () => {
path: filePath,
operations: [
{
splice: {
replace: {
sel: chunkPath,
crc: checksum,
beg: 4,
end: 3,
line: 4,
end_line: 3,
content: "const end = Date.now();",
},
},
{
splice: {
replace: {
sel: chunkPath,
crc: checksum2,
beg: 3,
end: 2,
line: 3,
end_line: 2,
content: "const start = Date.now();",
},
},
@@ -434,11 +434,11 @@ describe("chunk mode tools", () => {
path: filePath,
operations: [
{
splice: {
replace: {
sel: "class_Server.fn_handleError",
crc: checksum,
beg: 5,
end: 2,
line: 5,
end_line: 2,
content: " let total = 1;",
},
},
@@ -525,18 +525,18 @@ describe("chunk mode tools", () => {
},
},
{
splice: {
replace: {
sel: "class_Server.fn_handleError",
crc: checksum,
beg: 63,
end: 63,
line: 63,
end_line: 63,
content: " return err.message.toUpperCase();",
},
},
],
}),
).rejects.toThrow(
/Edit operation 2\/2 failed \(splice on "class_Server\.fn_handleError"\): Chunk "class_Server\.fn_handleError" was changed by an earlier batch operation/,
/Edit operation 2\/2 failed \(replace on "class_Server\.fn_handleError"\): Chunk "class_Server\.fn_handleError" was changed by an earlier batch operation/,
);
expect(await Bun.file(filePath).text()).toBe(originalSource);
@@ -554,11 +554,11 @@ describe("chunk mode tools", () => {
path: filePath,
operations: [
{
splice: {
replace: {
sel: "class_Server.fn_handleError",
// no crc!
beg: 3,
end: 3,
line: 3,
end_line: 3,
content: " let total = 1;",
},
} as never,