feat: implemented tree-sitter AST queries and landing correction for blocks

- Added `NodeSpan` struct and `nodeChainAt` function in `pi-ast` with native bindings to query AST node chains by line.
- Replaced regex-based annotation parsing with tree-sitter node queries and addedconstruct relocation validation.
- Implemented opener-escape landing correction for inserts anchored on block openers along with warning messages.
- Added comprehensive unit and integration tests covering node chain resolution and escape behavior.
This commit is contained in:
can1357
2026-08-20 03:16:05 +02:00
parent 12591dbde3
commit 1088cc349c
9 changed files with 475 additions and 34 deletions
+146
View File
@@ -274,6 +274,72 @@ pub fn enclosing_block_boundaries(options: EnclosingBoundaryOptions) -> Result<O
Ok(Some(boundaries.into_iter().collect()))
}
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
pub struct NodeSpan {
/// 1-indexed inclusive first line of the node.
pub start_line: u32,
/// 1-indexed inclusive last content line of the node.
pub end_line: u32,
/// Tree-sitter grammar node kind (e.g. `attribute_item`, `function_item`).
pub kind: String,
}
/// Named-node chain containing `options.line`, innermost-first, excluding the
/// whole-file root.
///
/// The chain descends through the line's first content character, so
/// single-line nodes beginning on the line (attributes, decorators, one-line
/// statements) come first, followed by every enclosing construct up to — but
/// not including — the file root. Callers use it to classify a line by grammar
/// node kind and to enumerate enclosing construct end lines.
///
/// ERROR and MISSING recovery nodes are skipped rather than failing the whole
/// chain: an unrelated syntax error elsewhere in the file leaves healthy
/// ancestors' spans meaningful.
///
/// Returns `None` when the language is unrecognized, the line is out of
/// range / blank, or the source fails to parse entirely.
pub fn node_chain_at(options: BlockRangeOptions) -> Result<Option<Vec<NodeSpan>>> {
let BlockRangeOptions { code, lang, path, line } = options;
if line == 0 || code.is_empty() {
return Ok(None);
}
let Some(language) = resolve_language(lang.as_deref(), path.as_deref()) else {
return Ok(None);
};
let row = (line - 1) as usize;
let Some(col) = first_content_column(&code, row) else {
return Ok(None);
};
let Some(tree) = parse_cached(&code, language)? else {
return Ok(None);
};
let root = tree.root_node();
// One-column-wide range for the same zero-width-node reason as
// `block_range_at` above.
let point = Point::new(row, col);
let point_end = Point::new(row, col + 1);
let Some(leaf) = root.named_descendant_for_point_range(point, point_end) else {
return Ok(None);
};
let mut chain = Vec::new();
let mut node = Some(leaf);
while let Some(current) = node {
if current.id() == root.id() {
break;
}
if current.is_named() && !current.is_error() && !current.is_missing() {
chain.push(NodeSpan {
start_line: node_start_line(current),
end_line: node_content_end_line(current),
kind: current.kind().to_string(),
});
}
node = current.parent();
}
Ok(Some(chain))
}
#[cfg(test)]
mod tests {
use super::*;
@@ -288,6 +354,86 @@ mod tests {
.expect("block resolution succeeds")
}
fn chain(code: &str, path: &str, line: u32) -> Vec<NodeSpan> {
node_chain_at(BlockRangeOptions {
code: code.to_string(),
lang: None,
path: Some(path.to_string()),
line,
})
.expect("chain resolution succeeds")
.expect("language recognized and line non-blank")
}
const RUST_ANNOTATED: &str = "mod m {\n impl S {\n #[napi]\n fn f(&self) -> u32 \
{\n 1\n }\n }\n}\n";
/// A consumer classifying an attribute row must find a single-line
/// `attribute_item` in the chain at that line.
#[test]
fn chain_names_rust_attribute_row() {
let spans = chain(RUST_ANNOTATED, "x.rs", 3);
assert!(
spans
.iter()
.any(|s| s.kind == "attribute_item" && s.start_line == 3 && s.end_line == 3),
"{spans:?}"
);
}
/// A consumer relocating past enclosing constructs needs the chain at a
/// construct's opening line to carry every enclosing end line,
/// innermost-first. Wrapper nodes (`block`, `declaration_list`) share end
/// lines with their construct; consumers dedupe.
#[test]
fn chain_orders_enclosing_ends_innermost_first() {
let spans = chain(RUST_ANNOTATED, "x.rs", 4);
let mut ends: Vec<u32> = spans
.iter()
.filter(|s| s.end_line > 4)
.map(|s| s.end_line)
.collect();
ends.dedup();
// fn f ends on 6, impl on 7, mod on 8.
assert_eq!(ends, vec![6, 7, 8]);
}
/// TypeScript decorators are children of the declaration they precede; the
/// chain at the decorator line must still surface the single-line
/// `decorator` node.
#[test]
fn chain_names_ts_decorator_row() {
let code = "/** d */\n@Injectable()\nclass Service {}\n";
let spans = chain(code, "x.ts", 2);
assert!(
spans
.iter()
.any(|s| s.kind == "decorator" && s.start_line == 2 && s.end_line == 2),
"{spans:?}"
);
}
/// Blank lines carry no chain; unknown languages resolve to `None`.
#[test]
fn chain_declines_blank_lines_and_unknown_languages() {
let blank = node_chain_at(BlockRangeOptions {
code: "a\n\nb\n".to_string(),
lang: None,
path: Some("x.ts".to_string()),
line: 2,
})
.unwrap();
assert_eq!(blank, None);
let unknown = node_chain_at(BlockRangeOptions {
code: "a\n".to_string(),
lang: None,
path: Some("x.unknownext".to_string()),
line: 1,
})
.unwrap();
assert_eq!(unknown, None);
}
const TS_EXAMPLE: &str = "function x() {\n if (y) {\n }\n}\n";
#[test]
+5 -1
View File
@@ -2,9 +2,13 @@
## [Unreleased]
### Added
- Added an opener-escape landing correction: a plain `PUT >N:` anchored on a construct's opening line (per tree-sitter) with a body that parses as one self-contained construct claiming a strictly shallower column depth (tab bodies in space files included) is landed after the innermost enclosing construct whose own depth admits the body as a sibling, verified by the syntax probe. Previously such an insert silently split the opener from its body — and could still parse (items are legal inside Rust fn bodies), so no advisory fired.
### Fixed
- Dropped one-sided boundary echoes on single-line replacement ranges when every echoed row is an attribute/decorator (`#[napi]`, `@Injectable()`): the authored result parses, so the syntax probe never fired and the attribute was silently duplicated. Under-filled annotation echoes are now rejected instead of applied.
- Dropped one-sided boundary echoes on single-line replacement ranges when every echoed row is an attribute/decorator/annotation node per the tree-sitter grammar (`#[napi]`, `@Injectable()`): the authored result parses, so the syntax probe never fired and the attribute was silently duplicated. Under-filled annotation echoes are now rejected instead of applied.
- Raised the default snapshot-store path capacity from 30 to 256 so tags minted early in a wide session no longer age out of the LRU and degrade a recoverable stale-tag mismatch into the misleading "hash is not from this session" rejection.
## [17.3.3] - 2026-08-14
+148 -32
View File
@@ -12,6 +12,7 @@
import { resolveClipboardEdits } from "./clipboard";
import {
afterInsertLandingShiftWarning,
afterInsertOpenerEscapeWarning,
ambiguousBoundaryEchoMessage,
ambiguousBoundaryPlacementMessage,
blockInsertLandingShiftWarning,
@@ -22,7 +23,7 @@ import {
UNRESOLVED_BLOCK_INTERNAL,
UNRESOLVED_CLIPBOARD_INTERNAL,
} from "./messages";
import { enclosingBoundaries, parsesCleanly } from "./syntax";
import { enclosingBoundaries, nodeChain, parsesCleanly } from "./syntax";
import { cloneCursor } from "./tokenizer";
import type { Anchor, ApplyResult, Clipboard, Cursor, Edit } from "./types";
@@ -154,21 +155,46 @@ function bucketAnchorEditsByLine(edits: IndexedEdit[]): Map<number, IndexedEdit[
export const STRUCTURAL_CLOSER_RE = /^\s*[)\]}]+[;,]?\s*$/;
/**
* A row that is nothing but an attribute or decorator: `#[napi]`,
* `#![allow(dead_code)]`, `@Injectable()`, `@property`. Statements may repeat
* verbatim on adjacent lines by intent, but annotations on one item never do —
* an exact adjacent copy is a boundary echo. This is the extra evidence that
* lets one-sided echo normalization act on single-line ranges (the
* Grammar node kinds for attribute/decorator/annotation rows, per bundled
* tree-sitter grammar. Statements may repeat verbatim on adjacent lines by
* intent, but annotations on one item never do — an exact adjacent copy is a
* boundary echo. This classification is the extra evidence that lets
* one-sided echo normalization act on single-line ranges (the
* doc-restoration incident: a `PUT N.=N` landed one line high and its body
* restated the `#[napi]` surviving just below the range, duplicating the
* attribute — a result that parses, so no syntax probe could catch it).
* attribute — a result that PARSES, attributes being repeatable syntax, so
* neither the probe advisory nor the variant search could catch it).
*
* Kinds are grammar-blessed names, verified against the bundled parsers —
* not lexical guesses; `@media` in CSS or `@x` inside a string never
* classify. Extend per grammar as needed.
*/
const ANNOTATION_ROW_RE = /^\s*(?:#!?\[.+\]|@[A-Za-z_$][\w$.]*(?:\(.*\))?)\s*$/;
const ANNOTATION_NODE_KINDS: Record<string, true> = {
attribute_item: true, // rust `#[...]`
inner_attribute_item: true, // rust `#![...]`
decorator: true, // typescript/tsx/javascript/python
annotation: true, // kotlin, java `@Foo(...)`
marker_annotation: true, // java `@Override`
attribute_list: true, // c# `[Fact]`
};
/** Every payload row in `[start, end)` is an annotation row. */
function isAnnotationEchoRun(payload: readonly string[], start: number, end: number): boolean {
for (let i = start; i < end; i++) {
if (!ANNOTATION_ROW_RE.test(payload[i])) return false;
/** File line `line` is exactly a single-line annotation node. */
function isAnnotationLine(fileLines: readonly string[], path: string, line: number): boolean {
return nodeChain(fileLines, path, line).some(
node => node.startLine === line && node.endLine === line && ANNOTATION_NODE_KINDS[node.kind] === true,
);
}
/** Every file line in the inclusive range is an annotation node. */
function isAnnotationEchoRun(
fileLines: readonly string[],
path: string | undefined,
first: number,
last: number,
): boolean {
if (path === undefined) return false;
for (let line = first; line <= last; line++) {
if (!isAnnotationLine(fileLines, path, line)) return false;
}
return true;
}
@@ -345,20 +371,22 @@ interface TextualBoundaryNormalization {
}
/**
* Normalize exact boundary echoes without interpreting language tokens.
* Normalize exact boundary echoes.
*
* Two-sided echoes are removed when stripping both copies leaves one payload
* row per deleted range line. One-sided echoes are removed when the remaining
* payload still covers the full range — on multi-line ranges from line
* equality alone, on single-line ranges only when every echoed row is an
* annotation ({@link ANNOTATION_ROW_RE}), where an adjacent duplicate is never
* intentional. An under-filled one-sided echo is recorded as ambiguous so the
* syntax-probe search gets first chance to resolve it, then rejected rather
* than silently dropping unique range content.
* equality alone, on single-line ranges only when every echoed row is a
* grammar-classified annotation ({@link ANNOTATION_NODE_KINDS}), where an
* adjacent duplicate is never intentional. An under-filled one-sided echo is
* recorded as ambiguous so the syntax-probe search gets first chance to
* resolve it, then rejected rather than silently dropping unique range
* content.
*/
function normalizeTextualBoundaryEchoes(
edits: readonly AppliedEdit[],
fileLines: readonly string[],
path: string | undefined,
): TextualBoundaryNormalization {
const out: AppliedEdit[] = [];
const warnings: string[] = [];
@@ -383,7 +411,10 @@ function normalizeTextualBoundaryEchoes(
dropLeading = leading;
dropTrailing = trailing;
}
} else if (leading > 0 && (rangeLength > 1 || isAnnotationEchoRun(group.payload, 0, leading))) {
} else if (
leading > 0 &&
(rangeLength > 1 || isAnnotationEchoRun(fileLines, path, group.startLine - leading, group.startLine - 1))
) {
if (group.payload.length - leading >= rangeLength) {
dropLeading = leading;
} else {
@@ -396,7 +427,7 @@ function normalizeTextualBoundaryEchoes(
}
} else if (
trailing > 0 &&
(rangeLength > 1 || isAnnotationEchoRun(group.payload, group.payload.length - trailing, group.payload.length))
(rangeLength > 1 || isAnnotationEchoRun(fileLines, path, group.endLine + 1, group.endLine + trailing))
) {
if (group.payload.length - trailing >= rangeLength) {
dropTrailing = trailing;
@@ -972,6 +1003,37 @@ function resolveShiftedLanding(
return landing === group.anchor ? undefined : { line: landing, crossed };
}
/**
* Body shape required for an opener-escape relocation: the rows tile into
* leading single-line nodes (comments, attributes) followed by one multi-line
* construct reaching the last content row — i.e. the body parses standalone
* as one self-contained `mod`/`fn`/`class` that can be moved past the block
* it was mis-anchored into without re-parenting anything. Bare statements and
* multi-statement bodies fail it and stay literal: an under-indented one-line
* body more likely names the inside of the block.
*/
function bodyIsRelocatableConstruct(rows: readonly string[], path: string): boolean {
let last = rows.length;
while (last > 0 && !hasNonWhitespace(rows[last - 1])) last--;
if (last === 0) return false;
let line = 1;
while (line <= last) {
if (!hasNonWhitespace(rows[line - 1])) {
line++;
continue;
}
const spans = nodeChain(rows, path, line);
let end = 0;
for (const span of spans) {
if (span.startLine === line && span.endLine > end) end = span.endLine;
}
if (end === 0) return false; // no node begins here — body does not parse as items
if (end >= last) return end > line; // final node must be one multi-line construct
line = end + 1;
}
return false;
}
/**
* Resolve where a block-lowered after-insert anchored on the block's closing
* line should land given a body depth `target` deeper than that closer: just
@@ -1015,14 +1077,17 @@ function resolveInwardLanding(
/**
* Slide mis-anchored after-insert hunks to the depth their body indentation
* claims: outward past the structural closer lines that follow the anchor
* when the body is shallower, or — for `insert_after_block N:` lowerings —
* inward across the block's trailing closers when the body is deeper than
* the block's closing line. Returns the corrected edit list plus one warning
* per shifted hunk.
* when the body is shallower; for plain inserts anchored on a block opener,
* past the whole block when the body is a balanced construct claiming a
* depth strictly above the opener (syntax-probe verified via `path`); or — for
* `insert_after_block N:` lowerings — inward across the block's trailing
* closers when the body is deeper than the block's closing line. Returns the
* corrected edit list plus one warning per shifted hunk.
*/
function repairAfterInsertLandings(
edits: readonly AppliedEdit[],
fileLines: readonly string[],
path: string | undefined,
): { edits: readonly AppliedEdit[]; warnings: string[] } {
// Group plain (non-replacement) after-anchor inserts per authored hunk:
// rows of one hunk share the anchor line and the patch header line.
@@ -1064,7 +1129,59 @@ function repairAfterInsertLandings(
warnings.push(afterInsertLandingShiftWarning(group.anchor, outward.line, outward.crossed));
continue;
}
if (group.blockStart === undefined) continue;
if (group.blockStart === undefined) {
// Opener-escape: `PUT >N:` anchored on a line that OPENS a construct,
// with a self-contained construct body claiming a column depth
// strictly above the opener — a landing between the opener and its
// first statement, which no such body can intend, yet one that can
// parse (items are legal inside Rust fn bodies). Candidates are the
// enclosing constructs' end lines from the node chain, innermost
// first, kept only when the construct's own opening depth sits at or
// above the body's claim (the body could be its sibling); the first
// candidate whose relocated result passes the syntax probe wins.
// Equal-depth bodies stay literal, matching the outward shift.
if (path === undefined) continue;
const anchorText = fileLines[group.anchor - 1] ?? "";
const targetCols = indentColumns(target);
if (targetCols >= indentColumns(anchorText)) continue;
const chain = nodeChain(fileLines, path, group.anchor);
if (!chain.some(node => node.startLine === group.anchor && node.endLine > group.anchor)) continue;
const rows = group.members.map(idx => insertEditAt(edits, idx).text);
if (!bodyIsRelocatableConstruct(rows, path)) continue;
const candidates = [
...new Set(
chain
.filter(
node =>
node.endLine > group.anchor && indentColumns(fileLines[node.startLine - 1] ?? "") <= targetCols,
)
.map(node => node.endLine),
),
].sort((a, b) => a - b);
for (const landing of candidates) {
// Never relocate across another hunk's target; farther
// candidates cross the same line, so stop outright.
let blocked = false;
for (const targeted of targetedLines) {
if (targeted > group.anchor && targeted <= landing) {
blocked = true;
break;
}
}
if (blocked) break;
const trial = [...(out ?? edits)];
for (const idx of group.members) {
const edit = insertEditAt(trial, idx);
trial[idx] = { ...edit, cursor: { kind: "after_anchor", anchor: { line: landing } } };
}
if (parsesCleanly(path, materializeEdits(fileLines, trial).text)) {
out = trial;
warnings.push(afterInsertOpenerEscapeWarning(group.anchor, landing));
break;
}
}
continue;
}
const inward = resolveInwardLanding(group, target, group.blockStart, fileLines, targetedLines);
if (inward === undefined) continue;
retarget(group, inward);
@@ -1095,7 +1212,6 @@ export interface ApplyEditsOptions {
interface Materialized {
text: string;
firstChangedLine: number | undefined;
warnings: string[];
}
/**
@@ -1105,7 +1221,6 @@ interface Materialized {
* veto.
*/
function materializeEdits(originalLines: readonly string[], edits: readonly AppliedEdit[]): Materialized {
const { edits: landed, warnings } = repairAfterInsertLandings(edits, originalLines);
const fileLines = [...originalLines];
const lineOrigins: LineOrigin[] = fileLines.map(() => "original");
@@ -1118,7 +1233,7 @@ function materializeEdits(originalLines: readonly string[], edits: readonly Appl
const bofLines: string[] = [];
const eofLines: string[] = [];
const anchorEdits: IndexedEdit[] = [];
landed.forEach((edit, idx) => {
edits.forEach((edit, idx) => {
if (edit.kind === "insert" && edit.cursor.kind === "bof") {
bofLines.push(edit.text);
} else if (edit.kind === "insert" && edit.cursor.kind === "eof") {
@@ -1182,7 +1297,7 @@ function materializeEdits(originalLines: readonly string[], edits: readonly Appl
const eofChangedLine = insertAtEnd(fileLines, lineOrigins, eofLines);
if (eofChangedLine !== undefined) trackFirstChanged(eofChangedLine);
return { text: fileLines.join("\n"), firstChangedLine, warnings };
return { text: fileLines.join("\n"), firstChangedLine };
}
/**
@@ -1226,13 +1341,14 @@ export function applyEdits(text: string, edits: readonly Edit[], options: ApplyE
);
validateLineBounds(targetEdits, fileLines);
const indentationWarnings = repairReplacementIndentation(targetEdits, fileLines);
const normalized = normalizeTextualBoundaryEchoes(targetEdits, fileLines);
const leading = [...clipboardWarnings, ...indentationWarnings, ...normalized.warnings];
const landed = repairAfterInsertLandings(targetEdits, fileLines, options.path);
const normalized = normalizeTextualBoundaryEchoes(landed.edits, fileLines, options.path);
const leading = [...clipboardWarnings, ...indentationWarnings, ...landed.warnings, ...normalized.warnings];
const authoredResult = materializeEdits(fileLines, normalized.edits);
const baselineParses = parsesCleanly(options.path, text);
const authoredParses = parsesCleanly(options.path, authoredResult.text);
const finish = (result: Materialized, warnings: string[]): ApplyResult => {
const merged = [...warnings, ...result.warnings];
const merged = [...warnings];
// Post-apply syntax advisory: the result stopped parsing while the
// pre-edit text parsed, so this patch demonstrably introduced the
// error. Catches misplacements no boundary variant can explain.
+14
View File
@@ -459,6 +459,20 @@ export function blockInsertLandingShiftWarning(blockStart: number, closerLine: n
return `PUT >${blockStart}*: body indented deeper than closing line ${closerLine}, so it was placed inside the block, after line ${landingLine}. \`PUT >N*\` lands AFTER the block at sibling depth — if inside was intended, use plain \`PUT >${closerLine}:\`.`;
}
/**
* Plain `PUT >N:` anchored on a block-opener line with a shallower construct
* body: the landing was moved past the whole block — anchoring on an opener
* places the body between the opener and its first statement, a position a
* body at the opener's depth or above never intends.
*/
export function afterInsertOpenerEscapeWarning(anchorLine: number, landingLine: number): string {
return (
`PUT >${anchorLine}: line ${anchorLine} opens a block, and the body's indentation claims a position ` +
`outside it, so the body was landed after line ${landingLine} (verified by the syntax probe). ` +
`To insert after a whole construct, anchor on its closing line or use \`PUT >N*:\`.`
);
}
/** `Recovery`: an external write matched a cached snapshot. */
export const RECOVERY_EXTERNAL_WARNING =
"Recovered from a stale file hash using a previous read snapshot (file changed externally between read and edit).";
+31 -1
View File
@@ -9,7 +9,9 @@
* normalization remains available.
*/
import { enclosingBlockBoundaries } from "@oh-my-pi/pi-natives";
import { enclosingBlockBoundaries, type NodeSpan, nodeChainAt } from "@oh-my-pi/pi-natives";
export type { NodeSpan };
/** Parse-result cache keyed by content hash + path; FIFO-bounded. */
const parseCache = new Map<string, boolean>();
@@ -17,6 +19,34 @@ const PARSE_CACHE_MAX = 256;
const boundaryCache = new Map<string, readonly number[]>();
const chainCache = new Map<string, readonly NodeSpan[]>();
/**
* Named-node chain (innermost-first) containing `line`, excluding the file
* root: single-line nodes beginning on the line (attributes, decorators)
* first, then every enclosing construct. Empty when the language is unknown,
* the line is blank, or the source cannot parse — callers treat empty as
* "no structural evidence", never as evidence about the line.
*/
export function nodeChain(lines: readonly string[], path: string, line: number): readonly NodeSpan[] {
const text = lines.join("\n");
const key = `${Bun.hash(text).toString(36)}:${text.length}:${path}:${line}`;
const cached = chainCache.get(key);
if (cached !== undefined) return cached;
let chain: readonly NodeSpan[];
try {
chain = nodeChainAt({ code: text, path, line }) ?? [];
} catch {
chain = [];
}
if (chainCache.size >= PARSE_CACHE_MAX) {
const oldest = chainCache.keys().next().value;
if (oldest !== undefined) chainCache.delete(oldest);
}
chainCache.set(key, chain);
return chain;
}
/** Syntactic node boundaries outside a visible source range. */
export function enclosingBoundaries(
lines: readonly string[],
@@ -217,3 +217,112 @@ describe("insert-after-block inward landing shift", () => {
expect(result.warnings?.some(w => /placed inside the block/.test(w)) ?? false).toBe(false);
});
});
/**
* Opener-escape landing correction — the stdout_policy incident: a plain
* `PUT >N:` anchored on the OPENING line of `fn clone_box` inserted a whole
* tab-indented test `mod` between the opener and its body. The result parsed
* (items are legal inside Rust fn bodies), so no probe warning fired and the
* corruption landed silently. Contract under test: a balanced construct body
* claiming a column depth strictly above the opener is landed after the first
* closer returning to that depth, verified by the syntax probe; statements,
* equal-depth bodies, unverifiable languages, and unparseable relocations
* stay literal.
*/
describe("opener-anchored after-insert escape (the stdout_policy incident)", () => {
// Mirrors crates/pi-builtins/src/host.rs: 3-space file, tab-indented body.
const RUST_FILE = [
"mod testing {", // 1
" struct MemStream;", // 2
"", // 3
" impl Stream for MemStream {", // 4
" fn clone_box(&self) -> u32 {", // 5
" 1", // 6
" }", // 7
"", // 8
" fn try_borrow(&self) -> u32 {", // 9
" 7", // 10
" }", // 11
" }", // 12
"}", // 13
"",
].join("\n");
const MOD_BODY = [
"PUT >5:",
"+",
"+\tmod stdout_policy {",
"+\t\tuse super::MemStream;",
"+",
"+\t\t#[test]",
"+\t\tfn line_policy() {",
"+\t\t\tassert!(true);",
"+\t\t}",
"+\t}",
].join("\n");
function applyRust(text: string, patch: string): { text: string; warnings: string[] } {
const result = applyEdits(text, parsePatch(patch).edits, { path: "fixture.rs" });
return { text: result.text, warnings: result.warnings ?? [] };
}
it("lands a tab-indented construct body after the block, not inside the opener", () => {
const { text, warnings } = applyRust(RUST_FILE, MOD_BODY);
const lines = text.split("\n");
// The fn body stays contiguous with its opener.
expect(lines[4]).toBe(" fn clone_box(&self) -> u32 {");
expect(lines[5]).toBe(" 1");
// The mod landed after the impl closer (line 12), inside `mod testing`.
expect(lines[11]).toBe(" }");
expect(lines[13]).toBe("\tmod stdout_policy {");
expect(warnings.some(w => /PUT >5: line 5 opens a block/.test(w))).toBe(true);
});
it("keeps a bare shallower statement literal (first-statement inserts survive)", () => {
const { text, warnings } = applyRust(RUST_FILE, "PUT >5:\n+ let x = 1;");
expect(text.split("\n")[5]).toBe(" let x = 1;");
expect(warnings.some(w => /opens a block/.test(w))).toBe(false);
});
it("keeps an equal-depth body literal (matches the outward shift's contract)", () => {
const body = ["PUT >5:", "+ fn extra(&self) -> u32 {", "+ 2", "+ }"].join("\n");
const { text, warnings } = applyRust(RUST_FILE, body);
expect(text.split("\n")[5]).toBe(" fn extra(&self) -> u32 {");
expect(warnings.some(w => /opens a block/.test(w))).toBe(false);
});
it("abandons the escape when the relocated result does not parse", () => {
// Balanced `{`-construct that is not a legal item: relocating it to
// `mod testing` scope would break the parse, so it stays literal.
const body = ["PUT >5:", "+ Some(1) => {", "+ }"].join("\n");
const { text, warnings } = applyRust(RUST_FILE, body);
expect(text.split("\n")[5]).toBe(" Some(1) => {");
expect(warnings.some(w => /opens a block/.test(w))).toBe(false);
});
it("stays literal without a parseable path (no probe, no relocation)", () => {
const { edits } = parsePatch(MOD_BODY);
const result = applyEdits(RUST_FILE, edits);
expect(result.text.split("\n")[6]).toBe("\tmod stdout_policy {");
expect((result.warnings ?? []).some(w => /opens a block/.test(w))).toBe(false);
});
it("escapes a shallower function body past a class in TypeScript", () => {
const file = ["class A {", " method() {", " return 1;", " }", "}", ""].join("\n");
const body = ["PUT >2:", "+function helper() {", "+ return 2;", "+}"].join("\n");
const result = applyEdits(file, parsePatch(body).edits, { path: "fixture.ts" });
expect(result.text).toBe(
[
"class A {",
" method() {",
" return 1;",
" }",
"}",
"function helper() {",
" return 2;",
"}",
"",
].join("\n"),
);
expect((result.warnings ?? []).some(w => /PUT >2: line 2 opens a block/.test(w))).toBe(true);
});
});
+1
View File
@@ -6,6 +6,7 @@
- Added `ClaudeV3`/`ClaudeV47`/`ClaudeV5` encodings to `countTokens`: a Rust rewrite of [ctok](https://github.com/sanderland/ctok) by Sander Land (MIT), reconstructing Anthropic's `count_tokens` offline. Counts are exact on ctok's ~3.4M-response measurement corpora; the port is validated against 493 Python-ctok reference fixtures covering all three families. The pipeline is byte-level throughout — markers occupy one byte, normalization borrows text no rule touches, ASCII and ideographs skip the Unicode tables, and pieces are matched with one Aho-Corasick transition per byte instead of a per-position vocabulary descent — which counts English prose at 64 MiB/s, markdown at 73 MiB/s, source code at 35 MiB/s and CJK at 49 MiB/s per core: 1.5× (CJK, already cheap per byte) to 5.5× (prose, markdown, digits) a straightforward character-level implementation of the same model, which is held to byte-for-byte identical counts across 2.4M randomized differential comparisons.
- Added zstd-embedded exact content tokenizers for Qwen 3.5+/3.6+/3.8, DeepSeek V3/V4/R1, Kimi K2/K3, and GLM-5 alongside the rebuilt OpenAI o200k/cl100k and Claude reconstructions. `countTokens` now reads JavaScript strings through a reusable UTF-16 buffer, so native counting does not allocate a UTF-8 temporary.
- Added `nodeChainAt`: the named tree-sitter node chain containing a line, innermost-first, with grammar kind and line span per node. Powers hashline's structural edit repairs (annotation-row classification, opener-anchored insert relocation) without lexical heuristics.
### Changed
+20
View File
@@ -1551,6 +1551,26 @@ export interface MinimizerResult {
*/
export declare function mmrRerankIndices(contents: Array<string>, scores: Float64Array, lambdaParam: number, topK: number): Uint32Array
/**
* Named-node chain containing `options.line`, innermost-first, excluding the
* whole-file root.
*
* Single-line nodes beginning on the line (attributes, decorators) come
* first, followed by every enclosing construct. ERROR/MISSING recovery nodes
* are skipped. Returns `null` when the language is unrecognized, the line is
* out of range / blank, or the source fails to parse entirely.
*/
export declare function nodeChainAt(options: BlockRangeOptions): Array<NodeSpan> | null
export interface NodeSpan {
/** 1-indexed inclusive first line of the node. */
startLine: number
/** 1-indexed inclusive last content line of the node. */
endLine: number
/** Tree-sitter grammar node kind (e.g. `attribute_item`, `function_item`). */
kind: string
}
/** Parsed Kitty keyboard protocol sequence result for a Kitty input sequence. */
export interface ParsedKittyResult {
/** Primary codepoint associated with the key. */
+1
View File
@@ -67,6 +67,7 @@ export const matchesKey = nativeBindings.matchesKey;
export const matchesKittySequence = nativeBindings.matchesKittySequence;
export const matchesLegacySequence = nativeBindings.matchesLegacySequence;
export const mmrRerankIndices = nativeBindings.mmrRerankIndices;
export const nodeChainAt = nativeBindings.nodeChainAt;
export const parseKey = nativeBindings.parseKey;
export const parseKittySequence = nativeBindings.parseKittySequence;
export const pdfToMarkdown = nativeBindings.pdfToMarkdown;