From 1e5017bf7d5891f285adcf6b2fd5dff553c72266 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 8 Jun 2026 14:29:59 +0200 Subject: [PATCH] feat: enabled block-boundary context support across read and diff previews - Added tree-sitter `enclosing_block_boundaries` API with line range models. - Added N-API `enclosingBlockBoundaries` bridge and exported JS declarations. - Replaced matching-bracket context resolution with source-aware block context in read and diff flows. - Passed source path through diff/read generators to surface native block boundary previews. --- crates/pi-ast/src/block.rs | 198 +++++++++++++++++- crates/pi-natives/src/block.rs | 48 +++++ packages/coding-agent/CHANGELOG.md | 2 +- packages/coding-agent/src/edit/diff.ts | 33 ++- .../coding-agent/src/edit/hashline/diff.ts | 2 +- .../coding-agent/src/edit/hashline/execute.ts | 2 +- packages/coding-agent/src/edit/modes/patch.ts | 8 +- .../coding-agent/src/edit/modes/replace.ts | 2 +- .../coding-agent/src/prompts/tools/read.md | 4 +- packages/coding-agent/src/tools/read.ts | 40 ++-- ...{matching-brackets.ts => block-context.ts} | 102 ++++++++- .../test/read-multi-range.test.ts | 31 +++ .../coding-agent/test/tools/edit-diff.test.ts | 3 +- packages/natives/CHANGELOG.md | 4 + packages/natives/native/index.d.ts | 32 +++ packages/natives/native/index.js | 1 + 16 files changed, 464 insertions(+), 48 deletions(-) rename packages/coding-agent/src/utils/{matching-brackets.ts => block-context.ts} (62%) diff --git a/crates/pi-ast/src/block.rs b/crates/pi-ast/src/block.rs index 567e9f95f..fd551d9a6 100644 --- a/crates/pi-ast/src/block.rs +++ b/crates/pi-ast/src/block.rs @@ -8,10 +8,12 @@ //! full span; pointing at a continuation line or a lone closing delimiter //! resolves to nothing. +use std::collections::BTreeSet; + use anyhow::{Result, anyhow}; use ast_grep_core::tree_sitter::LanguageExt; use serde::{Deserialize, Serialize}; -use tree_sitter::{Parser, Point}; +use tree_sitter::{Parser, Point, TreeCursor}; use crate::summary::{node_content_end_line, node_start_line, resolve_language}; @@ -111,6 +113,141 @@ pub fn block_range_at(options: BlockRangeOptions) -> Result> })) } +#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)] +pub struct LineRange { + /// 1-indexed inclusive first visible line. + pub start_line: u32, + /// 1-indexed inclusive last visible line. + pub end_line: u32, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct EnclosingBoundaryOptions { + /// Source code to inspect. + pub code: String, + /// Language alias (e.g. "rust", "typescript") used before path inference. + pub lang: Option, + /// File path used to infer language by extension when `lang` is omitted. + pub path: Option, + /// 1-indexed inclusive visible line ranges (the lines actually shown). + pub ranges: Vec, +} + +/// Sort, drop invalid, and merge adjacent/overlapping ranges so visibility +/// tests can binary-search a non-overlapping list. +fn normalize_ranges(mut ranges: Vec) -> Vec { + ranges.retain(|range| range.start_line > 0 && range.end_line >= range.start_line); + ranges.sort_by(|a, b| { + a.start_line + .cmp(&b.start_line) + .then(a.end_line.cmp(&b.end_line)) + }); + let mut merged: Vec = Vec::with_capacity(ranges.len()); + for range in ranges { + if let Some(last) = merged.last_mut() + && range.start_line <= last.end_line.saturating_add(1) + { + last.end_line = last.end_line.max(range.end_line); + continue; + } + merged.push(range); + } + merged +} + +fn is_visible(merged: &[LineRange], line: u32) -> bool { + merged + .binary_search_by(|range| { + if line < range.start_line { + std::cmp::Ordering::Greater + } else if line > range.end_line { + std::cmp::Ordering::Less + } else { + std::cmp::Ordering::Equal + } + }) + .is_ok() +} + +/// Depth-first walk collecting boundary lines from every multi-line named node +/// that straddles a visible-range edge. A single reused [`TreeCursor`] keeps +/// the traversal allocation-free. +fn collect_boundaries(cursor: &mut TreeCursor<'_>, merged: &[LineRange], out: &mut BTreeSet) { + let node = cursor.node(); + // Skip the whole-file root: its only "boundary" is EOF, never a useful + // matching line (mirrors `block_range_at` excluding the root). + if node.is_named() && node.parent().is_some() { + let start = node_start_line(node); + let end = node_content_end_line(node); + if end > start { + let start_visible = is_visible(merged, start); + let end_visible = is_visible(merged, end); + // Opener shown, closer off-window → surface the closer (and vice + // versa). A node fully inside or fully outside the window adds + // nothing. + if start_visible && !end_visible { + out.insert(end); + } else if end_visible && !start_visible { + out.insert(start); + } + } + } + if cursor.goto_first_child() { + loop { + collect_boundaries(cursor, merged, out); + if !cursor.goto_next_sibling() { + break; + } + } + cursor.goto_parent(); + } +} + +/// Generalize "show the matching bracket" to every tree-sitter block: for each +/// multi-line named node whose span crosses the visible window, return the +/// boundary line sitting *outside* that window. +/// +/// - node opens on a visible line but closes past the window → its closing line +/// - node closes on a visible line but opens before the window → its opening +/// line +/// +/// Because the trigger is an endpoint *inside* the window, the result is +/// bounded by the window size (not nesting depth), exactly like a bracket scan +/// — but it also covers indentation languages (Python) and uses real syntactic +/// spans. +/// +/// Returns `None` when the language is unrecognized or the source fails to +/// parse / carries a syntax error (caller falls back to a lexical bracket +/// scan); `Some(sorted unique boundary lines)` otherwise (possibly empty). +pub fn enclosing_block_boundaries(options: EnclosingBoundaryOptions) -> Result>> { + let EnclosingBoundaryOptions { code, lang, path, ranges } = options; + let merged = normalize_ranges(ranges); + if code.is_empty() || merged.is_empty() { + return Ok(Some(Vec::new())); + } + let Some(language) = resolve_language(lang.as_deref(), path.as_deref()) else { + return Ok(None); + }; + let mut parser = Parser::new(); + parser + .set_language(&language.get_ts_language()) + .map_err(|err| anyhow!("Failed to load tree-sitter language: {err}"))?; + let Some(tree) = parser.parse(&code, None) else { + return Ok(None); + }; + let root = tree.root_node(); + // A file-level syntax error makes error-recovery spans unreliable; defer to + // the lexical scanner rather than emit boundaries off a broken tree. + if root.has_error() { + return Ok(None); + } + + let mut boundaries = BTreeSet::new(); + let mut cursor = root.walk(); + collect_boundaries(&mut cursor, &merged, &mut boundaries); + Ok(Some(boundaries.into_iter().collect())) +} + #[cfg(test)] mod tests { use super::*; @@ -220,4 +357,63 @@ mod tests { let code = "struct A;\nstruct B {\n x: u32,\n}\n"; assert_eq!(resolve(code, "r.rs", 2), Some(BlockRange { start_line: 2, end_line: 4 })); } + + fn boundaries(code: &str, path: &str, ranges: &[(u32, u32)]) -> Option> { + enclosing_block_boundaries(EnclosingBoundaryOptions { + code: code.to_string(), + lang: None, + path: Some(path.to_string()), + ranges: ranges + .iter() + .map(|&(start_line, end_line)| LineRange { start_line, end_line }) + .collect(), + }) + .expect("boundary resolution succeeds") + } + + const TS_FN: &str = "function outer() {\n const a = 1;\n const b = 2;\n const c = 3;\n \ + return a + b + c;\n}\nafter();\n"; + + #[test] + fn surfaces_closing_brace_for_visible_opener() { + // Window is the opening line only; its block closes on line 6. + assert_eq!(boundaries(TS_FN, "x.ts", &[(1, 1)]), Some(vec![6])); + } + + #[test] + fn surfaces_opening_brace_for_visible_closer() { + // Window is the closing line only; its block opens on line 1. + assert_eq!(boundaries(TS_FN, "x.ts", &[(6, 6)]), Some(vec![1])); + } + + #[test] + fn interior_only_window_adds_no_boundary() { + // Neither the opener (1) nor the closer (6) is visible, so the bracket + // scan would add nothing — and neither do we. + assert_eq!(boundaries(TS_FN, "x.ts", &[(3, 4)]), Some(vec![])); + } + + #[test] + fn whole_file_window_adds_no_boundary() { + assert_eq!(boundaries(TS_FN, "x.ts", &[(1, 7)]), Some(vec![])); + } + + #[test] + fn python_indentation_block_uses_syntactic_span() { + // Python has no closing delimiter — the def's span ends at the last + // body line. Showing the `def` header surfaces that end line. + let code = "def greet(name):\n a = 1\n b = 2\n return a + b\n"; + assert_eq!(boundaries(code, "g.py", &[(1, 1)]), Some(vec![4])); + } + + #[test] + fn syntax_error_falls_back_to_none() { + let code = "function broken() {\n if (y) {\n"; + assert_eq!(boundaries(code, "b.ts", &[(1, 1)]), None); + } + + #[test] + fn unrecognized_language_falls_back_to_none() { + assert_eq!(boundaries(TS_FN, "x.unknownext", &[(1, 1)]), None); + } } diff --git a/crates/pi-natives/src/block.rs b/crates/pi-natives/src/block.rs index 88d942693..4bdd14e59 100644 --- a/crates/pi-natives/src/block.rs +++ b/crates/pi-natives/src/block.rs @@ -45,3 +45,51 @@ pub fn block_range_at(options: BlockRangeOptions) -> Result> .map(|range| range.map(Into::into)) .map_err(|error| Error::from_reason(error.to_string())) } + +#[napi(object)] +pub struct LineRange { + /// 1-indexed inclusive first visible line. + pub start_line: u32, + /// 1-indexed inclusive last visible line. + pub end_line: u32, +} + +#[napi(object)] +pub struct EnclosingBoundaryOptions { + /// Source code to inspect. + pub code: String, + /// Language alias (e.g. "rust", "typescript") used before path inference. + pub lang: Option, + /// File path used to infer language by extension when `lang` is omitted. + pub path: Option, + /// 1-indexed inclusive visible line ranges (the lines actually shown). + pub ranges: Vec, +} + +/// Matching-bracket context for an arbitrary tree-sitter language. +/// +/// For each multi-line named node whose span crosses the visible window, return +/// the boundary line sitting *outside* that window (the closer when the opener +/// is shown, the opener when the closer is shown). Covers brace and indentation +/// languages alike using real syntactic spans. +/// +/// Returns `null` when the language is unrecognized or the source fails to +/// parse / carries a syntax error (caller should fall back to a lexical scan); +/// a sorted, unique list of 1-indexed boundary lines otherwise. +#[napi] +pub fn enclosing_block_boundaries(options: EnclosingBoundaryOptions) -> Result>> { + pi_ast::block::enclosing_block_boundaries(pi_ast::block::EnclosingBoundaryOptions { + code: options.code, + lang: options.lang, + path: options.path, + ranges: options + .ranges + .into_iter() + .map(|range| pi_ast::block::LineRange { + start_line: range.start_line, + end_line: range.end_line, + }) + .collect(), + }) + .map_err(|error| Error::from_reason(error.to_string())) +} diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 670528061..a3ba00a7d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -15,7 +15,7 @@ ### Fixed - Fixed task-row shimmer timing so every running description starts its highlight on the first character together and reaches the last character together, regardless of text length. -- Fixed read and edit previews to include the matching bracket line behind an ellipsis when a shown opening/closing bracket pairs outside the displayed range. +- Fixed read and edit previews to surface the enclosing syntactic block's off-window boundary line (behind an ellipsis) when a shown line opens or closes a block whose other end falls outside the displayed range. Powered by a new tree-sitter `enclosingBlockBoundaries` native, so it covers brace languages and indentation languages (Python) using real syntactic spans, with a lexical bracket scan as fallback for unparseable sources. - Fixed `tab.screenshot({ selector })` hanging for the entire cell budget on continuously-animating pages (WebGL / `backdrop-filter` "glass" effects). The element-screenshot path no longer routes through puppeteer's `scrollIntoViewIfNeeded()`, whose `IntersectionObserver` promise can stall indefinitely under heavy rendering; it now does a single instant `scrollIntoView` and captures with `scrollIntoView: false` (relying on `captureBeyondViewport`), so off-screen elements are still captured without the stall. - Fixed follow-up message submissions to forward pending clipboard-pasted images to `session.prompt` in both streaming and non-streaming flows - Fixed follow-up handling to clear consumed clipboard image state after submission so pasted images are not silently carried into later messages diff --git a/packages/coding-agent/src/edit/diff.ts b/packages/coding-agent/src/edit/diff.ts index 1cd9b11c9..6759f5ae3 100644 --- a/packages/coding-agent/src/edit/diff.ts +++ b/packages/coding-agent/src/edit/diff.ts @@ -6,7 +6,7 @@ */ import * as Diff from "diff"; import { resolveToCwd } from "../tools/path-utils"; -import { findMatchingBracketContextLines } from "../utils/matching-brackets"; +import { type BlockContextSource, findBlockContextLines } from "../utils/block-context"; import { DEFAULT_FUZZY_THRESHOLD, EditMatchError, findMatch } from "./modes/replace"; import { adjustIndentation, normalizeToLF, stripBom } from "./normalize"; import { readEditFileText } from "./read-file"; @@ -127,7 +127,12 @@ function insertBracketContextRows( } } -function addMatchingBracketContextRows(rows: string[], oldLines: readonly string[], newLines: readonly string[]): void { +function addMatchingBracketContextRows( + rows: string[], + oldLines: readonly string[], + newLines: readonly string[], + source: BlockContextSource, +): void { const oldVisible: number[] = []; const newVisible: number[] = []; const seenRows = new Set(rows); @@ -139,15 +144,20 @@ function addMatchingBracketContextRows(rows: string[], oldLines: readonly string else newVisible.push(parsed.lineNumber); } - insertBracketContextRows(rows, "old", findMatchingBracketContextLines(oldLines, oldVisible), seenRows); - insertBracketContextRows(rows, "new", findMatchingBracketContextLines(newLines, newVisible), seenRows); + insertBracketContextRows(rows, "old", findBlockContextLines(oldLines, oldVisible, source), seenRows); + insertBracketContextRows(rows, "new", findBlockContextLines(newLines, newVisible, source), seenRows); } /** * Generate a unified diff string with line numbers and context. * Returns both the diff string and the first changed line number (in the new file). */ -export function generateDiffString(oldContent: string, newContent: string, contextLines = 2): DiffResult { +export function generateDiffString( + oldContent: string, + newContent: string, + contextLines = 2, + source: BlockContextSource = {}, +): DiffResult { const parts = Diff.diffLines(oldContent, newContent); const output: string[] = []; @@ -251,7 +261,7 @@ export function generateDiffString(oldContent: string, newContent: string, conte } } - addMatchingBracketContextRows(output, oldContent.split("\n"), newContent.split("\n")); + addMatchingBracketContextRows(output, oldContent.split("\n"), newContent.split("\n"), source); return { diff: output.join("\n"), firstChangedLine }; } @@ -280,7 +290,12 @@ export interface ReplaceResult { * Generate a unified diff string without file headers. * Returns both the diff string and the first changed line number (in the new file). */ -export function generateUnifiedDiffString(oldContent: string, newContent: string, contextLines = 3): DiffResult { +export function generateUnifiedDiffString( + oldContent: string, + newContent: string, + contextLines = 3, + source: BlockContextSource = {}, +): DiffResult { const patch = Diff.structuredPatch("", "", oldContent, newContent, "", "", { context: contextLines }); const output: string[] = []; let firstChangedLine: number | undefined; @@ -311,7 +326,7 @@ export function generateUnifiedDiffString(oldContent: string, newContent: string } } - addMatchingBracketContextRows(output, oldContent.split("\n"), newContent.split("\n")); + addMatchingBracketContextRows(output, oldContent.split("\n"), newContent.split("\n"), source); return { diff: output.join("\n"), firstChangedLine }; } @@ -900,7 +915,7 @@ export async function computeEditDiff( }; } - return generateDiffString(normalizedContent, result.content); + return generateDiffString(normalizedContent, result.content, undefined, { path }); } catch (err) { return { error: err instanceof Error ? err.message : String(err) }; } diff --git a/packages/coding-agent/src/edit/hashline/diff.ts b/packages/coding-agent/src/edit/hashline/diff.ts index 534aa43ef..fe3fecdda 100644 --- a/packages/coding-agent/src/edit/hashline/diff.ts +++ b/packages/coding-agent/src/edit/hashline/diff.ts @@ -230,7 +230,7 @@ export async function computeHashlineSectionDiff( if (options.streaming) return buildStreamingSectionDiff(section, normalized); const result = applyPreviewEdits({ section, absolutePath, normalized, snapshots, options }); if (normalized === result.text) return { error: `No changes would be made to ${section.path}.` }; - return generateDiffString(normalized, result.text); + return generateDiffString(normalized, result.text, undefined, { path: section.path }); } catch (err) { return { error: err instanceof Error ? err.message : String(err) }; } diff --git a/packages/coding-agent/src/edit/hashline/execute.ts b/packages/coding-agent/src/edit/hashline/execute.ts index dffdd61c3..54d091c94 100644 --- a/packages/coding-agent/src/edit/hashline/execute.ts +++ b/packages/coding-agent/src/edit/hashline/execute.ts @@ -97,7 +97,7 @@ function renderSection(result: PatchSectionResult, diagnostics: FileDiagnosticsR }; } - const diff = generateDiffString(result.before, result.after); + const diff = generateDiffString(result.before, result.after, undefined, { path: result.path }); const preview = buildCompactDiffPreview(diff.diff); const meta = outputMeta() .diagnostics(diagnostics?.summary ?? "", diagnostics?.messages ?? []) diff --git a/packages/coding-agent/src/edit/modes/patch.ts b/packages/coding-agent/src/edit/modes/patch.ts index 96d8a6a24..2734734f1 100644 --- a/packages/coding-agent/src/edit/modes/patch.ts +++ b/packages/coding-agent/src/edit/modes/patch.ts @@ -1571,7 +1571,9 @@ export async function computePatchDiff( if (!normalizedOld && !normalizedNew) { return { diff: "", firstChangedLine: undefined }; } - return generateUnifiedDiffString(normalizedOld, normalizedNew); + return generateUnifiedDiffString(normalizedOld, normalizedNew, undefined, { + path: result.change.newPath ?? result.change.path, + }); } catch (err) { return { error: err instanceof Error ? err.message : String(err) }; } @@ -1785,7 +1787,9 @@ export async function executePatchSingle( if (result.change.type === "update" && result.change.oldContent && result.change.newContent) { const normalizedOld = normalizeToLF(stripBom(result.change.oldContent).text); const normalizedNew = normalizeToLF(stripBom(result.change.newContent).text); - diffResult = generateUnifiedDiffString(normalizedOld, normalizedNew); + diffResult = generateUnifiedDiffString(normalizedOld, normalizedNew, undefined, { + path: result.change.newPath ?? result.change.path, + }); } let resultText: string; diff --git a/packages/coding-agent/src/edit/modes/replace.ts b/packages/coding-agent/src/edit/modes/replace.ts index be3fde872..4784bd75d 100644 --- a/packages/coding-agent/src/edit/modes/replace.ts +++ b/packages/coding-agent/src/edit/modes/replace.ts @@ -1078,7 +1078,7 @@ export async function executeReplaceSingle( ); invalidateFsScanAfterWrite(absolutePath); - const diffResult = generateDiffString(normalizedContent, result.content); + const diffResult = generateDiffString(normalizedContent, result.content, undefined, { path }); const resultText = result.count > 1 ? `Successfully replaced ${result.count} occurrences in ${path}.` diff --git a/packages/coding-agent/src/prompts/tools/read.md b/packages/coding-agent/src/prompts/tools/read.md index 70a658f3e..4bdb25d28 100644 --- a/packages/coding-agent/src/prompts/tools/read.md +++ b/packages/coding-agent/src/prompts/tools/read.md @@ -18,8 +18,8 @@ Append `:` to `path`. The bare path falls back to the default mode. - `:50` / `:50-` — read from line 50 onward. - `:50-200` — lines 50–200 inclusive. - `:50+150` — 150 lines starting at line 50. -- `:20+1` — anchor on line 20 (single-range reads expand by ≤1 leading and ≤3 trailing context lines; if an emitted bracket pairs with a line outside the window, that matching line is included behind an ellipsis). -- `:5-16,960-973` — multiple ranges in one call (sorted, overlaps merged). Multi-range mode returns exact requested bounds plus matching bracket lines when a shown bracket's pair sits outside those bounds. +- `:20+1` — anchor on line 20 (single-range reads expand by ≤1 leading and ≤3 trailing context lines). +- `:5-16,960-973` — multiple ranges in one call (sorted, overlaps merged). Multi-range mode returns exact bounds with no context padding. - `:raw` — verbatim text; no anchors, no summary, no line prefixes. - `:2-4:raw` or `:raw:2-4` — range AND verbatim; the two compose in either order. - `:conflicts` — one-line-per-block index of every unresolved git merge conflict. diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index e19e5587f..9b39f4b97 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -35,14 +35,10 @@ import { } from "../session/streaming-output"; import { fileHyperlink, renderCodeCell, renderMarkdownCell, renderStatusLine, tryResolveInternalUrlSync } from "../tui"; import { CachedOutputBlock, markFramedBlockComponent } from "../tui/output-block"; +import { buildLineEntriesWithBlockContext, type LineEntry, lineEntriesToPlainText } from "../utils/block-context"; import { resolveFileDisplayMode } from "../utils/file-display-mode"; import { ImageInputTooLargeError, loadImageInput, MAX_IMAGE_INPUT_BYTES } from "../utils/image-loading"; import { convertFileWithMarkit } from "../utils/markit"; -import { - buildLineEntriesWithMatchingBracketContext, - type LineEntry, - lineEntriesToPlainText, -} from "../utils/matching-brackets"; import { buildDirectoryTree, type DirectoryTree } from "../workspace-tree"; import { type ArchiveReader, formatArchiveEntryLines, openArchive, parseArchivePathCandidates } from "./archive-reader"; import { @@ -961,9 +957,9 @@ export class ReadTool implements AgentTool { return prependHashlineHeader(formatted, hashContext); }; const buildLineEntries = (endLineDisplay: number): LineEntry[] => - buildLineEntriesWithMatchingBracketContext(allLines, [ - { startLine: startLineDisplay, endLine: endLineDisplay }, - ]); + buildLineEntriesWithBlockContext(allLines, [{ startLine: startLineDisplay, endLine: endLineDisplay }], { + path: options.sourcePath, + }); let outputText: string; let truncationInfo: @@ -1090,7 +1086,7 @@ export class ReadTool implements AgentTool { if (options.raw === true) { outputText = rawParts.length > 0 ? rawParts.join("\n\n…\n\n") : ""; } else if (visibleSpans.length > 0) { - const entries = buildLineEntriesWithMatchingBracketContext(allLines, visibleSpans); + const entries = buildLineEntriesWithBlockContext(allLines, visibleSpans, { path: options.sourcePath }); const firstLine = entries.find(entry => entry.kind === "line"); if (firstLine?.kind === "line") { details.displayContent = { @@ -1220,16 +1216,21 @@ export class ReadTool implements AgentTool { let outputText: string; if (!rawSelector && fullLines && visibleSpans.length > 0) { - const entries = buildLineEntriesWithMatchingBracketContext(fullLines, visibleSpans, { - lineText: (lineNumber, sourceText) => { - const visibleText = displayLineByNumber.get(lineNumber); - if (visibleText !== undefined) return visibleText; - if (maxColumns <= 0) return sourceText; - const truncated = truncateLine(sourceText, maxColumns); - if (truncated.wasTruncated) columnTruncated = maxColumns; - return truncated.text; + const entries = buildLineEntriesWithBlockContext( + fullLines, + visibleSpans, + { path: absolutePath }, + { + lineText: (lineNumber, sourceText) => { + const visibleText = displayLineByNumber.get(lineNumber); + if (visibleText !== undefined) return visibleText; + if (maxColumns <= 0) return sourceText; + const truncated = truncateLine(sourceText, maxColumns); + if (truncated.wasTruncated) columnTruncated = maxColumns; + return truncated.text; + }, }, - }); + ); const firstLine = entries.find(entry => entry.kind === "line"); displayContent = { text: lineEntriesToPlainText(entries, BRACKET_CONTEXT_ELLIPSIS), @@ -2100,9 +2101,10 @@ export class ReadTool implements AgentTool { }; const formatBracketAwareText = (): string | undefined => { if (!bracketContextFullLines) return undefined; - const entries = buildLineEntriesWithMatchingBracketContext( + const entries = buildLineEntriesWithBlockContext( bracketContextFullLines, [{ startLine: startLineDisplay, endLine: displayedEndLine }], + { path: absolutePath }, { lineText: (lineNumber, sourceText) => { const visibleText = displayLineByNumber.get(lineNumber); diff --git a/packages/coding-agent/src/utils/matching-brackets.ts b/packages/coding-agent/src/utils/block-context.ts similarity index 62% rename from packages/coding-agent/src/utils/matching-brackets.ts rename to packages/coding-agent/src/utils/block-context.ts index 23105b99c..5b450cf97 100644 --- a/packages/coding-agent/src/utils/matching-brackets.ts +++ b/packages/coding-agent/src/utils/block-context.ts @@ -1,3 +1,6 @@ +import { enclosingBlockBoundaries } from "@oh-my-pi/pi-natives"; +import { logger } from "@oh-my-pi/pi-utils"; + const OPEN_TO_CLOSE: Record = { "(": ")", "[": "]", @@ -15,6 +18,12 @@ export interface LineSpan { endLine: number; } +/** Where the source came from, so tree-sitter can pick a grammar. */ +export interface BlockContextSource { + path?: string; + lang?: string; +} + export type LineEntry = { kind: "line"; lineNumber: number; text: string; context: boolean } | { kind: "ellipsis" }; interface StackEntry { @@ -63,6 +72,57 @@ function hasEveryLineVisible(visible: ReadonlySet, totalLines: number): return totalLines > 0 && visible.size >= totalLines; } +/** Collapse a set of visible line numbers into sorted, merged inclusive spans. */ +function visibleSetToSpans(visible: ReadonlySet): LineSpan[] { + const sorted = [...visible].sort((left, right) => left - right); + const spans: LineSpan[] = []; + for (const line of sorted) { + const previous = spans[spans.length - 1]; + if (previous && line <= previous.endLine + 1) { + previous.endLine = line; + continue; + } + spans.push({ startLine: line, endLine: line }); + } + return spans; +} + +/** + * Tree-sitter-backed block boundaries. For each multi-line named node whose + * span crosses the visible window, the native side returns the boundary line + * outside that window (closer when the opener is shown, opener when the closer + * is shown). Returns `null` when the language is unrecognized or the source has + * a syntax error so the caller can fall back to a lexical bracket scan. + */ +function nativeBlockContext( + fullLines: readonly string[], + visible: ReadonlySet, + source: BlockContextSource, +): Map | null { + if (!source.path && !source.lang) return null; + const ranges = visibleSetToSpans(visible); + if (ranges.length === 0) return new Map(); + let boundaries: number[] | null; + try { + boundaries = enclosingBlockBoundaries({ + code: fullLines.join("\n"), + path: source.path, + lang: source.lang, + ranges, + }); + } catch (error) { + logger.debug("enclosingBlockBoundaries failed; using lexical bracket fallback", { error }); + return null; + } + if (boundaries === null) return null; + const context = new Map(); + for (const lineNumber of boundaries) { + if (visible.has(lineNumber)) continue; + context.set(lineNumber, fullLines[lineNumber - 1] ?? ""); + } + return context; +} + function findMatchingStackIndex(stack: readonly StackEntry[], opener: string): number { for (let index = stack.length - 1; index >= 0; index--) { if (stack[index].opener === opener) return index; @@ -79,14 +139,14 @@ function isHashCommentStart(line: string, index: number): boolean { return true; } -export function findMatchingBracketContextLines( - fullLines: readonly string[], - visibleLinesInput: ReadonlySet | readonly number[], -): Map { - const visible = visibleLinesInput instanceof Set ? visibleLinesInput : new Set(visibleLinesInput); +/** + * Lexical bracket-matching fallback for sources tree-sitter can't parse + * (unknown extensions, syntax errors). Pairs `()[]{}` while skipping strings + * and line/block comments, and reports the matching line when one endpoint is + * visible and the other is not. + */ +function lexicalBracketContext(fullLines: readonly string[], visible: ReadonlySet): Map { const context = new Map(); - if (visible.size === 0 || hasEveryLineVisible(visible, fullLines.length)) return context; - const stack: StackEntry[] = []; let mode: ScannerMode = "code"; let escaped = false; @@ -189,16 +249,40 @@ export function findMatchingBracketContextLines( return context; } -export function buildLineEntriesWithMatchingBracketContext( +/** + * Resolve the off-window boundary lines for a visible window: tree-sitter + * syntactic spans first (covers brace and indentation languages), falling back + * to a lexical bracket scan when the grammar is unavailable. Returns a map of + * `lineNumber → source text` for the lines to surface, never including a line + * already visible. + */ +export function findBlockContextLines( + fullLines: readonly string[], + visibleInput: ReadonlySet | readonly number[], + source: BlockContextSource = {}, +): Map { + const visible = visibleInput instanceof Set ? visibleInput : new Set(visibleInput); + if (visible.size === 0 || hasEveryLineVisible(visible, fullLines.length)) return new Map(); + return nativeBlockContext(fullLines, visible, source) ?? lexicalBracketContext(fullLines, visible); +} + +/** + * Build display entries for `visibleSpans` plus any off-window block-boundary + * lines, in source order, with `{ kind: "ellipsis" }` markers inserted across + * non-contiguous gaps. `options.lineText` lets callers substitute display text + * (e.g. column-truncated lines) for a given line number. + */ +export function buildLineEntriesWithBlockContext( fullLines: readonly string[], visibleSpans: readonly LineSpan[], + source: BlockContextSource = {}, options: { lineText?: (lineNumber: number, sourceText: string, context: boolean) => string; } = {}, ): LineEntry[] { const spans = normalizeLineSpans(visibleSpans, fullLines.length); const visible = visibleLineNumbers(spans); - const context = findMatchingBracketContextLines(fullLines, visible); + const context = findBlockContextLines(fullLines, visible, source); const allLines = new Set(visible); for (const lineNumber of context.keys()) allLines.add(lineNumber); diff --git a/packages/coding-agent/test/read-multi-range.test.ts b/packages/coding-agent/test/read-multi-range.test.ts index 00fc40bf8..975fe0ce2 100644 --- a/packages/coding-agent/test/read-multi-range.test.ts +++ b/packages/coding-agent/test/read-multi-range.test.ts @@ -120,6 +120,37 @@ describe("read tool multi-range selector", () => { expect(text).not.toContain("const four = 4"); }); + it("uses tree-sitter syntactic spans for indentation languages (Python)", async () => { + const filePath = path.join(tmpDir, "module.py"); + await fs.writeFile( + filePath, + [ + "def greet(name):", + " a = 1", + " b = 2", + " c = 3", + " d = 4", + " e = 5", + " f = 6", + " g = 7", + " return a + b + c + d + e + f + g + len(name)", + "trailing = 1", + ].join("\n"), + ); + + const tool = new ReadTool(createSession(tmpDir)); + // Read only the `def` header (expands by a few trailing context lines). + // Python has no closing delimiter, so a bracket scan would surface + // nothing; tree-sitter surfaces the def's last body line (9) as the + // block boundary, behind an ellipsis for the skipped middle. + const text = textOutput(await tool.execute("call-py-def", { path: `${filePath}:1-1` })); + + expect(text).toContain("def greet(name):"); + expect(text).toContain("…"); + expect(text).toContain("return a + b + c + d + e + f + g + len(name)"); + expect(text).not.toContain("trailing = 1"); + }); + it("merges overlapping ranges into a single contiguous block", async () => { const filePath = path.join(tmpDir, "numbered.txt"); await fs.writeFile(filePath, makeNumberedContent(20)); diff --git a/packages/coding-agent/test/tools/edit-diff.test.ts b/packages/coding-agent/test/tools/edit-diff.test.ts index 964540acc..aec46fd04 100644 --- a/packages/coding-agent/test/tools/edit-diff.test.ts +++ b/packages/coding-agent/test/tools/edit-diff.test.ts @@ -35,8 +35,7 @@ describe("generateDiffString", () => { ]; const newLines = [...oldLines]; newLines[0] = "function renamed() {"; - - const result = generateDiffString(oldLines.join("\n"), newLines.join("\n"), 1); + const result = generateDiffString(oldLines.join("\n"), newLines.join("\n"), 1, { path: "sample.ts" }); const diffLines = result.diff.split("\n"); expect(diffLines).toContain("-1|function outer() {"); diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index 964643d9d..64fc2dea5 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added the `enclosingBlockBoundaries` native API (with `EnclosingBoundaryOptions` and `LineRange` types) that returns, for a set of visible line ranges, the off-window boundary lines of every multi-line tree-sitter node whose span crosses the window — the closer when an opener is shown and the opener when a closer is shown. Covers brace and indentation languages (Python) via real syntactic spans; returns `null` for unrecognized languages or sources with syntax errors so callers can fall back to a lexical scan. + ## [15.10.2] - 2026-06-08 ### Added diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index e8ee126e0..e462e5a0f 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -428,6 +428,31 @@ export declare enum Ellipsis { Omit = 2 } +/** + * Matching-bracket context for an arbitrary tree-sitter language. + * + * For each multi-line named node whose span crosses the visible window, return + * the boundary line sitting *outside* that window (the closer when the opener + * is shown, the opener when the closer is shown). Covers brace and indentation + * languages alike using real syntactic spans. + * + * Returns `null` when the language is unrecognized or the source fails to + * parse / carries a syntax error (caller should fall back to a lexical scan); + * a sorted, unique list of 1-indexed boundary lines otherwise. + */ +export declare function enclosingBlockBoundaries(options: EnclosingBoundaryOptions): Array | null + +export interface EnclosingBoundaryOptions { + /** Source code to inspect. */ + code: string + /** Language alias (e.g. "rust", "typescript") used before path inference. */ + lang?: string + /** File path used to infer language by extension when `lang` is omitted. */ + path?: string + /** 1-indexed inclusive visible line ranges (the lines actually shown). */ + ranges: Array +} + /** * Encode image bytes into a SIXEL escape sequence for terminal rendering. * @@ -907,6 +932,13 @@ export declare enum KeyEventType { Release = 3 } +export interface LineRange { + /** 1-indexed inclusive first visible line. */ + startLine: number + /** 1-indexed inclusive last visible line. */ + endLine: number +} + /** * Walk the workspace once and return tree entries plus AGENTS.md candidates. * diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index 938b782a9..a694fcdbc 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -31,6 +31,7 @@ export const blockRangeAt = nativeBindings.blockRangeAt; export const copyToClipboard = nativeBindings.copyToClipboard; export const countTokens = nativeBindings.countTokens; export const detectMacOSAppearance = nativeBindings.detectMacOSAppearance; +export const enclosingBlockBoundaries = nativeBindings.enclosingBlockBoundaries; export const encodeSixel = nativeBindings.encodeSixel; export const executeShell = nativeBindings.executeShell; export const extractSegments = nativeBindings.extractSegments;