diff --git a/crates/pi-natives/src/diff.rs b/crates/pi-natives/src/diff.rs index 0397bf9af..f97e9b82e 100644 --- a/crates/pi-natives/src/diff.rs +++ b/crates/pi-natives/src/diff.rs @@ -18,8 +18,27 @@ use std::{collections::HashMap, rc::Rc}; +use napi::{JsString, bindgen_prelude::*}; use napi_derive::napi; +/// Decode a JS string strictly: unpaired surrogates are rejected instead of +/// being replaced with U+FFFD, so callers can fall back to a UTF-16-aware +/// JS diff and both sides keep byte-identical jsdiff semantics. +fn strict_utf16_to_string(text: JsString) -> Result { + let utf16 = text.into_utf16()?; + // `as_slice` includes the trailing NUL terminator; drop it like napi's + // own `as_str` does (a legitimate U+0000 in the content sits before it). + let Some((_, units)) = utf16.as_slice().split_last() else { + return Ok(String::new()); + }; + String::from_utf16(units).map_err(|_| { + Error::new( + Status::InvalidArg, + "ill-formed UTF-16 input (unpaired surrogate); caller must fall back to a JS diff", + ) + }) +} + /// One jsdiff change object: a run of added, removed, or common tokens. #[napi(object)] pub struct DiffChange { @@ -308,9 +327,13 @@ fn diff_line_tokens<'a>(old_tokens: &[&'a str], new_tokens: &[&'a str]) -> Vec Vec { - let old_tokens = line_tokens(&old_text); - let new_tokens = line_tokens(&new_text); +pub fn diff_lines(old_text: JsString, new_text: JsString) -> Result> { + Ok(diff_lines_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?)) +} + +fn diff_lines_impl(old_text: &str, new_text: &str) -> Vec { + let old_tokens = line_tokens(old_text); + let new_tokens = line_tokens(new_text); let runs = diff_line_tokens(&old_tokens, &new_tokens); build_changes(&runs, &old_tokens, &new_tokens, |tokens| tokens.concat()) } @@ -322,7 +345,11 @@ pub fn diff_lines(old_text: String, new_text: String) -> Vec { /// Callers that map line numbers — like hashline recovery — need the counts, /// not another copy of the text. #[napi] -pub fn diff_line_runs(old_text: String, new_text: String) -> Vec { +pub fn diff_line_runs(old_text: JsString, new_text: JsString) -> Result> { + Ok(diff_line_runs_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?)) +} + +fn diff_line_runs_impl(old_text: &str, new_text: &str) -> Vec { let old_tokens: Vec<&str> = old_text.split('\n').collect(); let new_tokens: Vec<&str> = new_text.split('\n').collect(); let (old_ids, new_ids) = intern_exact(&old_tokens, &new_tokens); @@ -341,13 +368,25 @@ pub fn diff_line_runs(old_text: String, new_text: String) -> Vec { /// semantics. `context` defaults to 4 like jsdiff. #[napi] pub fn structured_patch_hunks( - old_text: String, - new_text: String, + old_text: JsString, + new_text: JsString, + context: Option, +) -> Result> { + Ok(structured_patch_hunks_impl( + &strict_utf16_to_string(old_text)?, + &strict_utf16_to_string(new_text)?, + context, + )) +} + +fn structured_patch_hunks_impl( + old_text: &str, + new_text: &str, context: Option, ) -> Vec { let context = context.map_or(4usize, |value| value as usize); - let old_tokens = line_tokens(&old_text); - let new_tokens = line_tokens(&new_text); + let old_tokens = line_tokens(old_text); + let new_tokens = line_tokens(new_text); let runs = diff_line_tokens(&old_tokens, &new_tokens); // Change list with per-change line slices; the trailing sentinel mirrors @@ -778,9 +817,13 @@ fn word_post_process(changes: &mut [DiffChange]) { /// Tokens carry surrounding whitespace, equality ignores it, and the /// post-pass dedupes whitespace across change boundaries. #[napi] -pub fn diff_words(old_text: String, new_text: String) -> Vec { - let old_tokens = word_tokens(&old_text); - let new_tokens = word_tokens(&new_text); +pub fn diff_words(old_text: JsString, new_text: JsString) -> Result> { + Ok(diff_words_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?)) +} + +fn diff_words_impl(old_text: &str, new_text: &str) -> Vec { + let old_tokens = word_tokens(old_text); + let new_tokens = word_tokens(new_text); let old_refs: Vec<&str> = old_tokens.iter().map(String::as_str).collect(); let new_refs: Vec<&str> = new_tokens.iter().map(String::as_str).collect(); // Equality is whitespace-insensitive: intern by trimmed text. @@ -798,7 +841,7 @@ mod tests { use super::*; fn lines(old: &str, new: &str) -> Vec<(String, bool, bool)> { - diff_lines(old.to_string(), new.to_string()) + diff_lines_impl(old, new) .into_iter() .map(|c| (c.value, c.added, c.removed)) .collect() @@ -825,7 +868,7 @@ mod tests { #[test] fn structured_patch_marks_missing_eof_newline() { - let hunks = structured_patch_hunks("a\nb".into(), "a\nc".into(), Some(3)); + let hunks = structured_patch_hunks_impl("a\nb", "a\nc", Some(3)); assert_eq!(hunks.len(), 1); assert_eq!(hunks[0].lines, vec![ " a", @@ -839,7 +882,7 @@ mod tests { #[test] fn word_diff_dedupes_boundary_whitespace() { // jsdiff's documented example 2: K:'foo ' D:'bar' I:'qux' K:' baz'. - let changes = diff_words("foo bar baz".into(), "foo qux baz".into()); + let changes = diff_words_impl("foo bar baz", "foo qux baz"); let shaped: Vec<(String, bool, bool)> = changes .into_iter() .map(|c| (c.value, c.added, c.removed)) @@ -854,7 +897,7 @@ mod tests { #[test] fn line_runs_preserve_empty_lines() { - let runs = diff_line_runs("a\n\nb".into(), "a\n\nc".into()); + let runs = diff_line_runs_impl("a\n\nb", "a\n\nc"); let shaped: Vec<(u32, bool, bool)> = runs .into_iter() .map(|r| (r.count, r.added, r.removed)) diff --git a/packages/coding-agent/src/edit/diff.ts b/packages/coding-agent/src/edit/diff.ts index b071ef77c..545823d05 100644 --- a/packages/coding-agent/src/edit/diff.ts +++ b/packages/coding-agent/src/edit/diff.ts @@ -4,13 +4,28 @@ * Provides diff string generation and the replace-mode edit logic * used when not in patch mode. */ -import { diffLines, structuredPatchHunks } from "@oh-my-pi/pi-natives"; +import { diffLines as nativeDiffLines, structuredPatchHunks as nativeStructuredPatchHunks } from "@oh-my-pi/pi-natives"; +import { diffLines as jsDiffLines, structuredPatch as jsStructuredPatch } from "diff"; import { resolveToCwd } from "../tools/path-utils"; import { type BlockContextSource, findBlockContextLines } from "../utils/block-context"; import { DEFAULT_FUZZY_THRESHOLD, EditMatchError, findMatch } from "./modes/replace"; import { adjustIndentation, normalizeToLF, stripBom } from "./normalize"; import { readEditFileText } from "./read-file"; +/** Native line diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */ +function diffLines(oldContent: string, newContent: string) { + return oldContent.isWellFormed() && newContent.isWellFormed() + ? nativeDiffLines(oldContent, newContent) + : jsDiffLines(oldContent, newContent); +} + +/** Native structured-patch hunks with the same ill-formed-UTF-16 fallback as {@link diffLines}. */ +function structuredPatchHunks(oldContent: string, newContent: string, context: number) { + return oldContent.isWellFormed() && newContent.isWellFormed() + ? nativeStructuredPatchHunks(oldContent, newContent, context) + : jsStructuredPatch("", "", oldContent, newContent, "", "", { context }).hunks; +} + export interface DiffResult { diff: string; firstChangedLine: number | undefined; diff --git a/packages/coding-agent/src/modes/components/diff.ts b/packages/coding-agent/src/modes/components/diff.ts index 917989511..f6f421703 100644 --- a/packages/coding-agent/src/modes/components/diff.ts +++ b/packages/coding-agent/src/modes/components/diff.ts @@ -1,8 +1,16 @@ -import { diffWords } from "@oh-my-pi/pi-natives"; +import { diffWords as nativeDiffWords } from "@oh-my-pi/pi-natives"; import { DEFAULT_TAB_WIDTH, sanitizeText } from "@oh-my-pi/pi-utils"; +import { diffWords as jsDiffWords } from "diff"; import { getLanguageFromPath, highlightCode, theme } from "../../modes/theme/theme"; import { type CodeFrameMarker, formatCodeFrameLine, replaceTabs } from "../../tools/render-utils"; +/** Native word diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */ +function diffWords(oldContent: string, newContent: string) { + return oldContent.isWellFormed() && newContent.isWellFormed() + ? nativeDiffWords(oldContent, newContent) + : jsDiffWords(oldContent, newContent); +} + /** SGR dim on / normal intensity — additive, preserves fg/bg colors. */ const DIM = "\x1b[2m"; const DIM_OFF = "\x1b[22m"; diff --git a/packages/hashline/src/recovery.ts b/packages/hashline/src/recovery.ts index f6456b23f..bca26ebfd 100644 --- a/packages/hashline/src/recovery.ts +++ b/packages/hashline/src/recovery.ts @@ -6,7 +6,8 @@ * Recovery fails closed when the target changed or became ambiguous. The * patcher then returns a mismatch with fresh context instead of guessing. */ -import { diffLineRuns } from "@oh-my-pi/pi-natives"; +import { diffLineRuns as nativeDiffLineRuns } from "@oh-my-pi/pi-natives"; +import { diffArrays } from "diff"; import { applyEdits } from "./apply"; import { RECOVERY_EXTERNAL_WARNING, RECOVERY_LINE_REMAP_WARNING, RECOVERY_SESSION_CHAIN_WARNING } from "./messages"; import type { SnapshotStore } from "./snapshots"; @@ -44,6 +45,21 @@ function getEditAnchors(edit: Edit): Anchor[] { return edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor" ? [edit.cursor.anchor] : []; } +/** Native line-run diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */ +function diffLineRuns( + previousText: string, + currentText: string, +): { count: number; added: boolean; removed: boolean }[] { + if (previousText.isWellFormed() && currentText.isWellFormed()) { + return nativeDiffLineRuns(previousText, currentText); + } + return diffArrays(previousText.split("\n"), currentText.split("\n")).map(change => ({ + count: change.count, + added: change.added, + removed: change.removed, + })); +} + function buildLineMap(previousText: string, currentText: string): Map { const changes = diffLineRuns(previousText, currentText); const map = new Map();