fix(natives): reject ill-formed UTF-16 in native diff and fall back to jsdiff

This commit is contained in:
Wolfgang Schoenberger
2026-07-22 04:32:46 -07:00
parent 55bba890ab
commit 8f17a0300d
4 changed files with 100 additions and 18 deletions
+58 -15
View File
@@ -18,8 +18,27 @@
use std::{collections::HashMap, rc::Rc};
use napi::{JsString, bindgen_prelude::*};
use napi_derive::napi;
/// Decode a JS string strictly: unpaired surrogates are rejected instead of
/// being replaced with U+FFFD, so callers can fall back to a UTF-16-aware
/// JS diff and both sides keep byte-identical jsdiff semantics.
fn strict_utf16_to_string(text: JsString) -> Result<String> {
let utf16 = text.into_utf16()?;
// `as_slice` includes the trailing NUL terminator; drop it like napi's
// own `as_str` does (a legitimate U+0000 in the content sits before it).
let Some((_, units)) = utf16.as_slice().split_last() else {
return Ok(String::new());
};
String::from_utf16(units).map_err(|_| {
Error::new(
Status::InvalidArg,
"ill-formed UTF-16 input (unpaired surrogate); caller must fall back to a JS diff",
)
})
}
/// One jsdiff change object: a run of added, removed, or common tokens.
#[napi(object)]
pub struct DiffChange {
@@ -308,9 +327,13 @@ fn diff_line_tokens<'a>(old_tokens: &[&'a str], new_tokens: &[&'a str]) -> Vec<R
/// options). Change values keep line terminators, and common runs are joined
/// from the new text.
#[napi]
pub fn diff_lines(old_text: String, new_text: String) -> Vec<DiffChange> {
let old_tokens = line_tokens(&old_text);
let new_tokens = line_tokens(&new_text);
pub fn diff_lines(old_text: JsString, new_text: JsString) -> Result<Vec<DiffChange>> {
Ok(diff_lines_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?))
}
fn diff_lines_impl(old_text: &str, new_text: &str) -> Vec<DiffChange> {
let old_tokens = line_tokens(old_text);
let new_tokens = line_tokens(new_text);
let runs = diff_line_tokens(&old_tokens, &new_tokens);
build_changes(&runs, &old_tokens, &new_tokens, |tokens| tokens.concat())
}
@@ -322,7 +345,11 @@ pub fn diff_lines(old_text: String, new_text: String) -> Vec<DiffChange> {
/// Callers that map line numbers — like hashline recovery — need the counts,
/// not another copy of the text.
#[napi]
pub fn diff_line_runs(old_text: String, new_text: String) -> Vec<DiffRun> {
pub fn diff_line_runs(old_text: JsString, new_text: JsString) -> Result<Vec<DiffRun>> {
Ok(diff_line_runs_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?))
}
fn diff_line_runs_impl(old_text: &str, new_text: &str) -> Vec<DiffRun> {
let old_tokens: Vec<&str> = old_text.split('\n').collect();
let new_tokens: Vec<&str> = new_text.split('\n').collect();
let (old_ids, new_ids) = intern_exact(&old_tokens, &new_tokens);
@@ -341,13 +368,25 @@ pub fn diff_line_runs(old_text: String, new_text: String) -> Vec<DiffRun> {
/// semantics. `context` defaults to 4 like jsdiff.
#[napi]
pub fn structured_patch_hunks(
old_text: String,
new_text: String,
old_text: JsString,
new_text: JsString,
context: Option<u32>,
) -> Result<Vec<PatchHunk>> {
Ok(structured_patch_hunks_impl(
&strict_utf16_to_string(old_text)?,
&strict_utf16_to_string(new_text)?,
context,
))
}
fn structured_patch_hunks_impl(
old_text: &str,
new_text: &str,
context: Option<u32>,
) -> Vec<PatchHunk> {
let context = context.map_or(4usize, |value| value as usize);
let old_tokens = line_tokens(&old_text);
let new_tokens = line_tokens(&new_text);
let old_tokens = line_tokens(old_text);
let new_tokens = line_tokens(new_text);
let runs = diff_line_tokens(&old_tokens, &new_tokens);
// Change list with per-change line slices; the trailing sentinel mirrors
@@ -778,9 +817,13 @@ fn word_post_process(changes: &mut [DiffChange]) {
/// Tokens carry surrounding whitespace, equality ignores it, and the
/// post-pass dedupes whitespace across change boundaries.
#[napi]
pub fn diff_words(old_text: String, new_text: String) -> Vec<DiffChange> {
let old_tokens = word_tokens(&old_text);
let new_tokens = word_tokens(&new_text);
pub fn diff_words(old_text: JsString, new_text: JsString) -> Result<Vec<DiffChange>> {
Ok(diff_words_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?))
}
fn diff_words_impl(old_text: &str, new_text: &str) -> Vec<DiffChange> {
let old_tokens = word_tokens(old_text);
let new_tokens = word_tokens(new_text);
let old_refs: Vec<&str> = old_tokens.iter().map(String::as_str).collect();
let new_refs: Vec<&str> = new_tokens.iter().map(String::as_str).collect();
// Equality is whitespace-insensitive: intern by trimmed text.
@@ -798,7 +841,7 @@ mod tests {
use super::*;
fn lines(old: &str, new: &str) -> Vec<(String, bool, bool)> {
diff_lines(old.to_string(), new.to_string())
diff_lines_impl(old, new)
.into_iter()
.map(|c| (c.value, c.added, c.removed))
.collect()
@@ -825,7 +868,7 @@ mod tests {
#[test]
fn structured_patch_marks_missing_eof_newline() {
let hunks = structured_patch_hunks("a\nb".into(), "a\nc".into(), Some(3));
let hunks = structured_patch_hunks_impl("a\nb", "a\nc", Some(3));
assert_eq!(hunks.len(), 1);
assert_eq!(hunks[0].lines, vec![
" a",
@@ -839,7 +882,7 @@ mod tests {
#[test]
fn word_diff_dedupes_boundary_whitespace() {
// jsdiff's documented example 2: K:'foo ' D:'bar' I:'qux' K:' baz'.
let changes = diff_words("foo bar baz".into(), "foo qux baz".into());
let changes = diff_words_impl("foo bar baz", "foo qux baz");
let shaped: Vec<(String, bool, bool)> = changes
.into_iter()
.map(|c| (c.value, c.added, c.removed))
@@ -854,7 +897,7 @@ mod tests {
#[test]
fn line_runs_preserve_empty_lines() {
let runs = diff_line_runs("a\n\nb".into(), "a\n\nc".into());
let runs = diff_line_runs_impl("a\n\nb", "a\n\nc");
let shaped: Vec<(u32, bool, bool)> = runs
.into_iter()
.map(|r| (r.count, r.added, r.removed))
+16 -1
View File
@@ -4,13 +4,28 @@
* Provides diff string generation and the replace-mode edit logic
* used when not in patch mode.
*/
import { diffLines, structuredPatchHunks } from "@oh-my-pi/pi-natives";
import { diffLines as nativeDiffLines, structuredPatchHunks as nativeStructuredPatchHunks } from "@oh-my-pi/pi-natives";
import { diffLines as jsDiffLines, structuredPatch as jsStructuredPatch } from "diff";
import { resolveToCwd } from "../tools/path-utils";
import { type BlockContextSource, findBlockContextLines } from "../utils/block-context";
import { DEFAULT_FUZZY_THRESHOLD, EditMatchError, findMatch } from "./modes/replace";
import { adjustIndentation, normalizeToLF, stripBom } from "./normalize";
import { readEditFileText } from "./read-file";
/** Native line diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */
function diffLines(oldContent: string, newContent: string) {
return oldContent.isWellFormed() && newContent.isWellFormed()
? nativeDiffLines(oldContent, newContent)
: jsDiffLines(oldContent, newContent);
}
/** Native structured-patch hunks with the same ill-formed-UTF-16 fallback as {@link diffLines}. */
function structuredPatchHunks(oldContent: string, newContent: string, context: number) {
return oldContent.isWellFormed() && newContent.isWellFormed()
? nativeStructuredPatchHunks(oldContent, newContent, context)
: jsStructuredPatch("", "", oldContent, newContent, "", "", { context }).hunks;
}
export interface DiffResult {
diff: string;
firstChangedLine: number | undefined;
@@ -1,8 +1,16 @@
import { diffWords } from "@oh-my-pi/pi-natives";
import { diffWords as nativeDiffWords } from "@oh-my-pi/pi-natives";
import { DEFAULT_TAB_WIDTH, sanitizeText } from "@oh-my-pi/pi-utils";
import { diffWords as jsDiffWords } from "diff";
import { getLanguageFromPath, highlightCode, theme } from "../../modes/theme/theme";
import { type CodeFrameMarker, formatCodeFrameLine, replaceTabs } from "../../tools/render-utils";
/** Native word diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */
function diffWords(oldContent: string, newContent: string) {
return oldContent.isWellFormed() && newContent.isWellFormed()
? nativeDiffWords(oldContent, newContent)
: jsDiffWords(oldContent, newContent);
}
/** SGR dim on / normal intensity — additive, preserves fg/bg colors. */
const DIM = "\x1b[2m";
const DIM_OFF = "\x1b[22m";
+17 -1
View File
@@ -6,7 +6,8 @@
* Recovery fails closed when the target changed or became ambiguous. The
* patcher then returns a mismatch with fresh context instead of guessing.
*/
import { diffLineRuns } from "@oh-my-pi/pi-natives";
import { diffLineRuns as nativeDiffLineRuns } from "@oh-my-pi/pi-natives";
import { diffArrays } from "diff";
import { applyEdits } from "./apply";
import { RECOVERY_EXTERNAL_WARNING, RECOVERY_LINE_REMAP_WARNING, RECOVERY_SESSION_CHAIN_WARNING } from "./messages";
import type { SnapshotStore } from "./snapshots";
@@ -44,6 +45,21 @@ function getEditAnchors(edit: Edit): Anchor[] {
return edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor" ? [edit.cursor.anchor] : [];
}
/** Native line-run diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */
function diffLineRuns(
previousText: string,
currentText: string,
): { count: number; added: boolean; removed: boolean }[] {
if (previousText.isWellFormed() && currentText.isWellFormed()) {
return nativeDiffLineRuns(previousText, currentText);
}
return diffArrays(previousText.split("\n"), currentText.split("\n")).map(change => ({
count: change.count,
added: change.added,
removed: change.removed,
}));
}
function buildLineMap(previousText: string, currentText: string): Map<number, number> {
const changes = diffLineRuns(previousText, currentText);
const map = new Map<number, number>();