fix(natives): reject ill-formed UTF-16 in native diff and fall back to jsdiff
This commit is contained in:
@@ -18,8 +18,27 @@
|
||||
|
||||
use std::{collections::HashMap, rc::Rc};
|
||||
|
||||
use napi::{JsString, bindgen_prelude::*};
|
||||
use napi_derive::napi;
|
||||
|
||||
/// Decode a JS string strictly: unpaired surrogates are rejected instead of
|
||||
/// being replaced with U+FFFD, so callers can fall back to a UTF-16-aware
|
||||
/// JS diff and both sides keep byte-identical jsdiff semantics.
|
||||
fn strict_utf16_to_string(text: JsString) -> Result<String> {
|
||||
let utf16 = text.into_utf16()?;
|
||||
// `as_slice` includes the trailing NUL terminator; drop it like napi's
|
||||
// own `as_str` does (a legitimate U+0000 in the content sits before it).
|
||||
let Some((_, units)) = utf16.as_slice().split_last() else {
|
||||
return Ok(String::new());
|
||||
};
|
||||
String::from_utf16(units).map_err(|_| {
|
||||
Error::new(
|
||||
Status::InvalidArg,
|
||||
"ill-formed UTF-16 input (unpaired surrogate); caller must fall back to a JS diff",
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
/// One jsdiff change object: a run of added, removed, or common tokens.
|
||||
#[napi(object)]
|
||||
pub struct DiffChange {
|
||||
@@ -308,9 +327,13 @@ fn diff_line_tokens<'a>(old_tokens: &[&'a str], new_tokens: &[&'a str]) -> Vec<R
|
||||
/// options). Change values keep line terminators, and common runs are joined
|
||||
/// from the new text.
|
||||
#[napi]
|
||||
pub fn diff_lines(old_text: String, new_text: String) -> Vec<DiffChange> {
|
||||
let old_tokens = line_tokens(&old_text);
|
||||
let new_tokens = line_tokens(&new_text);
|
||||
pub fn diff_lines(old_text: JsString, new_text: JsString) -> Result<Vec<DiffChange>> {
|
||||
Ok(diff_lines_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?))
|
||||
}
|
||||
|
||||
fn diff_lines_impl(old_text: &str, new_text: &str) -> Vec<DiffChange> {
|
||||
let old_tokens = line_tokens(old_text);
|
||||
let new_tokens = line_tokens(new_text);
|
||||
let runs = diff_line_tokens(&old_tokens, &new_tokens);
|
||||
build_changes(&runs, &old_tokens, &new_tokens, |tokens| tokens.concat())
|
||||
}
|
||||
@@ -322,7 +345,11 @@ pub fn diff_lines(old_text: String, new_text: String) -> Vec<DiffChange> {
|
||||
/// Callers that map line numbers — like hashline recovery — need the counts,
|
||||
/// not another copy of the text.
|
||||
#[napi]
|
||||
pub fn diff_line_runs(old_text: String, new_text: String) -> Vec<DiffRun> {
|
||||
pub fn diff_line_runs(old_text: JsString, new_text: JsString) -> Result<Vec<DiffRun>> {
|
||||
Ok(diff_line_runs_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?))
|
||||
}
|
||||
|
||||
fn diff_line_runs_impl(old_text: &str, new_text: &str) -> Vec<DiffRun> {
|
||||
let old_tokens: Vec<&str> = old_text.split('\n').collect();
|
||||
let new_tokens: Vec<&str> = new_text.split('\n').collect();
|
||||
let (old_ids, new_ids) = intern_exact(&old_tokens, &new_tokens);
|
||||
@@ -341,13 +368,25 @@ pub fn diff_line_runs(old_text: String, new_text: String) -> Vec<DiffRun> {
|
||||
/// semantics. `context` defaults to 4 like jsdiff.
|
||||
#[napi]
|
||||
pub fn structured_patch_hunks(
|
||||
old_text: String,
|
||||
new_text: String,
|
||||
old_text: JsString,
|
||||
new_text: JsString,
|
||||
context: Option<u32>,
|
||||
) -> Result<Vec<PatchHunk>> {
|
||||
Ok(structured_patch_hunks_impl(
|
||||
&strict_utf16_to_string(old_text)?,
|
||||
&strict_utf16_to_string(new_text)?,
|
||||
context,
|
||||
))
|
||||
}
|
||||
|
||||
fn structured_patch_hunks_impl(
|
||||
old_text: &str,
|
||||
new_text: &str,
|
||||
context: Option<u32>,
|
||||
) -> Vec<PatchHunk> {
|
||||
let context = context.map_or(4usize, |value| value as usize);
|
||||
let old_tokens = line_tokens(&old_text);
|
||||
let new_tokens = line_tokens(&new_text);
|
||||
let old_tokens = line_tokens(old_text);
|
||||
let new_tokens = line_tokens(new_text);
|
||||
let runs = diff_line_tokens(&old_tokens, &new_tokens);
|
||||
|
||||
// Change list with per-change line slices; the trailing sentinel mirrors
|
||||
@@ -778,9 +817,13 @@ fn word_post_process(changes: &mut [DiffChange]) {
|
||||
/// Tokens carry surrounding whitespace, equality ignores it, and the
|
||||
/// post-pass dedupes whitespace across change boundaries.
|
||||
#[napi]
|
||||
pub fn diff_words(old_text: String, new_text: String) -> Vec<DiffChange> {
|
||||
let old_tokens = word_tokens(&old_text);
|
||||
let new_tokens = word_tokens(&new_text);
|
||||
pub fn diff_words(old_text: JsString, new_text: JsString) -> Result<Vec<DiffChange>> {
|
||||
Ok(diff_words_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?))
|
||||
}
|
||||
|
||||
fn diff_words_impl(old_text: &str, new_text: &str) -> Vec<DiffChange> {
|
||||
let old_tokens = word_tokens(old_text);
|
||||
let new_tokens = word_tokens(new_text);
|
||||
let old_refs: Vec<&str> = old_tokens.iter().map(String::as_str).collect();
|
||||
let new_refs: Vec<&str> = new_tokens.iter().map(String::as_str).collect();
|
||||
// Equality is whitespace-insensitive: intern by trimmed text.
|
||||
@@ -798,7 +841,7 @@ mod tests {
|
||||
use super::*;
|
||||
|
||||
fn lines(old: &str, new: &str) -> Vec<(String, bool, bool)> {
|
||||
diff_lines(old.to_string(), new.to_string())
|
||||
diff_lines_impl(old, new)
|
||||
.into_iter()
|
||||
.map(|c| (c.value, c.added, c.removed))
|
||||
.collect()
|
||||
@@ -825,7 +868,7 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn structured_patch_marks_missing_eof_newline() {
|
||||
let hunks = structured_patch_hunks("a\nb".into(), "a\nc".into(), Some(3));
|
||||
let hunks = structured_patch_hunks_impl("a\nb", "a\nc", Some(3));
|
||||
assert_eq!(hunks.len(), 1);
|
||||
assert_eq!(hunks[0].lines, vec![
|
||||
" a",
|
||||
@@ -839,7 +882,7 @@ mod tests {
|
||||
#[test]
|
||||
fn word_diff_dedupes_boundary_whitespace() {
|
||||
// jsdiff's documented example 2: K:'foo ' D:'bar' I:'qux' K:' baz'.
|
||||
let changes = diff_words("foo bar baz".into(), "foo qux baz".into());
|
||||
let changes = diff_words_impl("foo bar baz", "foo qux baz");
|
||||
let shaped: Vec<(String, bool, bool)> = changes
|
||||
.into_iter()
|
||||
.map(|c| (c.value, c.added, c.removed))
|
||||
@@ -854,7 +897,7 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn line_runs_preserve_empty_lines() {
|
||||
let runs = diff_line_runs("a\n\nb".into(), "a\n\nc".into());
|
||||
let runs = diff_line_runs_impl("a\n\nb", "a\n\nc");
|
||||
let shaped: Vec<(u32, bool, bool)> = runs
|
||||
.into_iter()
|
||||
.map(|r| (r.count, r.added, r.removed))
|
||||
|
||||
@@ -4,13 +4,28 @@
|
||||
* Provides diff string generation and the replace-mode edit logic
|
||||
* used when not in patch mode.
|
||||
*/
|
||||
import { diffLines, structuredPatchHunks } from "@oh-my-pi/pi-natives";
|
||||
import { diffLines as nativeDiffLines, structuredPatchHunks as nativeStructuredPatchHunks } from "@oh-my-pi/pi-natives";
|
||||
import { diffLines as jsDiffLines, structuredPatch as jsStructuredPatch } from "diff";
|
||||
import { resolveToCwd } from "../tools/path-utils";
|
||||
import { type BlockContextSource, findBlockContextLines } from "../utils/block-context";
|
||||
import { DEFAULT_FUZZY_THRESHOLD, EditMatchError, findMatch } from "./modes/replace";
|
||||
import { adjustIndentation, normalizeToLF, stripBom } from "./normalize";
|
||||
import { readEditFileText } from "./read-file";
|
||||
|
||||
/** Native line diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */
|
||||
function diffLines(oldContent: string, newContent: string) {
|
||||
return oldContent.isWellFormed() && newContent.isWellFormed()
|
||||
? nativeDiffLines(oldContent, newContent)
|
||||
: jsDiffLines(oldContent, newContent);
|
||||
}
|
||||
|
||||
/** Native structured-patch hunks with the same ill-formed-UTF-16 fallback as {@link diffLines}. */
|
||||
function structuredPatchHunks(oldContent: string, newContent: string, context: number) {
|
||||
return oldContent.isWellFormed() && newContent.isWellFormed()
|
||||
? nativeStructuredPatchHunks(oldContent, newContent, context)
|
||||
: jsStructuredPatch("", "", oldContent, newContent, "", "", { context }).hunks;
|
||||
}
|
||||
|
||||
export interface DiffResult {
|
||||
diff: string;
|
||||
firstChangedLine: number | undefined;
|
||||
|
||||
@@ -1,8 +1,16 @@
|
||||
import { diffWords } from "@oh-my-pi/pi-natives";
|
||||
import { diffWords as nativeDiffWords } from "@oh-my-pi/pi-natives";
|
||||
import { DEFAULT_TAB_WIDTH, sanitizeText } from "@oh-my-pi/pi-utils";
|
||||
import { diffWords as jsDiffWords } from "diff";
|
||||
import { getLanguageFromPath, highlightCode, theme } from "../../modes/theme/theme";
|
||||
import { type CodeFrameMarker, formatCodeFrameLine, replaceTabs } from "../../tools/render-utils";
|
||||
|
||||
/** Native word diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */
|
||||
function diffWords(oldContent: string, newContent: string) {
|
||||
return oldContent.isWellFormed() && newContent.isWellFormed()
|
||||
? nativeDiffWords(oldContent, newContent)
|
||||
: jsDiffWords(oldContent, newContent);
|
||||
}
|
||||
|
||||
/** SGR dim on / normal intensity — additive, preserves fg/bg colors. */
|
||||
const DIM = "\x1b[2m";
|
||||
const DIM_OFF = "\x1b[22m";
|
||||
|
||||
@@ -6,7 +6,8 @@
|
||||
* Recovery fails closed when the target changed or became ambiguous. The
|
||||
* patcher then returns a mismatch with fresh context instead of guessing.
|
||||
*/
|
||||
import { diffLineRuns } from "@oh-my-pi/pi-natives";
|
||||
import { diffLineRuns as nativeDiffLineRuns } from "@oh-my-pi/pi-natives";
|
||||
import { diffArrays } from "diff";
|
||||
import { applyEdits } from "./apply";
|
||||
import { RECOVERY_EXTERNAL_WARNING, RECOVERY_LINE_REMAP_WARNING, RECOVERY_SESSION_CHAIN_WARNING } from "./messages";
|
||||
import type { SnapshotStore } from "./snapshots";
|
||||
@@ -44,6 +45,21 @@ function getEditAnchors(edit: Edit): Anchor[] {
|
||||
return edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor" ? [edit.cursor.anchor] : [];
|
||||
}
|
||||
|
||||
/** Native line-run diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */
|
||||
function diffLineRuns(
|
||||
previousText: string,
|
||||
currentText: string,
|
||||
): { count: number; added: boolean; removed: boolean }[] {
|
||||
if (previousText.isWellFormed() && currentText.isWellFormed()) {
|
||||
return nativeDiffLineRuns(previousText, currentText);
|
||||
}
|
||||
return diffArrays(previousText.split("\n"), currentText.split("\n")).map(change => ({
|
||||
count: change.count,
|
||||
added: change.added,
|
||||
removed: change.removed,
|
||||
}));
|
||||
}
|
||||
|
||||
function buildLineMap(previousText: string, currentText: string): Map<number, number> {
|
||||
const changes = diffLineRuns(previousText, currentText);
|
||||
const map = new Map<number, number>();
|
||||
|
||||
Reference in New Issue
Block a user