From 52ad6516ef1a0972a0723178999eb4cdea8e720c Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 23 Jul 2026 01:34:09 +0200 Subject: [PATCH] feat(diff): implemented native UTF-16 diff processing and removed jsdiff - Implemented native UTF-16 text processing in Rust diff module with support for unpaired surrogates. - Removed `similar` crate from Rust workspace and `diff` npm package from coding-agent, hashline, and natives. - Removed jsdiff fallback wrappers and `isWellFormed()` guards from TypeScript diff implementations. - Added comprehensive test suite for native diff functions covering random inputs and edge cases including surrogates and emoji. - Renamed model `codex-auto-review` to `gpt-5.3-codex-spark` with updated pricing and context window. --- Cargo.lock | 1 - bun.lock | 5 +- crates/pi-natives/Cargo.toml | 1 - crates/pi-natives/src/diff.rs | 552 +++++++++++------- packages/catalog/CHANGELOG.md | 10 + packages/catalog/src/models.json | 49 +- packages/coding-agent/CHANGELOG.md | 5 +- packages/coding-agent/package.json | 1 - packages/coding-agent/src/edit/diff.ts | 17 +- .../coding-agent/src/modes/components/diff.ts | 10 +- .../coding-agent/test/tools/edit-diff.test.ts | 8 +- .../test/tools/edit-renderer.test.ts | 7 +- packages/hashline/CHANGELOG.md | 4 + packages/hashline/package.json | 1 - packages/hashline/src/recovery.ts | 18 +- .../test/recovery-session-chain.test.ts | 7 +- packages/metaharness/package.json | 1 + packages/natives/CHANGELOG.md | 7 + packages/natives/bench/diff-results.md | 35 -- packages/natives/bench/diff.ts | 78 --- packages/natives/native/index.d.ts | 2 +- packages/natives/package.json | 3 +- packages/natives/test/diff-parity.test.ts | 192 ------ packages/natives/test/diff.test.ts | 254 ++++++++ .../typescript-edit-benchmark/package.json | 1 + 25 files changed, 658 insertions(+), 611 deletions(-) delete mode 100644 packages/natives/bench/diff-results.md delete mode 100644 packages/natives/bench/diff.ts delete mode 100644 packages/natives/test/diff-parity.test.ts create mode 100644 packages/natives/test/diff.test.ts diff --git a/Cargo.lock b/Cargo.lock index 558ef7026..2a7fe38a3 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3393,7 +3393,6 @@ dependencies = [ "regex", "serde", "serde_json", - "similar 3.1.1", "smallvec", "syntect", "tiktoken-rs", diff --git a/bun.lock b/bun.lock index 17dce7f92..4cde14918 100644 --- a/bun.lock +++ b/bun.lock @@ -104,7 +104,6 @@ "@xterm/headless": "catalog:", "arktype": "catalog:", "chalk": "catalog:", - "diff": "catalog:", "fast-xml-parser": "catalog:", "handlebars": "catalog:", "header-generator": "catalog:", @@ -147,7 +146,6 @@ "version": "17.0.8", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", - "diff": "catalog:", "lru-cache": "catalog:", }, "devDependencies": { @@ -166,6 +164,7 @@ "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", "@oh-my-pi/pi-coding-agent": "catalog:", + "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", "@oh-my-pi/typescript-edit-benchmark": "workspace:*", "clsx": "^2.1.1", @@ -219,7 +218,6 @@ "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", - "diff": "catalog:", }, }, "packages/snapcompact": { @@ -304,6 +302,7 @@ "@oh-my-pi/pi-agent-core": "catalog:", "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-coding-agent": "catalog:", + "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-tui": "catalog:", "@oh-my-pi/pi-utils": "catalog:", "diff": "catalog:", diff --git a/crates/pi-natives/Cargo.toml b/crates/pi-natives/Cargo.toml index f3d02430e..5ba89c14e 100644 --- a/crates/pi-natives/Cargo.toml +++ b/crates/pi-natives/Cargo.toml @@ -45,7 +45,6 @@ rayon.workspace = true regex.workspace = true serde.workspace = true serde_json.workspace = true -similar.workspace = true smallvec.workspace = true syntect.workspace = true tiktoken-rs.workspace = true diff --git a/crates/pi-natives/src/diff.rs b/crates/pi-natives/src/diff.rs index f97e9b82e..c357a7c37 100644 --- a/crates/pi-natives/src/diff.rs +++ b/crates/pi-natives/src/diff.rs @@ -8,6 +8,12 @@ //! (`minDiagonalToConsider` / `maxDiagonalToConsider`), so change coalescing //! matches jsdiff run-for-run rather than merely being "a" minimal diff. //! +//! Everything operates on UTF-16 code units end to end — [`Utf16String`] at +//! the N-API boundary, `&[u16]` internally — which is the exact value space of +//! JS strings. Ill-formed input (unpaired surrogates) is legal content that +//! diffs code-unit-for-code-unit like jsdiff, so callers never need a JS +//! fallback, and no UTF-8 conversion happens in either direction. +//! //! # Example //! ```ignore //! // JS: native.diffLines("a\nb\n", "a\nc\n") @@ -18,32 +24,17 @@ use std::{collections::HashMap, rc::Rc}; -use napi::{JsString, bindgen_prelude::*}; +use napi::bindgen_prelude::*; use napi_derive::napi; -/// Decode a JS string strictly: unpaired surrogates are rejected instead of -/// being replaced with U+FFFD, so callers can fall back to a UTF-16-aware -/// JS diff and both sides keep byte-identical jsdiff semantics. -fn strict_utf16_to_string(text: JsString) -> Result { - let utf16 = text.into_utf16()?; - // `as_slice` includes the trailing NUL terminator; drop it like napi's - // own `as_str` does (a legitimate U+0000 in the content sits before it). - let Some((_, units)) = utf16.as_slice().split_last() else { - return Ok(String::new()); - }; - String::from_utf16(units).map_err(|_| { - Error::new( - Status::InvalidArg, - "ill-formed UTF-16 input (unpaired surrogate); caller must fall back to a JS diff", - ) - }) -} +/// UTF-16 code unit for `\n`. +const LF: u16 = 0x000a; /// One jsdiff change object: a run of added, removed, or common tokens. #[napi(object)] pub struct DiffChange { /// Joined token text for this run (lines keep their `\n` terminators). - pub value: String, + pub value: Utf16String, /// Number of tokens in this run. pub count: u32, /// True when this run exists only in the new text. @@ -76,7 +67,7 @@ pub struct PatchHunk { pub new_lines: u32, /// Hunk body: `+`/`-`/` `-prefixed lines without trailing newlines, plus /// `\ No newline at end of file` markers where applicable. - pub lines: Vec, + pub lines: Vec, } // ═══════════════════════════════════════════════════════════════════════════ @@ -257,14 +248,15 @@ fn myers_diff(old: &[u32], new: &[u32]) -> Vec { unreachable!("Myers diff terminates within oldLen + newLen edits") } -/// Intern each token as a dense id under exact string equality, so the Myers -/// core compares `u32`s instead of re-hashing strings per probe. -fn intern_exact<'a>(old_tokens: &[&'a str], new_tokens: &[&'a str]) -> (Vec, Vec) { - fn assign<'a>(ids: &mut HashMap<&'a str, u32>, token: &'a str) -> u32 { +/// Intern each token as a dense id under exact code-unit equality, so the +/// Myers core compares `u32`s instead of re-hashing slices per probe. +fn intern_exact<'a>(old_tokens: &[&'a [u16]], new_tokens: &[&'a [u16]]) -> (Vec, Vec) { + fn assign<'a>(ids: &mut HashMap<&'a [u16], u32>, token: &'a [u16]) -> u32 { let next = ids.len() as u32; *ids.entry(token).or_insert(next) } - let mut ids: HashMap<&'a str, u32> = HashMap::with_capacity(old_tokens.len() + new_tokens.len()); + let mut ids: HashMap<&'a [u16], u32> = + HashMap::with_capacity(old_tokens.len() + new_tokens.len()); let old_ids = old_tokens .iter() .map(|token| assign(&mut ids, token)) @@ -281,9 +273,9 @@ fn intern_exact<'a>(old_tokens: &[&'a str], new_tokens: &[&'a str]) -> (Vec /// `buildValues` with `useLongestToken == false`. fn build_changes( runs: &[Run], - old_tokens: &[&str], - new_tokens: &[&str], - join: impl Fn(&[&str]) -> String, + old_tokens: &[&[u16]], + new_tokens: &[&[u16]], + join: impl Fn(&[&[u16]]) -> Vec, ) -> Vec { let mut old_pos = 0usize; let mut new_pos = 0usize; @@ -302,7 +294,12 @@ fn build_changes( } value }; - DiffChange { value, count: run.count as u32, added: run.added, removed: run.removed } + DiffChange { + value: value.into(), + count: run.count as u32, + added: run.added, + removed: run.removed, + } }) .collect() } @@ -314,44 +311,53 @@ fn build_changes( /// jsdiff line tokenization under default options: each token is a line /// including its `\n` (or `\r\n`) terminator; a final line without a newline /// is kept as-is; a lone `\r` never terminates a line. -fn line_tokens(text: &str) -> Vec<&str> { - text.split_inclusive('\n').collect() +fn line_tokens(text: &[u16]) -> Vec<&[u16]> { + text.split_inclusive(|&unit| unit == LF).collect() } -fn diff_line_tokens<'a>(old_tokens: &[&'a str], new_tokens: &[&'a str]) -> Vec { +fn diff_line_tokens(old_tokens: &[&[u16]], new_tokens: &[&[u16]]) -> Vec { let (old_ids, new_ids) = intern_exact(old_tokens, new_tokens); myers_diff(&old_ids, &new_ids) } +/// Concatenate token slices (jsdiff line `join`). +fn concat_tokens(tokens: &[&[u16]]) -> Vec { + let mut out = Vec::with_capacity(tokens.iter().map(|token| token.len()).sum()); + for token in tokens { + out.extend_from_slice(token); + } + out +} + /// Line diff with jsdiff `diffLines(oldText, newText)` semantics (default /// options). Change values keep line terminators, and common runs are joined /// from the new text. #[napi] -pub fn diff_lines(old_text: JsString, new_text: JsString) -> Result> { - Ok(diff_lines_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?)) +pub fn diff_lines(old_text: Utf16String, new_text: Utf16String) -> Vec { + diff_lines_impl(&old_text, &new_text) } -fn diff_lines_impl(old_text: &str, new_text: &str) -> Vec { +fn diff_lines_impl(old_text: &[u16], new_text: &[u16]) -> Vec { let old_tokens = line_tokens(old_text); let new_tokens = line_tokens(new_text); let runs = diff_line_tokens(&old_tokens, &new_tokens); - build_changes(&runs, &old_tokens, &new_tokens, |tokens| tokens.concat()) + build_changes(&runs, &old_tokens, &new_tokens, concat_tokens) } /// Diff `oldText.split("\n")` against `newText.split("\n")` with jsdiff -/// `diffArrays` semantics (exact string equality, empty lines preserved), +/// `diffArrays` semantics (exact code-unit equality, empty lines preserved), /// returning only run lengths. /// /// Callers that map line numbers — like hashline recovery — need the counts, /// not another copy of the text. #[napi] -pub fn diff_line_runs(old_text: JsString, new_text: JsString) -> Result> { - Ok(diff_line_runs_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?)) +pub fn diff_line_runs(old_text: Utf16String, new_text: Utf16String) -> Vec { + diff_line_runs_impl(&old_text, &new_text) } -fn diff_line_runs_impl(old_text: &str, new_text: &str) -> Vec { - let old_tokens: Vec<&str> = old_text.split('\n').collect(); - let new_tokens: Vec<&str> = new_text.split('\n').collect(); +fn diff_line_runs_impl(old_text: &[u16], new_text: &[u16]) -> Vec { + let old_tokens: Vec<&[u16]> = old_text.split(|&unit| unit == LF).collect(); + let new_tokens: Vec<&[u16]> = new_text.split(|&unit| unit == LF).collect(); let (old_ids, new_ids) = intern_exact(&old_tokens, &new_tokens); myers_diff(&old_ids, &new_ids) .into_iter() @@ -363,25 +369,34 @@ fn diff_line_runs_impl(old_text: &str, new_text: &str) -> Vec { // Structured patch (port of jsdiff patch/create.ts hunk builder) // ═══════════════════════════════════════════════════════════════════════════ +/// Prepend a `+`/`-`/` ` marker to a line's code units. +fn prefixed_line(prefix: u8, line: &[u16]) -> Vec { + let mut out = Vec::with_capacity(1 + line.len()); + out.push(u16::from(prefix)); + out.extend_from_slice(line); + out +} + +/// `\ No newline at end of file`, as UTF-16 code units. +fn no_newline_marker() -> Vec { + "\\ No newline at end of file".encode_utf16().collect() +} + /// Unified-diff hunks with jsdiff /// `structuredPatch(_, _, oldText, newText, _, _, { context }).hunks` /// semantics. `context` defaults to 4 like jsdiff. #[napi] pub fn structured_patch_hunks( - old_text: JsString, - new_text: JsString, + old_text: Utf16String, + new_text: Utf16String, context: Option, -) -> Result> { - Ok(structured_patch_hunks_impl( - &strict_utf16_to_string(old_text)?, - &strict_utf16_to_string(new_text)?, - context, - )) +) -> Vec { + structured_patch_hunks_impl(&old_text, &new_text, context) } fn structured_patch_hunks_impl( - old_text: &str, - new_text: &str, + old_text: &[u16], + new_text: &[u16], context: Option, ) -> Vec { let context = context.map_or(4usize, |value| value as usize); @@ -394,13 +409,13 @@ fn structured_patch_hunks_impl( struct ChangeLines<'a> { added: bool, removed: bool, - lines: &'a [&'a str], + lines: &'a [&'a [u16]], } let mut list: Vec = Vec::with_capacity(runs.len() + 1); let mut old_pos = 0usize; let mut new_pos = 0usize; for run in &runs { - let lines: &[&str] = if run.removed { + let lines: &[&[u16]] = if run.removed { let slice = &old_tokens[old_pos..old_pos + run.count]; old_pos += run.count; slice @@ -416,10 +431,19 @@ fn structured_patch_hunks_impl( } list.push(ChangeLines { added: false, removed: false, lines: &[] }); - let mut hunks: Vec = Vec::new(); + // Hunk skeleton before the trailing-newline post-pass; lines stay `Vec` + // so the pass below can pop terminators in place. + struct RawHunk { + old_start: usize, + old_lines: usize, + new_start: usize, + new_lines: usize, + lines: Vec>, + } + let mut hunks: Vec = Vec::new(); let mut old_range_start = 0usize; let mut new_range_start = 0usize; - let mut cur_range: Vec = Vec::new(); + let mut cur_range: Vec> = Vec::new(); let mut old_line = 1usize; let mut new_line = 1usize; for i in 0..list.len() { @@ -435,15 +459,15 @@ fn structured_patch_hunks_impl( let take = prev_lines.len().min(context); cur_range = prev_lines[prev_lines.len() - take..] .iter() - .map(|line| format!(" {line}")) + .map(|line| prefixed_line(b' ', line)) .collect(); old_range_start -= cur_range.len(); new_range_start -= cur_range.len(); } } - let marker = if current.added { '+' } else { '-' }; + let marker = if current.added { b'+' } else { b'-' }; for line in current.lines { - cur_range.push(format!("{marker}{line}")); + cur_range.push(prefixed_line(marker, line)); } if current.added { new_line += current.lines.len(); @@ -455,19 +479,19 @@ fn structured_patch_hunks_impl( if current.lines.len() <= context * 2 && i + 2 < list.len() { // Common run small enough to join adjacent hunks. for line in current.lines { - cur_range.push(format!(" {line}")); + cur_range.push(prefixed_line(b' ', line)); } } else { // Close the hunk with leading context. let context_size = current.lines.len().min(context); for line in ¤t.lines[..context_size] { - cur_range.push(format!(" {line}")); + cur_range.push(prefixed_line(b' ', line)); } - hunks.push(PatchHunk { - old_start: old_range_start as u32, - old_lines: (old_line - old_range_start + context_size) as u32, - new_start: new_range_start as u32, - new_lines: (new_line - new_range_start + context_size) as u32, + hunks.push(RawHunk { + old_start: old_range_start, + old_lines: old_line - old_range_start + context_size, + new_start: new_range_start, + new_lines: new_line - new_range_start + context_size, lines: std::mem::take(&mut cur_range), }); old_range_start = 0; @@ -483,94 +507,148 @@ fn structured_patch_hunks_impl( for hunk in &mut hunks { let mut i = 0; while i < hunk.lines.len() { - if hunk.lines[i].ends_with('\n') { + if hunk.lines[i].last() == Some(&LF) { hunk.lines[i].pop(); } else { - hunk - .lines - .insert(i + 1, "\\ No newline at end of file".to_string()); + hunk.lines.insert(i + 1, no_newline_marker()); i += 1; } i += 1; } } hunks + .into_iter() + .map(|hunk| PatchHunk { + old_start: hunk.old_start as u32, + old_lines: hunk.old_lines as u32, + new_start: hunk.new_start as u32, + new_lines: hunk.new_lines as u32, + lines: hunk.lines.into_iter().map(Utf16String::from).collect(), + }) + .collect() } // ═══════════════════════════════════════════════════════════════════════════ // Word diff (port of jsdiff word.ts, default options) // ═══════════════════════════════════════════════════════════════════════════ -/// jsdiff's `extendedWordChars` class: Latin-script word characters. -const fn is_word_char(c: char) -> bool { - matches!(c, - 'a'..='z' - | 'A'..='Z' - | '0'..='9' - | '_' - | '\u{ad}' - | '\u{c0}'..='\u{d6}' - | '\u{d8}'..='\u{f6}' - | '\u{f8}'..='\u{2c6}' - | '\u{2c8}'..='\u{2d7}' - | '\u{2de}'..='\u{2ff}' - | '\u{1e00}'..='\u{1eff}') +/// jsdiff's `extendedWordChars` class: Latin-script word characters. Takes a +/// code point so astral input classifies like a JS regex with the `u` flag — +/// never a word character (every member is BMP). +const fn is_word_char(cp: u32) -> bool { + matches!(cp, + 0x30..=0x39 // 0-9 + | 0x41..=0x5A // A-Z + | 0x5F // _ + | 0x61..=0x7A // a-z + | 0xAD + | 0xC0..=0xD6 + | 0xD8..=0xF6 + | 0xF8..=0x2C6 + | 0x2C8..=0x2D7 + | 0x2DE..=0x2FF + | 0x1E00..=0x1EFF) } /// JavaScript's `\s` / `String.prototype.trim` whitespace set (`WhiteSpace` + -/// `LineTerminator` productions). Every member is a single UTF-16 code unit, so -/// char-level scans here match jsdiff's code-unit-level scans exactly. -const fn is_js_whitespace(c: char) -> bool { +/// `LineTerminator` productions). Every member is a single UTF-16 code unit, +/// so unit-level scans here match jsdiff's code-unit-level scans exactly. +const fn is_js_whitespace(cp: u32) -> bool { matches!( - c, - '\t' | '\n' | '\u{b}' | '\u{c}' | '\r' | ' ' | '\u{a0}' | '\u{1680}' | '\u{2000}' - ..='\u{200a}' - | '\u{2028}' - | '\u{2029}' - | '\u{202f}' - | '\u{205f}' - | '\u{3000}' - | '\u{feff}' + cp, + 0x09 | 0x0a | 0x0b | 0x0c | 0x0d | 0x20 | 0xa0 | 0x1680 | 0x2000 + ..=0x200a | 0x2028 | 0x2029 | 0x202f | 0x205f | 0x3000 | 0xfeff ) } -fn leading_ws(s: &str) -> &str { - &s[..s.len() - s.trim_start_matches(is_js_whitespace).len()] +const fn is_ws_unit(unit: u16) -> bool { + is_js_whitespace(unit as u32) } -fn trailing_ws(s: &str) -> &str { - &s[s.trim_end_matches(is_js_whitespace).len()..] +fn trim_leading_ws(s: &[u16]) -> &[u16] { + let start = s + .iter() + .position(|&unit| !is_ws_unit(unit)) + .unwrap_or(s.len()); + &s[start..] } -fn js_trim(s: &str) -> &str { - s.trim_matches(is_js_whitespace) +fn trim_trailing_ws(s: &[u16]) -> &[u16] { + let end = s + .iter() + .rposition(|&unit| !is_ws_unit(unit)) + .map_or(0, |i| i + 1); + &s[..end] +} + +fn leading_ws(s: &[u16]) -> &[u16] { + &s[..s.len() - trim_leading_ws(s).len()] +} + +fn trailing_ws(s: &[u16]) -> &[u16] { + &s[trim_trailing_ws(s).len()..] +} + +fn js_trim(s: &[u16]) -> &[u16] { + trim_trailing_ws(trim_leading_ws(s)) +} + +/// Iterator over `(start, code_point, unit_len)` that pairs surrogates and +/// passes unpaired surrogates through as their own code points, exactly like +/// JS regex scanning under the `u` flag. +struct CodePoints<'a> { + text: &'a [u16], + pos: usize, +} + +impl Iterator for CodePoints<'_> { + type Item = (usize, u32, usize); + + fn next(&mut self) -> Option { + let &unit = self.text.get(self.pos)?; + let start = self.pos; + if matches!(unit, 0xd800..=0xdbff) + && let Some(&low) = self.text.get(start + 1) + && matches!(low, 0xdc00..=0xdfff) + { + self.pos += 2; + let cp = 0x10000 + ((u32::from(unit & 0x3ff) << 10) | u32::from(low & 0x3ff)); + return Some((start, cp, 2)); + } + self.pos += 1; + Some((start, u32::from(unit), 1)) + } +} + +const fn code_points(text: &[u16]) -> CodePoints<'_> { + CodePoints { text, pos: 0 } } /// Raw regex-equivalent scan: word runs, whitespace runs, or single other /// code points (jsdiff `tokenizeIncludingWhitespace` with the `u` flag). -fn word_parts(text: &str) -> Vec<&str> { +fn word_parts(text: &[u16]) -> Vec<&[u16]> { let mut parts = Vec::new(); - let mut iter = text.char_indices().peekable(); - while let Some((start, c)) = iter.next() { - let class = if is_word_char(c) { + let mut iter = code_points(text).peekable(); + while let Some((start, cp, len)) = iter.next() { + let class = if is_word_char(cp) { 1u8 - } else if is_js_whitespace(c) { + } else if is_js_whitespace(cp) { 2u8 } else { 0u8 }; - let mut end = start + c.len_utf8(); + let mut end = start + len; if class != 0 { - while let Some(&(next_start, next)) = iter.peek() { + while let Some(&(_, next_cp, next_len)) = iter.peek() { let same = if class == 1 { - is_word_char(next) + is_word_char(next_cp) } else { - is_js_whitespace(next) + is_js_whitespace(next_cp) }; if !same { break; } - end = next_start + next.len_utf8(); + end += next_len; iter.next(); } } @@ -581,32 +659,35 @@ fn word_parts(text: &str) -> Vec<&str> { /// jsdiff `WordDiff.tokenize`: stitch whitespace runs onto adjacent word or /// punctuation parts, duplicating interior whitespace into both neighbors. -fn word_tokens(text: &str) -> Vec { +fn word_tokens(text: &[u16]) -> Vec> { let parts = word_parts(text); - let mut tokens: Vec = Vec::with_capacity(parts.len()); - let mut prev_part: Option<&str> = None; + let mut tokens: Vec> = Vec::with_capacity(parts.len()); + let mut prev_part: Option<&[u16]> = None; for part in parts { - let part_is_ws = part.chars().next().is_some_and(is_js_whitespace); + let part_is_ws = part.first().is_some_and(|&unit| is_ws_unit(unit)); if part_is_ws { if prev_part.is_none() { - tokens.push(part.to_string()); + tokens.push(part.to_vec()); } else { let last = tokens .last_mut() .expect("tokens non-empty after first part"); - last.push_str(part); + last.extend_from_slice(part); } } else if let Some(prev) = - prev_part.filter(|p| p.chars().next().is_some_and(is_js_whitespace)) + prev_part.filter(|p| p.first().is_some_and(|&unit| is_ws_unit(unit))) { - if tokens.last().is_some_and(|last| last == prev) { + if tokens.last().is_some_and(|last| last.as_slice() == prev) { let last = tokens.last_mut().expect("checked non-empty"); - last.push_str(part); + last.extend_from_slice(part); } else { - tokens.push(format!("{prev}{part}")); + let mut token = Vec::with_capacity(prev.len() + part.len()); + token.extend_from_slice(prev); + token.extend_from_slice(part); + tokens.push(token); } } else { - tokens.push(part.to_string()); + tokens.push(part.to_vec()); } prev_part = Some(part); } @@ -615,102 +696,98 @@ fn word_tokens(text: &str) -> Vec { /// jsdiff `WordDiff.join`: concatenate, stripping leading whitespace from /// every token after the first. -fn word_join(tokens: &[&str]) -> String { - let mut out = String::new(); +fn word_join(tokens: &[&[u16]]) -> Vec { + let mut out = Vec::new(); for (i, token) in tokens.iter().enumerate() { if i == 0 { - out.push_str(token); + out.extend_from_slice(token); } else { - out.push_str(token.trim_start_matches(is_js_whitespace)); + out.extend_from_slice(trim_leading_ws(token)); } } out } -fn longest_common_prefix<'a>(a: &'a str, b: &str) -> &'a str { - let mut end = 0; - for (ca, cb) in a.chars().zip(b.chars()) { - if ca != cb { - break; - } - end += ca.len_utf8(); - } - &a[..end] +fn longest_common_prefix<'a>(a: &'a [u16], b: &[u16]) -> &'a [u16] { + let len = a.iter().zip(b).take_while(|(x, y)| x == y).count(); + &a[..len] } -fn longest_common_suffix<'a>(a: &'a str, b: &str) -> &'a str { - let mut start = a.len(); - for (ca, cb) in a.chars().rev().zip(b.chars().rev()) { - if ca != cb { - break; - } - start -= ca.len_utf8(); - } - &a[start..] +fn longest_common_suffix<'a>(a: &'a [u16], b: &[u16]) -> &'a [u16] { + let len = a + .iter() + .rev() + .zip(b.iter().rev()) + .take_while(|(x, y)| x == y) + .count(); + &a[a.len() - len..] } -fn remove_prefix(s: &str, prefix: &str) -> String { +fn remove_prefix(s: &[u16], prefix: &[u16]) -> Vec { s.strip_prefix(prefix) .expect("value must start with recorded prefix") - .to_string() + .to_vec() } -fn remove_suffix(s: &str, suffix: &str) -> String { +fn remove_suffix(s: &[u16], suffix: &[u16]) -> Vec { s.strip_suffix(suffix) .expect("value must end with recorded suffix") - .to_string() + .to_vec() } -fn replace_prefix(s: &str, old_prefix: &str, new_prefix: &str) -> String { +fn replace_prefix(s: &[u16], old_prefix: &[u16], new_prefix: &[u16]) -> Vec { let rest = s .strip_prefix(old_prefix) .expect("value must start with recorded prefix"); - format!("{new_prefix}{rest}") + let mut out = Vec::with_capacity(new_prefix.len() + rest.len()); + out.extend_from_slice(new_prefix); + out.extend_from_slice(rest); + out } -fn replace_suffix(s: &str, old_suffix: &str, new_suffix: &str) -> String { +fn replace_suffix(s: &[u16], old_suffix: &[u16], new_suffix: &[u16]) -> Vec { let rest = s .strip_suffix(old_suffix) .expect("value must end with recorded suffix"); - format!("{rest}{new_suffix}") + let mut out = Vec::with_capacity(rest.len() + new_suffix.len()); + out.extend_from_slice(rest); + out.extend_from_slice(new_suffix); + out } /// jsdiff `maximumOverlap`: the longest prefix of `b` that is also a suffix -/// of `a`, via the KMP failure function. -fn maximum_overlap<'a>(a: &str, b: &'a str) -> &'a str { - let a_chars: Vec = a.chars().collect(); - let b_chars: Vec = b.chars().collect(); - let start_a = a_chars.len().saturating_sub(b_chars.len()); - let end_b = b_chars.len().min(a_chars.len()); +/// of `a`, via the KMP failure function over code units. +fn maximum_overlap<'a>(a: &[u16], b: &'a [u16]) -> &'a [u16] { + let start_a = a.len().saturating_sub(b.len()); + let end_b = b.len().min(a.len()); if end_b == 0 { - return ""; + return &[]; } let mut map = vec![0usize; end_b]; let mut k = 0usize; for j in 1..end_b { - if b_chars[j] == b_chars[k] { + if b[j] == b[k] { map[j] = map[k]; } else { map[j] = k; } - while k > 0 && b_chars[j] != b_chars[k] { + while k > 0 && b[j] != b[k] { k = map[k]; } - if b_chars[j] == b_chars[k] { + if b[j] == b[k] { k += 1; } } k = 0; - for &c in &a_chars[start_a..] { - while k > 0 && c != b_chars[k] { + for &unit in &a[start_a..] { + while k > 0 && unit != b[k] { k = map[k]; } - if c == b_chars[k] { + if unit == b[k] { k += 1; } } - let byte_end: usize = b_chars[..k].iter().map(|c| c.len_utf8()).sum(); - &b[..byte_end] + &b[..k] } /// jsdiff `dedupeWhitespaceInChangeObjects` (no segmenter): trim whitespace @@ -724,62 +801,62 @@ fn dedupe_whitespace( ) { match (deletion, insertion) { (Some(del), Some(ins)) => { - let old_ws_prefix = leading_ws(&changes[del].value).to_string(); - let old_ws_suffix = trailing_ws(&changes[del].value).to_string(); - let new_ws_prefix = leading_ws(&changes[ins].value).to_string(); - let new_ws_suffix = trailing_ws(&changes[ins].value).to_string(); + let old_ws_prefix = leading_ws(&changes[del].value).to_vec(); + let old_ws_suffix = trailing_ws(&changes[del].value).to_vec(); + let new_ws_prefix = leading_ws(&changes[ins].value).to_vec(); + let new_ws_suffix = trailing_ws(&changes[ins].value).to_vec(); if let Some(start) = start_keep { - let common_ws_prefix = - longest_common_prefix(&old_ws_prefix, &new_ws_prefix).to_string(); + let common_ws_prefix = longest_common_prefix(&old_ws_prefix, &new_ws_prefix).to_vec(); changes[start].value = - replace_suffix(&changes[start].value, &new_ws_prefix, &common_ws_prefix); - changes[del].value = remove_prefix(&changes[del].value, &common_ws_prefix); - changes[ins].value = remove_prefix(&changes[ins].value, &common_ws_prefix); + replace_suffix(&changes[start].value, &new_ws_prefix, &common_ws_prefix).into(); + changes[del].value = remove_prefix(&changes[del].value, &common_ws_prefix).into(); + changes[ins].value = remove_prefix(&changes[ins].value, &common_ws_prefix).into(); } if let Some(end) = end_keep { - let common_ws_suffix = - longest_common_suffix(&old_ws_suffix, &new_ws_suffix).to_string(); + let common_ws_suffix = longest_common_suffix(&old_ws_suffix, &new_ws_suffix).to_vec(); changes[end].value = - replace_prefix(&changes[end].value, &new_ws_suffix, &common_ws_suffix); - changes[del].value = remove_suffix(&changes[del].value, &common_ws_suffix); - changes[ins].value = remove_suffix(&changes[ins].value, &common_ws_suffix); + replace_prefix(&changes[end].value, &new_ws_suffix, &common_ws_suffix).into(); + changes[del].value = remove_suffix(&changes[del].value, &common_ws_suffix).into(); + changes[ins].value = remove_suffix(&changes[ins].value, &common_ws_suffix).into(); } }, (None, Some(ins)) => { if start_keep.is_some() { - let ws = leading_ws(&changes[ins].value).to_string(); - changes[ins].value = changes[ins].value[ws.len()..].to_string(); + let ws_len = leading_ws(&changes[ins].value).len(); + changes[ins].value = changes[ins].value[ws_len..].to_vec().into(); } if let Some(end) = end_keep { - let ws = leading_ws(&changes[end].value).to_string(); - changes[end].value = changes[end].value[ws.len()..].to_string(); + let ws_len = leading_ws(&changes[end].value).len(); + changes[end].value = changes[end].value[ws_len..].to_vec().into(); } }, (Some(del), None) => match (start_keep, end_keep) { (Some(start), Some(end)) => { - let new_ws_full = leading_ws(&changes[end].value).to_string(); - let del_ws_start = leading_ws(&changes[del].value).to_string(); - let del_ws_end = trailing_ws(&changes[del].value).to_string(); - let new_ws_start = longest_common_prefix(&new_ws_full, &del_ws_start).to_string(); - changes[del].value = remove_prefix(&changes[del].value, &new_ws_start); + let new_ws_full = leading_ws(&changes[end].value).to_vec(); + let del_ws_start = leading_ws(&changes[del].value).to_vec(); + let del_ws_end = trailing_ws(&changes[del].value).to_vec(); + let new_ws_start = longest_common_prefix(&new_ws_full, &del_ws_start).to_vec(); + changes[del].value = remove_prefix(&changes[del].value, &new_ws_start).into(); let new_ws_end = - longest_common_suffix(&new_ws_full[new_ws_start.len()..], &del_ws_end).to_string(); - changes[del].value = remove_suffix(&changes[del].value, &new_ws_end); - changes[end].value = replace_prefix(&changes[end].value, &new_ws_full, &new_ws_end); + longest_common_suffix(&new_ws_full[new_ws_start.len()..], &del_ws_end).to_vec(); + changes[del].value = remove_suffix(&changes[del].value, &new_ws_end).into(); + changes[end].value = + replace_prefix(&changes[end].value, &new_ws_full, &new_ws_end).into(); let start_ws = &new_ws_full[..new_ws_full.len() - new_ws_end.len()]; - changes[start].value = replace_suffix(&changes[start].value, &new_ws_full, start_ws); + changes[start].value = + replace_suffix(&changes[start].value, &new_ws_full, start_ws).into(); }, (None, Some(end)) => { - let end_keep_ws_prefix = leading_ws(&changes[end].value).to_string(); - let deletion_ws_suffix = trailing_ws(&changes[del].value).to_string(); - let overlap = maximum_overlap(&deletion_ws_suffix, &end_keep_ws_prefix).to_string(); - changes[del].value = remove_suffix(&changes[del].value, &overlap); + let end_keep_ws_prefix = leading_ws(&changes[end].value).to_vec(); + let deletion_ws_suffix = trailing_ws(&changes[del].value).to_vec(); + let overlap = maximum_overlap(&deletion_ws_suffix, &end_keep_ws_prefix).to_vec(); + changes[del].value = remove_suffix(&changes[del].value, &overlap).into(); }, (Some(start), None) => { - let start_keep_ws_suffix = trailing_ws(&changes[start].value).to_string(); - let deletion_ws_prefix = leading_ws(&changes[del].value).to_string(); - let overlap = maximum_overlap(&start_keep_ws_suffix, &deletion_ws_prefix).to_string(); - changes[del].value = remove_prefix(&changes[del].value, &overlap); + let start_keep_ws_suffix = trailing_ws(&changes[start].value).to_vec(); + let deletion_ws_prefix = leading_ws(&changes[del].value).to_vec(); + let overlap = maximum_overlap(&start_keep_ws_suffix, &deletion_ws_prefix).to_vec(); + changes[del].value = remove_prefix(&changes[del].value, &overlap).into(); }, (None, None) => {}, }, @@ -817,18 +894,18 @@ fn word_post_process(changes: &mut [DiffChange]) { /// Tokens carry surrounding whitespace, equality ignores it, and the /// post-pass dedupes whitespace across change boundaries. #[napi] -pub fn diff_words(old_text: JsString, new_text: JsString) -> Result> { - Ok(diff_words_impl(&strict_utf16_to_string(old_text)?, &strict_utf16_to_string(new_text)?)) +pub fn diff_words(old_text: Utf16String, new_text: Utf16String) -> Vec { + diff_words_impl(&old_text, &new_text) } -fn diff_words_impl(old_text: &str, new_text: &str) -> Vec { +fn diff_words_impl(old_text: &[u16], new_text: &[u16]) -> Vec { let old_tokens = word_tokens(old_text); let new_tokens = word_tokens(new_text); - let old_refs: Vec<&str> = old_tokens.iter().map(String::as_str).collect(); - let new_refs: Vec<&str> = new_tokens.iter().map(String::as_str).collect(); + let old_refs: Vec<&[u16]> = old_tokens.iter().map(Vec::as_slice).collect(); + let new_refs: Vec<&[u16]> = new_tokens.iter().map(Vec::as_slice).collect(); // Equality is whitespace-insensitive: intern by trimmed text. - let old_keys: Vec<&str> = old_refs.iter().map(|token| js_trim(token)).collect(); - let new_keys: Vec<&str> = new_refs.iter().map(|token| js_trim(token)).collect(); + let old_keys: Vec<&[u16]> = old_refs.iter().map(|token| js_trim(token)).collect(); + let new_keys: Vec<&[u16]> = new_refs.iter().map(|token| js_trim(token)).collect(); let (old_ids, new_ids) = intern_exact(&old_keys, &new_keys); let runs = myers_diff(&old_ids, &new_ids); let mut changes = build_changes(&runs, &old_refs, &new_refs, word_join); @@ -840,10 +917,14 @@ fn diff_words_impl(old_text: &str, new_text: &str) -> Vec { mod tests { use super::*; + fn u16s(text: &str) -> Vec { + text.encode_utf16().collect() + } + fn lines(old: &str, new: &str) -> Vec<(String, bool, bool)> { - diff_lines_impl(old, new) + diff_lines_impl(&u16s(old), &u16s(new)) .into_iter() - .map(|c| (c.value, c.added, c.removed)) + .map(|c| (String::from_utf16(&c.value).unwrap(), c.added, c.removed)) .collect() } @@ -868,9 +949,14 @@ mod tests { #[test] fn structured_patch_marks_missing_eof_newline() { - let hunks = structured_patch_hunks_impl("a\nb", "a\nc", Some(3)); + let hunks = structured_patch_hunks_impl(&u16s("a\nb"), &u16s("a\nc"), Some(3)); assert_eq!(hunks.len(), 1); - assert_eq!(hunks[0].lines, vec![ + let body: Vec = hunks[0] + .lines + .iter() + .map(|line| String::from_utf16(line).unwrap()) + .collect(); + assert_eq!(body, vec![ " a", "-b", "\\ No newline at end of file", @@ -882,10 +968,10 @@ mod tests { #[test] fn word_diff_dedupes_boundary_whitespace() { // jsdiff's documented example 2: K:'foo ' D:'bar' I:'qux' K:' baz'. - let changes = diff_words_impl("foo bar baz", "foo qux baz"); + let changes = diff_words_impl(&u16s("foo bar baz"), &u16s("foo qux baz")); let shaped: Vec<(String, bool, bool)> = changes .into_iter() - .map(|c| (c.value, c.added, c.removed)) + .map(|c| (String::from_utf16(&c.value).unwrap(), c.added, c.removed)) .collect(); assert_eq!(shaped, vec![ ("foo ".into(), false, false), @@ -897,11 +983,43 @@ mod tests { #[test] fn line_runs_preserve_empty_lines() { - let runs = diff_line_runs_impl("a\n\nb", "a\n\nc"); + let runs = diff_line_runs_impl(&u16s("a\n\nb"), &u16s("a\n\nc")); let shaped: Vec<(u32, bool, bool)> = runs .into_iter() .map(|r| (r.count, r.added, r.removed)) .collect(); assert_eq!(shaped, vec![(2, false, false), (1, false, true), (1, true, false)]); } + + #[test] + fn unpaired_surrogates_diff_as_distinct_content() { + // Lone surrogates are legal JS string content; they must compare by + // code unit instead of failing (or lossily surviving) a UTF-8 round + // trip. + let old = [0x61, 0xd800, LF]; + let new = [0x61, 0xd801, LF]; + let shaped: Vec<(Vec, bool, bool)> = diff_lines_impl(&old, &new) + .into_iter() + .map(|c| (c.value.to_vec(), c.added, c.removed)) + .collect(); + assert_eq!(shaped, vec![(old.to_vec(), false, true), (new.to_vec(), true, false)]); + } + + #[test] + fn word_scan_keeps_lone_surrogate_before_astral_pair_separate() { + // "\u{D800}🚀" is a lone high surrogate directly followed by a valid + // pair; the `u`-flag scan must yield two "other" tokens, so replacing + // only the rocket leaves the lone surrogate as common content. + let old: Vec = [0xd800, 0xd83d, 0xde80].to_vec(); // "\u{D800}🚀" + let new: Vec = [0xd800, 0x78].to_vec(); // "\u{D800}x" + let shaped: Vec<(Vec, bool, bool)> = diff_words_impl(&old, &new) + .into_iter() + .map(|c| (c.value.to_vec(), c.added, c.removed)) + .collect(); + assert_eq!(shaped, vec![ + (vec![0xd800], false, false), + (vec![0xd83d, 0xde80], false, true), + (vec![0x78], true, false), + ]); + } } diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 14522ef6e..54026822e 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,16 @@ ## [Unreleased] +### Changed + +- Renamed `codex-auto-review` model to `GPT-5.3 Codex Spark` with updated pricing and capabilities +- Removed image input support from GPT-5.3 Codex Spark (text-only) +- Reduced GPT-5.3 Codex Spark context window from 272K to 128K tokens +- Changed GPT-5.3 Codex Spark thinking efforts from `["minimal", "low", "medium", "high", "xhigh"]` to `["low", "medium", "high", "xhigh"]` +- Updated pricing for multiple AI models across providers (costs adjusted in models.json) +- Reduced max output tokens for an unspecified model from 16384 to 8192 +- Added image input support to Venice AI text model + ## [17.0.8] - 2026-07-22 ### Added diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 7ff298032..24eb18fe9 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -62703,21 +62703,20 @@ } }, "openai-codex": { - "codex-auto-review": { - "id": "codex-auto-review", - "name": "Codex Auto Review", + "gpt-5.3-codex-spark": { + "id": "gpt-5.3-codex-spark", + "name": "GPT-5.3 Codex Spark", "api": "openai-codex-responses", "provider": "openai-codex", "baseUrl": "https://chatgpt.com/backend-api", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.75, + "output": 14, + "cacheRead": 0.175, "cacheWrite": 0 }, "remoteCompaction": { @@ -62725,20 +62724,21 @@ "api": "openai-codex-responses", "v2StreamingEnabled": true }, - "contextWindow": 272000, + "contextWindow": 128000, "maxTokens": 128000, "preferWebsockets": true, - "priority": 43, + "priority": 26, + "applyPatchToolType": "freeform", "thinking": { "mode": "effort", "efforts": [ - "minimal", "low", "medium", "high", "xhigh" ] - } + }, + "contextPromotionTarget": "openai-codex/gpt-5.5" }, "gpt-5.4": { "id": "gpt-5.4", @@ -69138,8 +69138,8 @@ "text" ], "cost": { - "input": 0.3, - "output": 1.2, + "input": 0.255, + "output": 1.02, "cacheRead": 0.03, "cacheWrite": 0 }, @@ -72905,13 +72905,13 @@ "text" ], "cost": { - "input": 0.12, - "output": 0.24, + "input": 0.22749999999999998, + "output": 0.9099999999999999, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 16384, + "maxTokens": 8192, "thinking": { "mode": "effort", "efforts": [ @@ -75480,9 +75480,9 @@ "text" ], "cost": { - "input": 0.798, - "output": 2.508, - "cacheRead": 0.1482, + "input": 0.7756000000000001, + "output": 2.4376, + "cacheRead": 0.14404, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -79935,7 +79935,8 @@ "baseUrl": "https://api.venice.ai/api/v1", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.15, @@ -86599,9 +86600,9 @@ "text" ], "cost": { - "input": 1.5, - "output": 5.125, - "cacheRead": 0.25, + "input": 1.75, + "output": 5.5, + "cacheRead": 0.325, "cacheWrite": 0 }, "contextWindow": 1048576, diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 35f768d99..ca8ad644e 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -13,8 +13,11 @@ ### Changed - Optimized edit-tool previews, diff components, and intra-line word highlighting to compute line and word diffs natively, reducing synchronous diff times by 2-10x on large inputs. +- Updated diff generation and rendering components to rely exclusively on native UTF-16 diff bindings, removing `isWellFormed()` guards and JS fallback code paths. -### Fixed +### Removed + +- Removed npm `diff` dependency. - Fixed an issue where `Ctrl+V` clipboard paste was ignored while API-key and other modal prompts had focus. - Fixed `scripts/install.sh` incorrectly installing an x86_64 build on Apple Silicon when running under Rosetta. diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index f12fe1564..530b7a001 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -79,7 +79,6 @@ "@xterm/headless": "catalog:", "arktype": "catalog:", "chalk": "catalog:", - "diff": "catalog:", "fast-xml-parser": "catalog:", "handlebars": "catalog:", "header-generator": "catalog:", diff --git a/packages/coding-agent/src/edit/diff.ts b/packages/coding-agent/src/edit/diff.ts index 545823d05..b071ef77c 100644 --- a/packages/coding-agent/src/edit/diff.ts +++ b/packages/coding-agent/src/edit/diff.ts @@ -4,28 +4,13 @@ * Provides diff string generation and the replace-mode edit logic * used when not in patch mode. */ -import { diffLines as nativeDiffLines, structuredPatchHunks as nativeStructuredPatchHunks } from "@oh-my-pi/pi-natives"; -import { diffLines as jsDiffLines, structuredPatch as jsStructuredPatch } from "diff"; +import { diffLines, structuredPatchHunks } from "@oh-my-pi/pi-natives"; import { resolveToCwd } from "../tools/path-utils"; import { type BlockContextSource, findBlockContextLines } from "../utils/block-context"; import { DEFAULT_FUZZY_THRESHOLD, EditMatchError, findMatch } from "./modes/replace"; import { adjustIndentation, normalizeToLF, stripBom } from "./normalize"; import { readEditFileText } from "./read-file"; -/** Native line diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */ -function diffLines(oldContent: string, newContent: string) { - return oldContent.isWellFormed() && newContent.isWellFormed() - ? nativeDiffLines(oldContent, newContent) - : jsDiffLines(oldContent, newContent); -} - -/** Native structured-patch hunks with the same ill-formed-UTF-16 fallback as {@link diffLines}. */ -function structuredPatchHunks(oldContent: string, newContent: string, context: number) { - return oldContent.isWellFormed() && newContent.isWellFormed() - ? nativeStructuredPatchHunks(oldContent, newContent, context) - : jsStructuredPatch("", "", oldContent, newContent, "", "", { context }).hunks; -} - export interface DiffResult { diff: string; firstChangedLine: number | undefined; diff --git a/packages/coding-agent/src/modes/components/diff.ts b/packages/coding-agent/src/modes/components/diff.ts index f6f421703..917989511 100644 --- a/packages/coding-agent/src/modes/components/diff.ts +++ b/packages/coding-agent/src/modes/components/diff.ts @@ -1,16 +1,8 @@ -import { diffWords as nativeDiffWords } from "@oh-my-pi/pi-natives"; +import { diffWords } from "@oh-my-pi/pi-natives"; import { DEFAULT_TAB_WIDTH, sanitizeText } from "@oh-my-pi/pi-utils"; -import { diffWords as jsDiffWords } from "diff"; import { getLanguageFromPath, highlightCode, theme } from "../../modes/theme/theme"; import { type CodeFrameMarker, formatCodeFrameLine, replaceTabs } from "../../tools/render-utils"; -/** Native word diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */ -function diffWords(oldContent: string, newContent: string) { - return oldContent.isWellFormed() && newContent.isWellFormed() - ? nativeDiffWords(oldContent, newContent) - : jsDiffWords(oldContent, newContent); -} - /** SGR dim on / normal intensity — additive, preserves fg/bg colors. */ const DIM = "\x1b[2m"; const DIM_OFF = "\x1b[22m"; diff --git a/packages/coding-agent/test/tools/edit-diff.test.ts b/packages/coding-agent/test/tools/edit-diff.test.ts index 64e25c3b2..b25a498e2 100644 --- a/packages/coding-agent/test/tools/edit-diff.test.ts +++ b/packages/coding-agent/test/tools/edit-diff.test.ts @@ -167,10 +167,10 @@ describe("generateDiffString", () => { expect(diffLines.filter(line => line.includes("| const keep = 2;"))).toEqual([" 3| const keep = 2;"]); }); - it("detects changes between ill-formed UTF-16 inputs via the JS diff fallback", () => { - // The native binding rejects unpaired surrogates; the isWellFormed() - // guard must fall back to jsdiff instead of reporting no change (or - // throwing) for two distinct ill-formed lines. + it("detects changes between ill-formed UTF-16 inputs natively", () => { + // Native diffs operate directly over UTF-16 code units, so lone + // surrogates compare code-unit for code-unit without throwing or + // collapsing distinct lines. const result = generateDiffString("a\ud800b", "a\ud801b", 3); const rows = result.diff.split("\n"); expect(rows.some(row => row.startsWith("-1|"))).toBe(true); diff --git a/packages/coding-agent/test/tools/edit-renderer.test.ts b/packages/coding-agent/test/tools/edit-renderer.test.ts index f6f662b92..bf237828f 100644 --- a/packages/coding-agent/test/tools/edit-renderer.test.ts +++ b/packages/coding-agent/test/tools/edit-renderer.test.ts @@ -602,10 +602,9 @@ describe("editToolRenderer diff line wrapping", () => { for (const row of rows.slice(1, -1)) expect(row).toMatch(/^│\s*[+-]?\s*\d*│/); }); - it("renders ill-formed UTF-16 replacements through the JS word-diff fallback", async () => { - // Lone surrogates are rejected by the native diffWords binding; the - // component's isWellFormed() guard must route to jsdiff instead of - // throwing mid-render. + it("renders ill-formed UTF-16 replacements natively", async () => { + // Native word diffs operate directly over UTF-16 code units, so lone + // surrogates render and highlight without throwing mid-render. const rows = (await renderSingleLineReplacement("alpha \ud800 beta", "alpha \ud801 beta", 100)).map(row => Bun.stripANSI(row), ); diff --git a/packages/hashline/CHANGELOG.md b/packages/hashline/CHANGELOG.md index afeacfef0..3160a4c9a 100644 --- a/packages/hashline/CHANGELOG.md +++ b/packages/hashline/CHANGELOG.md @@ -7,7 +7,11 @@ ### Changed - Improved snapshot recovery line remapping by utilizing native line diffing. +- Switched line anchor recovery diffs to native `diffLineRuns`, processing UTF-16 code units directly and removing JS diff fallback. +### Removed + +- Removed npm `diff` dependency. ## [17.0.4] - 2026-07-18 ### Fixed diff --git a/packages/hashline/package.json b/packages/hashline/package.json index f235b737e..cc684bfb9 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -34,7 +34,6 @@ }, "dependencies": { "@oh-my-pi/pi-natives": "catalog:", - "diff": "catalog:", "lru-cache": "catalog:" }, "devDependencies": { diff --git a/packages/hashline/src/recovery.ts b/packages/hashline/src/recovery.ts index bca26ebfd..f6456b23f 100644 --- a/packages/hashline/src/recovery.ts +++ b/packages/hashline/src/recovery.ts @@ -6,8 +6,7 @@ * Recovery fails closed when the target changed or became ambiguous. The * patcher then returns a mismatch with fresh context instead of guessing. */ -import { diffLineRuns as nativeDiffLineRuns } from "@oh-my-pi/pi-natives"; -import { diffArrays } from "diff"; +import { diffLineRuns } from "@oh-my-pi/pi-natives"; import { applyEdits } from "./apply"; import { RECOVERY_EXTERNAL_WARNING, RECOVERY_LINE_REMAP_WARNING, RECOVERY_SESSION_CHAIN_WARNING } from "./messages"; import type { SnapshotStore } from "./snapshots"; @@ -45,21 +44,6 @@ function getEditAnchors(edit: Edit): Anchor[] { return edit.cursor.kind === "before_anchor" || edit.cursor.kind === "after_anchor" ? [edit.cursor.anchor] : []; } -/** Native line-run diff when both inputs are well-formed UTF-16; the native binding rejects unpaired surrogates, so ill-formed inputs fall back to jsdiff. */ -function diffLineRuns( - previousText: string, - currentText: string, -): { count: number; added: boolean; removed: boolean }[] { - if (previousText.isWellFormed() && currentText.isWellFormed()) { - return nativeDiffLineRuns(previousText, currentText); - } - return diffArrays(previousText.split("\n"), currentText.split("\n")).map(change => ({ - count: change.count, - added: change.added, - removed: change.removed, - })); -} - function buildLineMap(previousText: string, currentText: string): Map { const changes = diffLineRuns(previousText, currentText); const map = new Map(); diff --git a/packages/hashline/test/recovery-session-chain.test.ts b/packages/hashline/test/recovery-session-chain.test.ts index ca2159f79..9957ced76 100644 --- a/packages/hashline/test/recovery-session-chain.test.ts +++ b/packages/hashline/test/recovery-session-chain.test.ts @@ -264,10 +264,9 @@ describe("Recovery — colliding snapshot tags", () => { }); describe("Recovery — ill-formed UTF-16 content", () => { - it("remaps line anchors through the JS diff fallback when the file contains lone surrogates", () => { - // The native diffLineRuns binding rejects unpaired surrogates; the - // isWellFormed() guard must fall back to jsdiff so recovery still - // remaps anchors instead of throwing or refusing. + it("remaps line anchors natively when the file contains lone surrogates", () => { + // Native diffLineRuns operates directly over UTF-16 code units, so + // recovery remaps anchors natively when files contain lone surrogates. const store = new InMemorySnapshotStore(); const snapshotText = lines("head", "lone \ud800 surrogate", "target line", "tail"); const hash = store.record(PATH, snapshotText); diff --git a/packages/metaharness/package.json b/packages/metaharness/package.json index 1aaff47fc..c41e98cb0 100644 --- a/packages/metaharness/package.json +++ b/packages/metaharness/package.json @@ -35,6 +35,7 @@ "d3-scale": "^4.0.2", "d3-shape": "^3.2.0", "diff": "catalog:", + "@oh-my-pi/pi-natives": "catalog:", "motion": "^12.15.0", "react": "^19.1.0", "react-dom": "^19.1.0", diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index 3b36a6ca6..055d3e727 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -9,10 +9,17 @@ - Added jsdiff-compatible native diff exports: `diffLines`, `diffWords`, `diffLineRuns`, and `structuredPatchHunks`. - Added batch vector kernels for mnemopi recall paths: `cosineSimilarityPairs`, `vectorIndexTopK`, and `mmrRerankIndices`. +### Changed + +- Updated diff functions (`diffLines`, `diffWords`, `diffLineRuns`, `structuredPatchHunks`) to process UTF-16 code units natively end to end via `Utf16String`, supporting ill-formed JS strings with unpaired surrogates without throwing or converting to UTF-8. + ### Fixed - Fixed a critical issue where the in-process `rm` builtin treated an empty path operand as the current working directory, causing `rm -rf ""` to recursively delete the current directory. Empty operands are now rejected, matching GNU `rm` behavior. +### Removed + +- Removed unused `similar` crate dependency and dev-dependency on npm `diff`. ## [17.0.5] - 2026-07-18 ### Added diff --git a/packages/natives/bench/diff-results.md b/packages/natives/bench/diff-results.md deleted file mode 100644 index fab335b72..000000000 --- a/packages/natives/bench/diff-results.md +++ /dev/null @@ -1,35 +0,0 @@ -# Native diff benchmark - -- Source: `973bad0c6170d2ff2ad08cb52c574002f1d02ec8` (clean tree, deployed via `git archive`) -- Host: Apple M1, darwin-arm64, bun 1.3.14, idle machine (load < 7) -- Native build: ci profile, `RUSTFLAGS="-C target-cpu=apple-m1"` -- Method: seeded synthetic docs, per-scenario warmup + timed iterations, serial runs, crossing-inclusive. Native timings include the `isWellFormed()` guards the production call sites pay before choosing the native path. -- Command: `PI_COMPILED=1 bun packages/natives/bench/diff.ts` (`BENCH_WARMUP` / `BENCH_ITERATIONS` / `BENCH_MAX_LINES` / `BENCH_SHA` env overrides) - -## High-precision run (warmup 20, iterations 300/scenario, scenarios ≤ 5000 lines) - -| scenario | jsdiff diffLines | native diffLines | speedup | jsdiff structuredPatch | native hunks | speedup | -|---|---|---|---|---|---|---| -| 100 lines / 1% edits | 37.3µs | 21.7µs | 1.7x | 47.9µs | 21.7µs | 2.2x | -| 100 lines / 20% edits | 63.1µs | 38.0µs | 1.7x | 69.9µs | 38.0µs | 1.8x | -| 5000 lines / 1% edits | 1.10ms | 827.7µs | 1.3x | 1.35ms | 825.4µs | 1.6x | -| 5000 lines / 20% edits | 150.04ms | 25.30ms | 5.9x | 150.66ms | 26.03ms | 5.8x | - -## Full run including 50k-line scenarios (warmup 2, iterations 10/scenario) - -Reduced iterations because jsdiff needs ~22s per iteration on the heaviest row. - -| scenario | jsdiff diffLines | native diffLines | speedup | jsdiff structuredPatch | native hunks | speedup | -|---|---|---|---|---|---|---| -| 100 lines / 1% edits | 57.2µs | 22.6µs | 2.5x | 64.3µs | 21.2µs | 3.0x | -| 100 lines / 20% edits | 66.3µs | 37.2µs | 1.8x | 104.4µs | 34.4µs | 3.0x | -| 5000 lines / 1% edits | 1.41ms | 970.6µs | 1.5x | 1.71ms | 889.7µs | 1.9x | -| 5000 lines / 20% edits | 152.44ms | 26.32ms | 5.8x | 154.58ms | 25.20ms | 6.1x | -| 50000 lines / 1% edits | 46.17ms | 16.03ms | 2.9x | 50.95ms | 15.34ms | 3.3x | -| 50000 lines / 20% edits | 22410.26ms | 2973.66ms | 7.5x | 23010.35ms | 2996.85ms | 7.7x | - -## Notes - -- Native wins at every measured size, guards included; no crossover where the N-API crossing plus the well-formedness scan dominates. -- The worst jsdiff case (50k lines / 20% edit density) is a ~22s synchronous stall on the render path vs ~3s native. -- Behavior parity with jsdiff is defended by `packages/natives/test/diff-parity.test.ts` (fixed edge cases, seeded random documents including CRLF/unicode/no-trailing-newline, seeded random word diffs, a 10k-line document, and explicit ill-formed UTF-16 rejection) plus the call-site fallback regressions in `edit-diff.test.ts`, `edit-renderer.test.ts`, and `recovery-session-chain.test.ts`, all run with `PI_COMPILED=1 bun test`. diff --git a/packages/natives/bench/diff.ts b/packages/natives/bench/diff.ts deleted file mode 100644 index 3db13b1a8..000000000 --- a/packages/natives/bench/diff.ts +++ /dev/null @@ -1,78 +0,0 @@ -/** - * Benchmark: native diff (N-API) vs jsdiff for line diffs and structured patches. - * - * Usage: bun bench/diff.ts - * Records git sha, scenario, and iteration counts; prints a markdown table. - */ -import * as os from "node:os"; -import { diffLines as nativeDiffLines, structuredPatchHunks } from "../native/index.js"; -import * as Diff from "diff"; - -const WARMUP = Number(Bun.env.BENCH_WARMUP ?? 5); -const ITERATIONS = Number(Bun.env.BENCH_ITERATIONS ?? 50); - -function makeRng(seed: number) { - let state = seed >>> 0; - return () => { - state = (state * 1664525 + 1013904223) >>> 0; - return state / 0x1_0000_0000; - }; -} - -function buildDoc(rng: () => number, lines: number): string { - const words = ["alpha", "beta", "gamma", "delta", "epsilon", "zeta", "eta", "theta"]; - const out: string[] = []; - for (let i = 0; i < lines; i++) { - out.push(`${words[Math.floor(rng() * words.length)]} ${words[Math.floor(rng() * words.length)]} line${i}`); - } - return `${out.join("\n")}\n`; -} - -function mutate(rng: () => number, text: string, density: number): string { - const lines = text.split("\n"); - for (let i = 0; i < lines.length; i++) { - const roll = rng(); - if (roll < density / 3) lines[i] = `${lines[i]} edited`; - else if (roll < (density * 2) / 3) { - lines.splice(i, 1); - i--; - } else if (roll < density) lines.splice(i, 0, `ins ${Math.floor(rng() * 1e6)}`); - } - return lines.join("\n"); -} - -function bench(fn: () => unknown): { meanMs: number; iterations: number } { - for (let i = 0; i < WARMUP; i++) fn(); - const start = performance.now(); - for (let i = 0; i < ITERATIONS; i++) fn(); - return { meanMs: (performance.now() - start) / ITERATIONS, iterations: ITERATIONS }; -} - -const sha = Bun.env.BENCH_SHA ?? Bun.spawnSync(["git", "rev-parse", "HEAD"]).stdout.toString().trim(); -console.log(`# diff bench — sha ${sha}, warmup ${WARMUP}, iterations ${ITERATIONS}/scenario`); -console.log(`# host: ${os.cpus()[0]?.model ?? "unknown"}, ${os.platform()}-${os.arch()}, bun ${Bun.version}\n`); -console.log("| scenario | jsdiff diffLines | native diffLines | speedup | jsdiff structuredPatch | native hunks | speedup |"); -console.log("|---|---|---|---|---|---|---|"); - -const MAX_LINES = Number(Bun.env.BENCH_MAX_LINES ?? Number.POSITIVE_INFINITY); -for (const lines of [100, 5_000, 50_000].filter(n => n <= MAX_LINES)) { - for (const density of [0.01, 0.2]) { - const rng = makeRng(lines * 31 + density * 1000); - const oldText = buildDoc(rng, lines); - const newText = mutate(rng, oldText, density); - const jsLines = bench(() => Diff.diffLines(oldText, newText)); - // Native timings include the isWellFormed() guards the production call - // sites pay before choosing the native path. - const natLines = bench(() => - oldText.isWellFormed() && newText.isWellFormed() ? nativeDiffLines(oldText, newText) : undefined, - ); - const jsPatch = bench(() => Diff.structuredPatch("", "", oldText, newText, "", "", { context: 3 })); - const natPatch = bench(() => - oldText.isWellFormed() && newText.isWellFormed() ? structuredPatchHunks(oldText, newText, 3) : undefined, - ); - const fmt = (ms: number) => (ms >= 1 ? `${ms.toFixed(2)}ms` : `${(ms * 1000).toFixed(1)}µs`); - console.log( - `| ${lines} lines / ${density * 100}% edits | ${fmt(jsLines.meanMs)} | ${fmt(natLines.meanMs)} | ${(jsLines.meanMs / natLines.meanMs).toFixed(1)}x | ${fmt(jsPatch.meanMs)} | ${fmt(natPatch.meanMs)} | ${(jsPatch.meanMs / natPatch.meanMs).toFixed(1)}x |`, - ); - } -} diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index c6737ea51..068eca625 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -518,7 +518,7 @@ export interface DiffChange { * Diff `oldText.split(" ")` against `newText.split(" ")` with jsdiff - * `diffArrays` semantics (exact string equality, empty lines preserved), + * `diffArrays` semantics (exact code-unit equality, empty lines preserved), * returning only run lengths. * * Callers that map line numbers — like hashline recovery — need the counts, diff --git a/packages/natives/package.json b/packages/natives/package.json index 34f4ebb46..f2293962f 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -43,8 +43,7 @@ }, "devDependencies": { "@napi-rs/cli": "catalog:", - "@types/bun": "catalog:", - "diff": "catalog:" + "@types/bun": "catalog:" }, "engines": { "bun": ">=1.3.14" diff --git a/packages/natives/test/diff-parity.test.ts b/packages/natives/test/diff-parity.test.ts deleted file mode 100644 index c0aa9b7ef..000000000 --- a/packages/natives/test/diff-parity.test.ts +++ /dev/null @@ -1,192 +0,0 @@ -import { describe, expect, test } from "bun:test"; -import { diffLineRuns, diffLines, diffWords, structuredPatchHunks } from "@oh-my-pi/pi-natives"; -import * as Diff from "diff"; - -/** Normalize jsdiff change objects (added/removed may be undefined). */ -function jsChanges(changes: Diff.Change[]) { - return changes.map(c => ({ value: c.value, count: c.count, added: !!c.added, removed: !!c.removed })); -} - -function natChanges(changes: { value: string; count: number; added: boolean; removed: boolean }[]) { - return changes.map(c => ({ value: c.value, count: c.count, added: c.added, removed: c.removed })); -} - -function jsHunks(oldText: string, newText: string, context: number) { - return Diff.structuredPatch("", "", oldText, newText, "", "", { context }).hunks.map(h => ({ - oldStart: h.oldStart, - oldLines: h.oldLines, - newStart: h.newStart, - newLines: h.newLines, - lines: h.lines, - })); -} - -function assertParity(oldText: string, newText: string) { - expect(natChanges(diffLines(oldText, newText))).toEqual(jsChanges(Diff.diffLines(oldText, newText))); - for (const context of [0, 3, 4]) { - expect(structuredPatchHunks(oldText, newText, context)).toEqual(jsHunks(oldText, newText, context)); - } - const jsRuns = Diff.diffArrays(oldText.split("\n"), newText.split("\n")).map(c => ({ - count: c.value.length, - added: !!c.added, - removed: !!c.removed, - })); - expect(diffLineRuns(oldText, newText).map(c => ({ count: c.count, added: c.added, removed: c.removed }))).toEqual( - jsRuns, - ); -} - -/** Deterministic LCG so failures are reproducible. */ -function makeRng(seed: number) { - let state = seed >>> 0; - return () => { - state = (state * 1664525 + 1013904223) >>> 0; - return state / 0x1_0000_0000; - }; -} - -function randomText(rng: () => number, lines: number, opts: { crlf?: boolean; unicode?: boolean } = {}) { - const words = opts.unicode - ? ["alpha", "béta", "γάμμα", "デルタ", "🚀rocket", "ω"] - : ["alpha", "beta", "gamma", "delta", "epsilon", "zeta"]; - const eol = opts.crlf ? "\r\n" : "\n"; - const out: string[] = []; - for (let i = 0; i < lines; i++) { - const n = 1 + Math.floor(rng() * 4); - const parts: string[] = []; - for (let j = 0; j < n; j++) parts.push(words[Math.floor(rng() * words.length)]!); - out.push(parts.join(" ")); - } - let text = out.join(eol); - if (rng() > 0.5) text += eol; - return text; -} - -function mutate(rng: () => number, text: string, density: number) { - const lines = text.split("\n"); - for (let i = 0; i < lines.length; i++) { - const roll = rng(); - if (roll < density / 3) lines[i] = `${lines[i]} edited`; - else if (roll < (density * 2) / 3) { - lines.splice(i, 1); - i--; - } else if (roll < density) lines.splice(i, 0, `inserted ${Math.floor(rng() * 1000)}`); - } - return lines.join("\n"); -} - -describe("native diff parity with jsdiff", () => { - test("fixed edge cases", () => { - assertParity("", ""); - assertParity("", "a\nb\n"); - assertParity("a\nb\n", ""); - assertParity("same\ntext\n", "same\ntext\n"); - assertParity("a\nb\nc", "a\nx\nc"); // no trailing newline - assertParity("a\r\nb\r\nc\r\n", "a\r\nx\r\nc\r\n"); // CRLF - assertParity("a\nb\n", "a\r\nb\r\n"); // mixed EOL rewrite - assertParity("líne ünicode 🚀\nsecond\n", "líne ünicode 🚀\nsécond\n"); - assertParity("lone\rcarriage\n", "lone\rreturn\n"); // \r is not a line break - }); - - test("word diff parity", () => { - const cases: [string, string][] = [ - ["foo bar baz", "foo qux baz"], - [" leading space", "leading space"], - ["tab\tsep", "tab sep"], - ["punct, mark!", "punct; mark?"], - ["ünïcode wörds", "ünïcode words"], - ["", "new words"], - ["same same", "same same"], - ]; - for (const [a, b] of cases) { - expect(natChanges(diffWords(a, b))).toEqual(jsChanges(Diff.diffWords(a, b))); - } - }); - - test("ill-formed UTF-16 is rejected, never silently normalized", () => { - // N-API UTF-8 conversion would replace unpaired surrogates with U+FFFD - // and collapse distinct inputs into "no change"; the native exports - // throw instead so callers fall back to a UTF-16-aware JS diff. - const a = "a\ud800b"; - const b = "a\ud801b"; - expect(() => diffLines(a, b)).toThrow(/ill-formed UTF-16/); - expect(() => diffLineRuns(a, b)).toThrow(/ill-formed UTF-16/); - expect(() => diffWords(a, b)).toThrow(/ill-formed UTF-16/); - expect(() => structuredPatchHunks(a, b, 3)).toThrow(/ill-formed UTF-16/); - expect(() => diffLines("well formed", b)).toThrow(/ill-formed UTF-16/); - // A legitimate embedded NUL is content, not a terminator. - expect(natChanges(diffLines("a\u0000\nb\n", "a\u0000\nc\n"))).toEqual( - jsChanges(Diff.diffLines("a\u0000\nb\n", "a\u0000\nc\n")), - ); - }); - - test("seeded random word diffs", () => { - // Token pool stresses jsdiff's word/whitespace boundary rules: repeated - // and mixed whitespace, tabs, newlines, punctuation runs, Latin - // extended, non-Latin scripts, emoji (surrogate pairs), and digits. - const tokens = [ - "word", - "Wörter", - "naïve", - "λέξη", - "слово", - "単語", - "🚀", - "👍🏽", - "can't", - "co-op", - "...", - "!?", - ";", - "(x)", - "42", - "3.14", - " ", - " ", - "\t", - " \t ", - "\n", - "\n\n", - " \n ", - ]; - const rng = makeRng(0xd1ff); - const build = (len: number) => { - let s = ""; - for (let i = 0; i < len; i++) s += tokens[Math.floor(rng() * tokens.length)]; - return s; - }; - for (let round = 0; round < 200; round++) { - const a = build(Math.floor(rng() * 40)); - // Mix related pairs (mutations of `a`) with unrelated pairs. - let b: string; - if (round % 2 === 0) { - b = build(Math.floor(rng() * 40)); - } else { - const parts = a.split(/(\s+)/).filter(Boolean); - for (let i = 0; i < parts.length; i++) { - const roll = rng(); - if (roll < 0.15) parts[i] = tokens[Math.floor(rng() * tokens.length)] ?? ""; - else if (roll < 0.25) parts[i] = ""; - } - b = parts.join(""); - } - expect(natChanges(diffWords(a, b))).toEqual(jsChanges(Diff.diffWords(a, b))); - } - }); - - test("seeded random documents", () => { - const rng = makeRng(0xc0ffee); - for (let round = 0; round < 30; round++) { - const crlf = round % 3 === 1; - const unicode = round % 4 === 2; - const base = randomText(rng, 5 + Math.floor(rng() * 120), { crlf, unicode }); - assertParity(base, mutate(rng, base, round % 2 === 0 ? 0.01 : 0.2)); - } - }); - - test("10k-line document", () => { - const rng = makeRng(0xbeef); - const base = randomText(rng, 10_000); - assertParity(base, mutate(rng, base, 0.01)); - }); -}); diff --git a/packages/natives/test/diff.test.ts b/packages/natives/test/diff.test.ts new file mode 100644 index 000000000..f9ad9b6c7 --- /dev/null +++ b/packages/natives/test/diff.test.ts @@ -0,0 +1,254 @@ +import { describe, expect, test } from "bun:test"; +import { diffLineRuns, diffLines, diffWords, type PatchHunk, structuredPatchHunks } from "@oh-my-pi/pi-natives"; + +function applyHunks(oldText: string, hunks: PatchHunk[]): string { + if (hunks.length === 0) return oldText; + const oldLines = oldText.split("\n"); + const resultLines: string[] = []; + let oldIdx = 0; + let removeEofNewline = false; + + for (const hunk of hunks) { + const targetOldIdx = hunk.oldStart > 0 ? hunk.oldStart - 1 : 0; + while (oldIdx < targetOldIdx && oldIdx < oldLines.length) { + resultLines.push(oldLines[oldIdx]!); + oldIdx++; + } + for (let i = 0; i < hunk.lines.length; i++) { + const line = hunk.lines[i]!; + if (line.startsWith("\\ No newline at end of file")) { + const prevLine = hunk.lines[i - 1]; + if (prevLine && (prevLine.startsWith("+") || prevLine.startsWith(" "))) { + removeEofNewline = true; + } + continue; + } + if (line.startsWith("-")) { + oldIdx++; + } else if (line.startsWith("+")) { + resultLines.push(line.slice(1)); + } else if (line.startsWith(" ")) { + resultLines.push(line.slice(1)); + oldIdx++; + } + } + } + while (oldIdx < oldLines.length) { + resultLines.push(oldLines[oldIdx]!); + oldIdx++; + } + + if (removeEofNewline && resultLines[resultLines.length - 1] === "") { + resultLines.pop(); + } + return resultLines.join("\n"); +} + +function verifyDiffInvariants(oldText: string, newText: string) { + // 1. diffLines reconstruction + const lineChanges = diffLines(oldText, newText); + let reconstructedOld = ""; + let reconstructedNew = ""; + for (const change of lineChanges) { + if (change.removed) { + reconstructedOld += change.value; + } else if (change.added) { + reconstructedNew += change.value; + } else { + reconstructedOld += change.value; + reconstructedNew += change.value; + } + } + expect(reconstructedOld).toBe(oldText); + expect(reconstructedNew).toBe(newText); + + // 2. structuredPatchHunks reconstruction + for (const context of [0, 3, 4]) { + const hunks = structuredPatchHunks(oldText, newText, context); + const patchedText = applyHunks(oldText, hunks); + expect(patchedText).toBe(newText); + } + + // 3. diffLineRuns counts + const oldLineCount = oldText.split("\n").length; + const newLineCount = newText.split("\n").length; + const runs = diffLineRuns(oldText, newText); + let sumOld = 0; + let sumNew = 0; + for (const run of runs) { + if (run.removed) sumOld += run.count; + else if (run.added) sumNew += run.count; + else { + sumOld += run.count; + sumNew += run.count; + } + } + expect(sumOld).toBe(oldLineCount); + expect(sumNew).toBe(newLineCount); +} + +/** Deterministic LCG so failures are reproducible. */ +function makeRng(seed: number) { + let state = seed >>> 0; + return () => { + state = (state * 1664525 + 1013904223) >>> 0; + return state / 0x1_0000_0000; + }; +} + +function randomText(rng: () => number, lines: number, opts: { crlf?: boolean; unicode?: boolean } = {}) { + const words = opts.unicode + ? ["alpha", "béta", "γάμμα", "デルタ", "🚀rocket", "ω"] + : ["alpha", "beta", "gamma", "delta", "epsilon", "zeta"]; + const eol = opts.crlf ? "\r\n" : "\n"; + const out: string[] = []; + for (let i = 0; i < lines; i++) { + const n = 1 + Math.floor(rng() * 4); + const parts: string[] = []; + for (let j = 0; j < n; j++) parts.push(words[Math.floor(rng() * words.length)]!); + out.push(parts.join(" ")); + } + let text = out.join(eol); + if (rng() > 0.5) text += eol; + return text; +} + +function mutate(rng: () => number, text: string, density: number) { + const lines = text.split("\n"); + for (let i = 0; i < lines.length; i++) { + const roll = rng(); + if (roll < density / 3) lines[i] = `${lines[i]} edited`; + else if (roll < (density * 2) / 3) { + lines.splice(i, 1); + i--; + } else if (roll < density) lines.splice(i, 0, `inserted ${Math.floor(rng() * 1000)}`); + } + return lines.join("\n"); +} + +describe("native diff correctness", () => { + test("fixed edge cases", () => { + verifyDiffInvariants("", ""); + verifyDiffInvariants("", "a\nb\n"); + verifyDiffInvariants("a\nb\n", ""); + verifyDiffInvariants("same\ntext\n", "same\ntext\n"); + verifyDiffInvariants("a\nb\nc", "a\nx\nc"); + verifyDiffInvariants("a\r\nb\r\nc\r\n", "a\r\nx\r\nc\r\n"); + verifyDiffInvariants("a\nb\n", "a\r\nb\r\n"); + verifyDiffInvariants("líne ünicode 🚀\nsecond\n", "líne ünicode 🚀\nsécond\n"); + verifyDiffInvariants("lone\rcarriage\n", "lone\rreturn\n"); + // A legitimate embedded NUL is content, not a terminator. + verifyDiffInvariants("a\u0000\nb\n", "a\u0000\nc\n"); + }); + + test("word diff behavior", () => { + const changes = diffWords("foo bar baz", "foo qux baz"); + expect(changes).toEqual([ + { value: "foo ", count: 1, added: false, removed: false }, + { value: "bar", count: 1, added: false, removed: true }, + { value: "qux", count: 1, added: true, removed: false }, + { value: " baz", count: 1, added: false, removed: false }, + ]); + }); + + test("ill-formed UTF-16 is legal content, preserved code unit for code unit", () => { + const a = "a\ud800b"; + const b = "a\ud801b"; + // Distinct lone surrogates must survive the N-API round trip untouched + // (UTF-8 conversion would collapse both into U+FFFD and report "no + // change") and diff as distinct content. + expect(diffLines(a, b)).toEqual([ + { value: a, count: 1, added: false, removed: true }, + { value: b, count: 1, added: true, removed: false }, + ]); + expect(diffWords("alpha \ud800 beta", "alpha \ud801 beta")).toEqual([ + { value: "alpha ", count: 1, added: false, removed: false }, + { value: "\ud800", count: 1, added: false, removed: true }, + { value: "\ud801", count: 1, added: true, removed: false }, + { value: " beta", count: 1, added: false, removed: false }, + ]); + expect(diffLineRuns(a, b)).toEqual([ + { count: 1, added: false, removed: true }, + { count: 1, added: true, removed: false }, + ]); + expect(structuredPatchHunks(a, b, 3).length).toBeGreaterThan(0); + verifyDiffInvariants(a, b); + }); + + test("seeded random word diffs reconstruct the new text exactly", () => { + // Token pool stresses word/whitespace boundary rules: repeated and + // mixed whitespace, tabs, newlines, punctuation runs, Latin extended, + // non-Latin scripts, emoji (surrogate pairs), digits, and lone + // surrogates. + const tokens = [ + "word", + "Wörter", + "naïve", + "λέξη", + "слово", + "単語", + "🚀", + "👍🏽", + "can't", + "co-op", + "...", + "!?", + ";", + "(x)", + "42", + "3.14", + " ", + " ", + "\t", + " \t ", + "\n", + "\n\n", + " \n ", + "\ud800", + "\udc00", + ]; + const rng = makeRng(0xd1ff); + const build = (len: number) => { + let s = ""; + for (let i = 0; i < len; i++) s += tokens[Math.floor(rng() * tokens.length)]; + return s; + }; + for (let round = 0; round < 200; round++) { + const a = build(Math.floor(rng() * 40)); + const b = build(Math.floor(rng() * 40)); + const changes = diffWords(a, b); + // Common tokens carry the new text's whitespace and the post-pass + // dedupes boundary whitespace, so the kept+added side concatenates + // back to the new text byte for byte... + expect( + changes + .filter(c => !c.removed) + .map(c => c.value) + .join(""), + ).toBe(b); + // ...while the kept+removed side preserves all non-whitespace + // content of the old text in order. + const oldSide = changes + .filter(c => !c.added) + .map(c => c.value) + .join(""); + expect(oldSide.replace(/\s+/gu, "")).toBe(a.replace(/\s+/gu, "")); + } + }); + + test("seeded random documents", () => { + const rng = makeRng(0xc0ffee); + for (let round = 0; round < 30; round++) { + const crlf = round % 3 === 1; + const unicode = round % 4 === 2; + const base = randomText(rng, 5 + Math.floor(rng() * 120), { crlf, unicode }); + verifyDiffInvariants(base, mutate(rng, base, round % 2 === 0 ? 0.01 : 0.2)); + } + }); + + test("10k-line document", () => { + const rng = makeRng(0xbeef); + const base = randomText(rng, 10_000); + verifyDiffInvariants(base, mutate(rng, base, 0.01)); + }); +}); diff --git a/packages/typescript-edit-benchmark/package.json b/packages/typescript-edit-benchmark/package.json index 1c7cbe408..8ec81cceb 100644 --- a/packages/typescript-edit-benchmark/package.json +++ b/packages/typescript-edit-benchmark/package.json @@ -38,6 +38,7 @@ "@oh-my-pi/pi-coding-agent": "catalog:", "@oh-my-pi/pi-utils": "catalog:", "diff": "catalog:", + "@oh-my-pi/pi-natives": "catalog:", "prettier": "catalog:", "regexp-tree": "catalog:", "@oh-my-pi/pi-ai": "catalog:",