From 23201555eef2db48585762a8e50979b340e5eb1e Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 01:14:57 -0700 Subject: [PATCH 01/16] feat(natives): add batch vector kernels for mnemopi recall paths Batch cosine similarity, pairwise similarity clustering, exact vector index top-k, packed binary Hamming distance (plain and dimension-masked), and MMR rerank index selection. One N-API crossing per recall operation: flattened Float32/Float64Array candidate matrices with explicit dim, packed Uint8Array binary batches with stride, and MMR taking scores plus contents returning selected indices. Semantics mirror the TS reference implementations bit-exactly (accumulation order, non-finite handling, stable-sort tie-breaking, NaN-round MMR selection). --- crates/pi-natives/src/lib.rs | 1 + crates/pi-natives/src/vectors.rs | 476 +++++++++++++++++++++++++++++ packages/natives/native/index.d.ts | 91 ++++++ packages/natives/native/index.js | 6 + 4 files changed, 574 insertions(+) create mode 100644 crates/pi-natives/src/vectors.rs diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index f354ba3bf..e49739178 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -53,6 +53,7 @@ pub(crate) mod testing; pub mod text; pub mod tokens; pub(crate) mod utils; +pub mod vectors; pub mod workspace; #[cfg(target_os = "windows")] diff --git a/crates/pi-natives/src/vectors.rs b/crates/pi-natives/src/vectors.rs new file mode 100644 index 000000000..0610e023b --- /dev/null +++ b/crates/pi-natives/src/vectors.rs @@ -0,0 +1,476 @@ +//! Batch numeric vector kernels for mnemopi recall paths. +//! +//! Every export processes an entire candidate batch per N-API crossing so the +//! crossing cost is amortized over the whole recall operation. Semantics +//! mirror the TypeScript reference implementations in +//! `packages/mnemopi/src/core` exactly — same accumulation order, same +//! non-finite handling, same tie-breaking — so float scores are +//! bit-identical to the TS versions and integer results are exactly equal. + +use napi::{ + Error, Result, Status, + bindgen_prelude::{Float32Array, Float64Array, Uint8Array, Uint32Array}, +}; +use napi_derive::napi; + +fn invalid(message: &str) -> Result { + Err(Error::new(Status::InvalidArg, message)) +} + +#[inline] +fn finite_or_zero(value: f64) -> f64 { + if value.is_finite() { value } else { 0.0 } +} + +/// Cosine similarity with the exact semantics of mnemopi's TS +/// `cosineSimilarity`: iterate `max(len_a, len_b)` elements, treat missing +/// and non-finite entries as `0`, return `0` when either norm is zero. +/// +/// Splitting the shared prefix from the tails preserves bit-exactness: tail +/// terms of the shorter side only ever add `±0.0` to `dot` and `+0.0` to its +/// own norm, in the same index order as the TS loop. +#[inline] +fn cosine_one(a: &[f64], b: &[f64]) -> f64 { + if a.is_empty() && b.is_empty() { + return 0.0; + } + let shared = a.len().min(b.len()); + let mut dot = 0.0f64; + let mut norm_a = 0.0f64; + let mut norm_b = 0.0f64; + for i in 0..shared { + let av = finite_or_zero(a[i]); + let bv = finite_or_zero(b[i]); + dot += av * bv; + norm_a += av * av; + norm_b += bv * bv; + } + for &raw in &a[shared..] { + let av = finite_or_zero(raw); + norm_a += av * av; + } + for &raw in &b[shared..] { + let bv = finite_or_zero(raw); + norm_b += bv * bv; + } + if norm_a == 0.0 || norm_b == 0.0 { + return 0.0; + } + dot / (norm_a.sqrt() * norm_b.sqrt()) +} + +/// Cosine similarity of `query` against a batch of candidate vectors. +/// +/// `candidates` is `n` vectors flattened row-major at `dim` elements per row +/// (callers zero-pad shorter vectors, which matches the TS `?? 0` missing +/// element semantics). Returns one score per candidate, bit-identical to +/// calling the TS `cosineSimilarity(query, candidate)` per row. +#[napi] +pub fn cosine_similarity_batch( + query: Float64Array, + candidates: Float64Array, + dim: u32, +) -> Result { + let dim = dim as usize; + let cands: &[f64] = &candidates; + if dim == 0 { + if !cands.is_empty() { + return invalid("candidates must be empty when dim is 0"); + } + return Ok(Float64Array::new(Vec::new())); + } + if cands.len() % dim != 0 { + return invalid("candidates length must be a multiple of dim"); + } + let q: &[f64] = &query; + let scores: Vec = cands + .chunks_exact(dim) + .map(|row| cosine_one(q, row)) + .collect(); + Ok(Float64Array::new(scores)) +} + +/// All pairs `(i, j)` with `i < j` whose cosine similarity meets `threshold`. +/// +/// `vectors` is `count` vectors flattened row-major at `dim` `f32` elements +/// per row (zero-padded). Element values are widened to `f64` before any +/// arithmetic, exactly like JS reads from a `Float32Array`, so the similarity +/// is bit-identical to the TS pairwise loop in `clusterBySimilarity`. +/// Returns pairs flattened as `[i0, j0, i1, j1, ...]` in the same `(i, j)` +/// visit order as the TS nested loop. +#[napi] +pub fn cosine_similarity_pairs( + vectors: Float32Array, + count: u32, + dim: u32, + threshold: f64, +) -> Result { + let count = count as usize; + let dim = dim as usize; + let data: &[f32] = &vectors; + if data.len() != count * dim { + return invalid("vectors length must equal count * dim"); + } + let widened: Vec = data.iter().map(|&v| f64::from(v)).collect(); + let mut pairs: Vec = Vec::new(); + for i in 0..count { + let left = &widened[i * dim..(i + 1) * dim]; + for j in (i + 1)..count { + let right = &widened[j * dim..(j + 1) * dim]; + if cosine_one(left, right) >= threshold { + pairs.push(i as u32); + pairs.push(j as u32); + } + } + } + Ok(Uint32Array::new(pairs)) +} + +/// Top-k rows of a normalized vector matrix ranked by dot product with a +/// normalized query. +#[napi(object)] +pub struct VectorTopK { + /// Row indices of the selected hits, best score first. + pub indices: Uint32Array, + /// Scores aligned with `indices`. + pub scores: Float64Array, +} + +/// Score every row of a normalized `f32` matrix against `query` and return +/// the top `limit` rows. +/// +/// Mirrors the TS `searchExactVectorIndex` loop bit-exactly: the query is +/// normalized by the L2 norm of its *full* length, each row score sums +/// `matrix[row][col] * (query[col] / norm)` over +/// `min(query.len, dimensions)` columns in column order. Ranking matches the +/// TS stable sort: score descending, lower row index first on exact ties +/// (`-0.0` and `+0.0` compare equal). Callers are expected to enforce the TS +/// guards first (finite query with a positive norm, non-empty matrix). +#[napi] +pub fn vector_index_top_k( + matrix: Float32Array, + dimensions: u32, + query: Float64Array, + limit: u32, +) -> Result { + let dims = dimensions as usize; + let data: &[f32] = &matrix; + if dims == 0 || data.len() % dims != 0 { + return invalid("matrix length must be a positive multiple of dimensions"); + } + let count = data.len() / dims; + let q: &[f64] = &query; + let mut norm_sq = 0.0f64; + for &value in q { + norm_sq += value * value; + } + let norm = norm_sq.sqrt(); + // Hoisting the per-column division out of the row loop is bitwise + // identical to the TS per-row `query[col] / queryNorm`. + let query_dims = q.len().min(dims); + let normalized: Vec = q[..query_dims].iter().map(|&v| v / norm).collect(); + + let mut order: Vec<(f64, u32)> = Vec::with_capacity(count); + for row in 0..count { + let base = row * dims; + let mut score = 0.0f64; + for (col, &qv) in normalized.iter().enumerate() { + score += f64::from(data[base + col]) * qv; + } + order.push((score, row as u32)); + } + // JS comparator `(a, b) => b.score - a.score` under a stable sort: strict + // score ordering, otherwise (equal, including ±0.0) original row order. + order.sort_by(|a, b| { + let diff = b.0 - a.0; + if diff > 0.0 { + core::cmp::Ordering::Greater + } else if diff < 0.0 { + core::cmp::Ordering::Less + } else { + a.1.cmp(&b.1) + } + }); + let take = (limit as usize).min(order.len()); + order.truncate(take); + let indices: Vec = order.iter().map(|&(_, row)| row).collect(); + let scores: Vec = order.iter().map(|&(score, _)| score).collect(); + Ok(VectorTopK { indices: Uint32Array::new(indices), scores: Float64Array::new(scores) }) +} + +#[inline] +fn hamming_one(query: &[u8], row: &[u8]) -> u32 { + let shared = query.len().min(row.len()); + let mut distance = 0u32; + for i in 0..shared { + distance += (query[i] ^ row[i]).count_ones(); + } + for &byte in &query[shared..] { + distance += byte.count_ones(); + } + for &byte in &row[shared..] { + distance += byte.count_ones(); + } + distance +} + +/// Hamming distance of `query` against a batch of packed binary vectors. +/// +/// `candidates` is flattened row-major at `stride` bytes per row. +/// `lengths[i]` gives the meaningful byte length of row `i` (clamped to +/// `stride`); omit it when every row is exactly `stride` bytes. Semantics +/// match the TS `hammingDistance` exactly, including popcounting the +/// unmatched tail of whichever side is longer. +#[napi] +pub fn hamming_distance_batch( + query: Uint8Array, + candidates: Uint8Array, + stride: u32, + lengths: Option, +) -> Result { + let stride = stride as usize; + let cands: &[u8] = &candidates; + let q: &[u8] = &query; + let count = match &lengths { + Some(lens) => lens.len(), + None if stride == 0 => { + if cands.is_empty() { + return Ok(Uint32Array::new(Vec::new())); + } + return invalid("stride must be positive when lengths is omitted"); + }, + None => { + if cands.len() % stride != 0 { + return invalid("candidates length must be a multiple of stride"); + } + cands.len() / stride + }, + }; + if count * stride > cands.len() { + return invalid("candidates shorter than count * stride"); + } + let mut distances: Vec = Vec::with_capacity(count); + for i in 0..count { + let len = match &lengths { + Some(lens) => (lens[i] as usize).min(stride), + None => stride, + }; + let row = &cands[i * stride..i * stride + len]; + distances.push(hamming_one(q, row)); + } + Ok(Uint32Array::new(distances)) +} + +/// Dimension-masked Hamming distance of `query` against a batch of packed +/// binary vectors. +/// +/// `candidates` is flattened row-major at `stride` bytes per row and +/// `dims[i]` is the bit dimension compared for row `i`. Bytes beyond either +/// side's data read as `0`, and a trailing partial byte is masked to the top +/// `dims[i] % 8` bits — exactly the TS `hammingDistanceForDimension`. +#[napi] +pub fn hamming_distance_for_dim_batch( + query: Uint8Array, + candidates: Uint8Array, + stride: u32, + dims: Uint32Array, +) -> Result { + let stride = stride as usize; + let cands: &[u8] = &candidates; + let q: &[u8] = &query; + let count = dims.len(); + if count * stride > cands.len() { + return invalid("candidates shorter than dims.length * stride"); + } + let mut distances: Vec = Vec::with_capacity(count); + for i in 0..count { + let row = &cands[i * stride..(i + 1) * stride]; + let dim = dims[i] as usize; + let whole_bytes = dim >> 3; + let mut distance = 0u32; + for byte in 0..whole_bytes { + let a = q.get(byte).copied().unwrap_or(0); + let b = row.get(byte).copied().unwrap_or(0); + distance += (a ^ b).count_ones(); + } + let remaining_bits = dim & 7; + if remaining_bits > 0 { + let mask = (0xffu16 << (8 - remaining_bits)) as u8; + let a = q.get(whole_bytes).copied().unwrap_or(0); + let b = row.get(whole_bytes).copied().unwrap_or(0); + distance += ((a ^ b) & mask).count_ones(); + } + distances.push(distance); + } + Ok(Uint32Array::new(distances)) +} + +/// ECMA-262 `\s` (WhiteSpace ∪ LineTerminator), which differs from Rust's +/// `char::is_whitespace` (JS additionally includes U+FEFF). +#[inline] +const fn is_js_whitespace(c: char) -> bool { + matches!( + c, + '\u{0009}' + | '\u{000a}' + | '\u{000b}' + | '\u{000c}' + | '\u{000d}' + | '\u{0020}' + | '\u{00a0}' + | '\u{1680}' + | '\u{2000}' + ..='\u{200a}' + | '\u{2028}' + | '\u{2029}' + | '\u{202f}' + | '\u{205f}' + | '\u{3000}' + | '\u{feff}' + ) +} + +/// Lowercased word set per the TS `jaccardSimilarity` tokenizer: +/// `text.toLowerCase().split(/\s+/).filter(Boolean)` into a `Set`. +/// Returned sorted and deduplicated for merge-based intersection counting. +fn word_set(text: &str) -> Vec> { + let lower = text.to_lowercase(); + let mut words: Vec> = lower + .split(is_js_whitespace) + .filter(|w| !w.is_empty()) + .map(Box::from) + .collect(); + words.sort_unstable(); + words.dedup(); + words +} + +/// Jaccard similarity of two sorted, deduplicated word sets. Matches the TS +/// `jaccardSimilarity`: `0` when either set is empty, otherwise +/// `|A ∩ B| / (|A| + |B| - |A ∩ B|)` with exact integer counts. +fn jaccard_sorted(a: &[Box], b: &[Box]) -> f64 { + if a.is_empty() || b.is_empty() { + return 0.0; + } + let mut intersection = 0usize; + let (mut i, mut j) = (0usize, 0usize); + while i < a.len() && j < b.len() { + match a[i].cmp(&b[j]) { + core::cmp::Ordering::Less => i += 1, + core::cmp::Ordering::Greater => j += 1, + core::cmp::Ordering::Equal => { + intersection += 1; + i += 1; + j += 1; + }, + } + } + intersection as f64 / (a.len() + b.len() - intersection) as f64 +} + +/// MMR selection over pre-sorted candidates using Jaccard word similarity. +/// +/// `contents[i]` and `scores[i]` describe candidate `i`, already sorted by +/// relevance exactly as the TS `mmrRerank` sorts them (the JS stable sort +/// stays on the TS side so its tie and NaN semantics are preserved). +/// Replicates the TS selection loop exactly: candidate `0` is always taken +/// first; each round picks the remaining candidate maximizing +/// `lambda * score - (1 - lambda) * maxSimilarity(selected)` with strict +/// `>` comparisons, so ties keep the earliest remaining candidate — and a +/// round where every score is `NaN` picks the first remaining candidate, +/// matching the TS `bestIdx = 0` initialisation. Returns the selected +/// indices into the input order. +/// +/// Word tokenization matches `text.toLowerCase().split(/\s+/)` (ECMA `\s`, +/// Unicode default full case conversion). Known divergence: unpaired +/// surrogates arrive here as U+FFFD, while JS keeps the lone surrogate; both +/// tokenize to a single non-whitespace word so Jaccard counts still agree +/// unless a text mixes U+FFFD words with lone-surrogate words. +#[napi] +pub fn mmr_rerank_indices( + contents: Vec, + scores: Float64Array, + lambda_param: f64, + top_k: u32, +) -> Result { + if scores.len() != contents.len() { + return invalid("scores length must equal contents length"); + } + let limit = top_k as usize; + let count = contents.len(); + if limit == 0 || count == 0 { + return Ok(Uint32Array::new(Vec::new())); + } + let sets: Vec>> = contents.iter().map(|text| word_set(text)).collect(); + let mut selected: Vec = Vec::with_capacity(limit.min(count)); + selected.push(0); + let mut remaining: Vec = (1..count as u32).collect(); + + while !remaining.is_empty() && selected.len() < limit { + let mut best_idx = 0usize; + let mut best_score = f64::NEG_INFINITY; + for (idx, &candidate) in remaining.iter().enumerate() { + let mut max_similarity = 0.0f64; + for &picked in &selected { + let similarity = jaccard_sorted(&sets[candidate as usize], &sets[picked as usize]); + if similarity > max_similarity { + max_similarity = similarity; + } + } + let relevance = scores[candidate as usize]; + let mmr_score = lambda_param * relevance - (1.0 - lambda_param) * max_similarity; + if mmr_score > best_score { + best_score = mmr_score; + best_idx = idx; + } + } + selected.push(remaining.remove(best_idx)); + } + if selected.len() < limit { + selected.extend(remaining); + selected.truncate(limit); + } + Ok(Uint32Array::new(selected)) +} + +#[cfg(test)] +mod tests { + use super::{cosine_one, hamming_one, is_js_whitespace, jaccard_sorted, word_set}; + + #[test] + fn cosine_matches_reference_semantics() { + assert_eq!(cosine_one(&[], &[]), 0.0); + assert_eq!(cosine_one(&[1.0, 0.0], &[0.0, 0.0]), 0.0); + let same = cosine_one(&[1.0, 2.0, 3.0], &[1.0, 2.0, 3.0]); + assert!((same - 1.0).abs() < 1e-12); + // Non-finite entries are zeroed, mismatched lengths pad with zero. + let sim = cosine_one(&[f64::NAN, 1.0], &[0.5, 1.0, 2.0]); + let expect = 1.0 / (1.0f64.sqrt() * (0.25f64 + 1.0 + 4.0).sqrt()); + assert!((sim - expect).abs() < 1e-12); + } + + #[test] + fn hamming_counts_unmatched_tails() { + assert_eq!(hamming_one(&[0xff, 0x0f], &[0x0f]), 4 + 4); + assert_eq!(hamming_one(&[], &[0xff]), 8); + assert_eq!(hamming_one(&[0b1010], &[0b0101]), 4); + } + + #[test] + fn word_set_matches_js_tokenizer() { + let set = word_set("Hello\u{00a0}WORLD hello\u{feff}world"); + assert_eq!(set, vec![Box::from("hello"), Box::from("world")]); + assert!(word_set("").is_empty()); + assert!(word_set(" \t\n").is_empty()); + assert!(!is_js_whitespace('\u{200b}')); // ZWSP is not JS \s + } + + #[test] + fn jaccard_matches_reference() { + let a = word_set("the quick brown fox"); + let b = word_set("the lazy brown dog"); + let sim = jaccard_sorted(&a, &b); + assert!((sim - 2.0 / 6.0).abs() < 1e-12); + assert_eq!(jaccard_sorted(&a, &word_set("")), 0.0); + } +} diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index c7a42b2ba..f38776c88 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -470,6 +470,28 @@ export interface ContextLine { */ export declare function copyToClipboard(text: string): void +/** + * Cosine similarity of `query` against a batch of candidate vectors. + * + * `candidates` is `n` vectors flattened row-major at `dim` elements per row + * (callers zero-pad shorter vectors, which matches the TS `?? 0` missing + * element semantics). Returns one score per candidate, bit-identical to + * calling the TS `cosineSimilarity(query, candidate)` per row. + */ +export declare function cosineSimilarityBatch(query: Float64Array, candidates: Float64Array, dim: number): Float64Array + +/** + * All pairs `(i, j)` with `i < j` whose cosine similarity meets `threshold`. + * + * `vectors` is `count` vectors flattened row-major at `dim` `f32` elements + * per row (zero-padded). Element values are widened to `f64` before any + * arithmetic, exactly like JS reads from a `Float32Array`, so the similarity + * is bit-identical to the TS pairwise loop in `clusterBySimilarity`. + * Returns pairs flattened as `[i0, j0, i1, j1, ...]` in the same `(i, j)` + * visit order as the TS nested loop. + */ +export declare function cosineSimilarityPairs(vectors: Float32Array, count: number, dim: number, threshold: number): Uint32Array + /** * Count tokens in `input`. * @@ -809,6 +831,28 @@ export interface GrepResult { skippedOversized?: number } +/** + * Hamming distance of `query` against a batch of packed binary vectors. + * + * `candidates` is flattened row-major at `stride` bytes per row. + * `lengths[i]` gives the meaningful byte length of row `i` (clamped to + * `stride`); omit it when every row is exactly `stride` bytes. Semantics + * match the TS `hammingDistance` exactly, including popcounting the + * unmatched tail of whichever side is longer. + */ +export declare function hammingDistanceBatch(query: Uint8Array, candidates: Uint8Array, stride: number, lengths?: Uint32Array | undefined | null): Uint32Array + +/** + * Dimension-masked Hamming distance of `query` against a batch of packed + * binary vectors. + * + * `candidates` is flattened row-major at `stride` bytes per row and + * `dims[i]` is the bit dimension compared for row `i`. Bytes beyond either + * side's data read as `0`, and a trailing partial byte is masked to the top + * `dims[i] % 8` bits — exactly the TS `hammingDistanceForDimension`. + */ +export declare function hammingDistanceForDimBatch(query: Uint8Array, candidates: Uint8Array, stride: number, dims: Uint32Array): Uint32Array + /** * Quick check if content matches a pattern. * @@ -1199,6 +1243,28 @@ export interface MinimizerResult { outputBytes: number } +/** + * MMR selection over pre-sorted candidates using Jaccard word similarity. + * + * `contents[i]` and `scores[i]` describe candidate `i`, already sorted by + * relevance exactly as the TS `mmrRerank` sorts them (the JS stable sort + * stays on the TS side so its tie and NaN semantics are preserved). + * Replicates the TS selection loop exactly: candidate `0` is always taken + * first; each round picks the remaining candidate maximizing + * `lambda * score - (1 - lambda) * maxSimilarity(selected)` with strict + * `>` comparisons, so ties keep the earliest remaining candidate — and a + * round where every score is `NaN` picks the first remaining candidate, + * matching the TS `bestIdx = 0` initialisation. Returns the selected + * indices into the input order. + * + * Word tokenization matches `text.toLowerCase().split(/\s+/)` (ECMA `\s`, + * Unicode default full case conversion). Known divergence: unpaired + * surrogates arrive here as U+FFFD, while JS keeps the lone surrogate; both + * tokenize to a single non-whitespace word so Jaccard counts still agree + * unless a text mixes U+FFFD words with lone-surrogate words. + */ +export declare function mmrRerankIndices(contents: Array, scores: Float64Array, lambdaParam: number, topK: number): Uint32Array + /** Parsed Kitty keyboard protocol sequence result for a Kitty input sequence. */ export interface ParsedKittyResult { /** Primary codepoint associated with the key. */ @@ -1593,6 +1659,31 @@ export declare function supportsLanguage(lang: string): boolean */ export declare function truncateToWidth(text: string, maxWidth: number, ellipsisKind: Ellipsis | undefined | null, pad: boolean | undefined | null, tabWidth: number): string +/** + * Score every row of a normalized `f32` matrix against `query` and return + * the top `limit` rows. + * + * Mirrors the TS `searchExactVectorIndex` loop bit-exactly: the query is + * normalized by the L2 norm of its *full* length, each row score sums + * `matrix[row][col] * (query[col] / norm)` over + * `min(query.len, dimensions)` columns in column order. Ranking matches the + * TS stable sort: score descending, lower row index first on exact ties + * (`-0.0` and `+0.0` compare equal). Callers are expected to enforce the TS + * guards first (finite query with a positive norm, non-empty matrix). + */ +export declare function vectorIndexTopK(matrix: Float32Array, dimensions: number, query: Float64Array, limit: number): VectorTopK + +/** + * Top-k rows of a normalized vector matrix ranked by dot product with a + * normalized query. + */ +export interface VectorTopK { + /** Row indices of the selected hits, best score first. */ + indices: Uint32Array + /** Scores aligned with `indices`. */ + scores: Float64Array +} + /** * Calculate visible width of text, excluding ANSI escape sequences. * diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index a41c6cddb..773db1f60 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -30,6 +30,8 @@ export const astGrep = nativeBindings.astGrep; export const astMatch = nativeBindings.astMatch; export const blockRangeAt = nativeBindings.blockRangeAt; export const copyToClipboard = nativeBindings.copyToClipboard; +export const cosineSimilarityBatch = nativeBindings.cosineSimilarityBatch; +export const cosineSimilarityPairs = nativeBindings.cosineSimilarityPairs; export const countTokens = nativeBindings.countTokens; export const detectMacOSAppearance = nativeBindings.detectMacOSAppearance; export const enclosingBlockBoundaries = nativeBindings.enclosingBlockBoundaries; @@ -41,6 +43,8 @@ export const getSupportedLanguages = nativeBindings.getSupportedLanguages; export const getWorkProfile = nativeBindings.getWorkProfile; export const glob = nativeBindings.glob; export const grep = nativeBindings.grep; +export const hammingDistanceBatch = nativeBindings.hammingDistanceBatch; +export const hammingDistanceForDimBatch = nativeBindings.hammingDistanceForDimBatch; export const hasMatch = nativeBindings.hasMatch; export const highlightCode = nativeBindings.highlightCode; export const htmlToMarkdown = nativeBindings.htmlToMarkdown; @@ -56,6 +60,7 @@ export const listWorkspace = nativeBindings.listWorkspace; export const matchesKey = nativeBindings.matchesKey; export const matchesKittySequence = nativeBindings.matchesKittySequence; export const matchesLegacySequence = nativeBindings.matchesLegacySequence; +export const mmrRerankIndices = nativeBindings.mmrRerankIndices; export const parseKey = nativeBindings.parseKey; export const parseKittySequence = nativeBindings.parseKittySequence; export const readImageFromClipboard = nativeBindings.readImageFromClipboard; @@ -67,6 +72,7 @@ export const snapcompactSupportedChars = nativeBindings.snapcompactSupportedChar export const summarizeCode = nativeBindings.summarizeCode; export const supportsLanguage = nativeBindings.supportsLanguage; export const truncateToWidth = nativeBindings.truncateToWidth; +export const vectorIndexTopK = nativeBindings.vectorIndexTopK; export const visibleWidth = nativeBindings.visibleWidth; export const wrapTextWithAnsi = nativeBindings.wrapTextWithAnsi; From 22a5fb3d9ff9dfbe63049b6bfa36ce16755d66a8 Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 01:24:37 -0700 Subject: [PATCH 02/16] perf(mnemopi): use native batch vector kernels in recall hot loops Swap the batch-shaped scoring loops to the pi-natives kernels behind the existing TS function signatures, so every caller keeps its guards and observable behavior: mmrRerank (default Jaccard path), searchExactVectorIndex, clusterBySimilarity, BinaryVectorStore.search, and FastBinarySearch.search. cosine_similarity_pairs now takes Float64Array so f64 vectors round-trip without f32 narrowing. Scalar one-off call sites stay in TS (query-cache cosine probe, shmr centroid/confidence loops, recall per-row cosine map, custom-similarity mmrRerank). Adds a seeded parity suite: 1e-9 relative tolerance for floats, exact equality for Hamming distances and pair lists, identical MMR index sequences across lambda/topK grids. --- crates/pi-natives/src/vectors.rs | 17 +- .../mnemopi/bench/native-vectors.bench.ts | 122 ++++++++++++ packages/mnemopi/package.json | 1 + packages/mnemopi/src/core/binary-vectors.ts | 44 ++++- packages/mnemopi/src/core/mmr.ts | 16 ++ packages/mnemopi/src/core/shmr.ts | 29 +-- packages/mnemopi/src/core/vector-index.ts | 23 +-- .../mnemopi/test/native-vector-parity.test.ts | 184 ++++++++++++++++++ packages/natives/native/index.d.ts | 13 +- 9 files changed, 402 insertions(+), 47 deletions(-) create mode 100644 packages/mnemopi/bench/native-vectors.bench.ts create mode 100644 packages/mnemopi/test/native-vector-parity.test.ts diff --git a/crates/pi-natives/src/vectors.rs b/crates/pi-natives/src/vectors.rs index 0610e023b..0dc6b66c7 100644 --- a/crates/pi-natives/src/vectors.rs +++ b/crates/pi-natives/src/vectors.rs @@ -92,26 +92,25 @@ pub fn cosine_similarity_batch( /// All pairs `(i, j)` with `i < j` whose cosine similarity meets `threshold`. /// -/// `vectors` is `count` vectors flattened row-major at `dim` `f32` elements -/// per row (zero-padded). Element values are widened to `f64` before any -/// arithmetic, exactly like JS reads from a `Float32Array`, so the similarity -/// is bit-identical to the TS pairwise loop in `clusterBySimilarity`. -/// Returns pairs flattened as `[i0, j0, i1, j1, ...]` in the same `(i, j)` -/// visit order as the TS nested loop. +/// `vectors` is `count` vectors flattened row-major at `dim` `f64` elements +/// per row (zero-padded, which matches the TS `?? 0` missing-element +/// semantics), so the similarity is bit-identical to the TS pairwise loop in +/// `clusterBySimilarity`. Returns pairs flattened as `[i0, j0, i1, j1, ...]` +/// in the same `(i, j)` visit order as the TS nested loop. #[napi] pub fn cosine_similarity_pairs( - vectors: Float32Array, + vectors: Float64Array, count: u32, dim: u32, threshold: f64, ) -> Result { let count = count as usize; let dim = dim as usize; - let data: &[f32] = &vectors; + let data: &[f64] = &vectors; if data.len() != count * dim { return invalid("vectors length must equal count * dim"); } - let widened: Vec = data.iter().map(|&v| f64::from(v)).collect(); + let widened: &[f64] = data; let mut pairs: Vec = Vec::new(); for i in 0..count { let left = &widened[i * dim..(i + 1) * dim]; diff --git a/packages/mnemopi/bench/native-vectors.bench.ts b/packages/mnemopi/bench/native-vectors.bench.ts new file mode 100644 index 000000000..b52de396a --- /dev/null +++ b/packages/mnemopi/bench/native-vectors.bench.ts @@ -0,0 +1,122 @@ +/** + * Crossing-inclusive benchmark of native batch vector kernels vs the TS + * reference loops, at realistic mnemopi recall shapes (fastembed + * bge-small-en-v1.5, dim=384; binarized stride=48 bytes). + * + * Run from the repo root: `bun packages/mnemopi/bench/native-vectors.bench.ts` + */ +import { execSync } from "node:child_process"; +import { cosineSimilarityBatch, hammingDistanceBatch, vectorIndexTopK } from "@oh-my-pi/pi-natives"; +import { hammingDistance } from "../src/core/binary-vectors"; +import { cosineSimilarity } from "../src/core/vector-math"; + +const DIM = 384; +const STRIDE = DIM / 8; +const COUNTS = [10, 100, 1000, 10000]; +const WARMUP = 20; +const ITERATIONS = 200; + +function makeRng(seed: number): () => number { + let state = seed >>> 0; + return () => { + state = (state * 1664525 + 1013904223) >>> 0; + return state / 4294967296; + }; +} + +let sink = 0; + +function timeNs(fn: () => void): number { + for (let i = 0; i < WARMUP; i += 1) fn(); + const start = Bun.nanoseconds(); + for (let i = 0; i < ITERATIONS; i += 1) fn(); + return (Bun.nanoseconds() - start) / ITERATIONS; +} + +interface Row { + kernel: string; + count: number; + tsNs: number; + nativeNs: number; + speedup: number; +} + +const rows: Row[] = []; +const rng = makeRng(0xbe4c4); + +for (const count of COUNTS) { + const query = Float64Array.from({ length: DIM }, () => rng() * 2 - 1); + const flat = new Float64Array(count * DIM); + for (let i = 0; i < flat.length; i += 1) flat[i] = rng() * 2 - 1; + + const tsNs = timeNs(() => { + for (let row = 0; row < count; row += 1) { + sink += cosineSimilarity(query, flat.subarray(row * DIM, (row + 1) * DIM)); + } + }); + const nativeNs = timeNs(() => { + sink += cosineSimilarityBatch(query, flat, DIM)[0] ?? 0; + }); + rows.push({ kernel: "cosineSimilarityBatch", count, tsNs, nativeNs, speedup: tsNs / nativeNs }); +} + +for (const count of COUNTS) { + const matrix = new Float32Array(count * DIM); + for (let i = 0; i < matrix.length; i += 1) matrix[i] = rng() * 2 - 1; + const query = Float64Array.from({ length: DIM }, () => rng() * 2 - 1); + let normSq = 0; + for (const v of query) normSq += v * v; + const norm = Math.sqrt(normSq); + const limit = 10; + + const tsNs = timeNs(() => { + const hits: Array<{ row: number; score: number }> = []; + for (let row = 0; row < count; row += 1) { + let score = 0; + for (let col = 0; col < DIM; col += 1) { + score += (matrix[row * DIM + col] ?? 0) * ((query[col] ?? 0) / norm); + } + hits.push({ row, score }); + } + hits.sort((a, b) => b.score - a.score); + sink += hits[0]?.score ?? 0; + }); + const nativeNs = timeNs(() => { + sink += vectorIndexTopK(matrix, DIM, query, limit).scores[0] ?? 0; + }); + rows.push({ kernel: "vectorIndexTopK", count, tsNs, nativeNs, speedup: tsNs / nativeNs }); +} + +for (const count of COUNTS) { + const query = Uint8Array.from({ length: STRIDE }, () => Math.floor(rng() * 256)); + const packed = new Uint8Array(count * STRIDE); + for (let i = 0; i < packed.length; i += 1) packed[i] = Math.floor(rng() * 256); + const vectors: Uint8Array[] = []; + for (let i = 0; i < count; i += 1) vectors.push(packed.subarray(i * STRIDE, (i + 1) * STRIDE)); + + const tsNs = timeNs(() => { + for (let i = 0; i < count; i += 1) sink += hammingDistance(query, vectors[i] ?? new Uint8Array()); + }); + const nativeNs = timeNs(() => { + sink += hammingDistanceBatch(query, packed, STRIDE)[0] ?? 0; + }); + rows.push({ kernel: "hammingDistanceBatch", count, tsNs, nativeNs, speedup: tsNs / nativeNs }); +} + +const sha = execSync("git rev-parse HEAD").toString().trim(); +const report = { + sha, + date: new Date().toISOString(), + scenario: `dim=${DIM}, stride=${STRIDE}B, warmup=${WARMUP}, iterations=${ITERATIONS}, crossing-inclusive`, + runtime: `bun ${Bun.version}`, + rows: rows.map(r => ({ + kernel: r.kernel, + count: r.count, + ts_us: +(r.tsNs / 1000).toFixed(2), + native_us: +(r.nativeNs / 1000).toFixed(2), + speedup: +r.speedup.toFixed(2), + })), + sink, +}; + +console.log(JSON.stringify(report, null, 2)); diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index 4d5b505af..c38357420 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -41,6 +41,7 @@ "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", + "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", "lru-cache": "catalog:" }, diff --git a/packages/mnemopi/src/core/binary-vectors.ts b/packages/mnemopi/src/core/binary-vectors.ts index 6964a5e70..21d26718c 100644 --- a/packages/mnemopi/src/core/binary-vectors.ts +++ b/packages/mnemopi/src/core/binary-vectors.ts @@ -1,4 +1,5 @@ import type { Database } from "bun:sqlite"; +import { hammingDistanceBatch, hammingDistanceForDimBatch } from "@oh-my-pi/pi-natives"; import { embeddingDim, type VecType } from "../config"; import { closeQuietly, type DatabasePath, openDatabase } from "../db"; @@ -147,7 +148,7 @@ export function hammingDistance(binaryA: Uint8Array | ArrayBuffer, binaryB: Uint return distance; } -function hammingDistanceForDimension( +export function hammingDistanceForDimension( binaryA: Uint8Array | ArrayBuffer, binaryB: Uint8Array | ArrayBuffer, dim: number, @@ -226,15 +227,31 @@ export class BinaryVectorStore { const rows = this.conn .query(`SELECT memory_id, binary_vector, original_dim, magnitude FROM ${this.tableName}`) .all() as VectorRow[]; - const results: BinaryVectorSearchResult[] = []; - for (const row of rows) { + // Native batch kernel: pack all rows at a fixed stride (zero-padded, + // matching the TS `?? 0` out-of-range reads) and compute every + // dimension-masked distance in one N-API crossing. + const stride = BYTES_PER_VECTOR; + const packed = new Uint8Array(rows.length * stride); + const dims = new Uint32Array(rows.length); + const comparedDims: number[] = new Array(rows.length); + for (let i = 0; i < rows.length; i += 1) { + const row = rows[i]; + if (row === undefined) continue; const storedDim = Math.max(0, Math.min(EMBEDDING_DIM, Math.trunc(toFiniteNumber(row.original_dim)))); const comparedDim = Math.min(queryDim, storedDim); - const distance = hammingDistanceForDimension(queryBinary, bytesFromBlob(row.binary_vector), comparedDim); + comparedDims[i] = comparedDim; + dims[i] = comparedDim; + const bytes = bytesFromBlob(row.binary_vector); + packed.set(bytes.length > stride ? bytes.subarray(0, stride) : bytes, i * stride); + } + const distances = hammingDistanceForDimBatch(queryBinary, packed, stride, dims); + const results: BinaryVectorSearchResult[] = []; + for (let i = 0; i < rows.length; i += 1) { + const distance = distances[i] ?? 0; results.push({ - memory_id: row.memory_id, + memory_id: rows[i]?.memory_id ?? "", distance, - score: informationTheoreticScore(distance, comparedDim), + score: informationTheoreticScore(distance, comparedDims[i] ?? 0), }); } results.sort((a, b) => b.score - a.score || a.memory_id.localeCompare(b.memory_id)); @@ -302,9 +319,22 @@ export class FastBinarySearch { search(queryBinary: Uint8Array | ArrayBuffer, topK = 10): BinaryVectorSearchResult[] { const query = queryBinary instanceof Uint8Array ? queryBinary : new Uint8Array(queryBinary); + // Native batch kernel: pack all vectors at the max byte length and + // compute every Hamming distance in one N-API crossing. Per-row + // lengths preserve the TS unmatched-tail popcount semantics. + let stride = 0; + for (const vector of this.vectors) if (vector.length > stride) stride = vector.length; + const packed = new Uint8Array(this.vectors.length * stride); + const lengths = new Uint32Array(this.vectors.length); + for (let i = 0; i < this.vectors.length; i += 1) { + const vector = this.vectors[i] ?? new Uint8Array(); + lengths[i] = vector.length; + packed.set(vector, i * stride); + } + const distances = hammingDistanceBatch(query, packed, stride, lengths); const results: BinaryVectorSearchResult[] = []; for (let i = 0; i < this.vectors.length; i += 1) { - const distance = hammingDistance(query, this.vectors[i] ?? new Uint8Array()); + const distance = distances[i] ?? 0; results.push({ memory_id: this.memoryIds[i] ?? "", distance, diff --git a/packages/mnemopi/src/core/mmr.ts b/packages/mnemopi/src/core/mmr.ts index 6cc87322b..9f706943c 100644 --- a/packages/mnemopi/src/core/mmr.ts +++ b/packages/mnemopi/src/core/mmr.ts @@ -1,3 +1,5 @@ +import { mmrRerankIndices } from "@oh-my-pi/pi-natives"; + export interface MmrResult { readonly content?: string; readonly score?: number; @@ -33,6 +35,20 @@ export function mmrRerank( const first = sortedResults[0]; if (first === undefined) return []; + // Native batch kernel: one N-API crossing selects all indices. Only valid + // for the default Jaccard similarity; custom similarity functions stay in TS. + if (similarityFn === jaccardSimilarity) { + const contents = sortedResults.map(result => result.content ?? ""); + const scores = Float64Array.from(sortedResults, result => result.score ?? 0); + const picked = mmrRerankIndices(contents, scores, lambdaParam, limit); + const out: T[] = []; + for (const index of picked) { + const item = sortedResults[index]; + if (item !== undefined) out.push(item); + } + return out; + } + const selected: T[] = [first]; const remaining = sortedResults.slice(1); diff --git a/packages/mnemopi/src/core/shmr.ts b/packages/mnemopi/src/core/shmr.ts index dca63ba24..e6497ed9b 100644 --- a/packages/mnemopi/src/core/shmr.ts +++ b/packages/mnemopi/src/core/shmr.ts @@ -1,5 +1,6 @@ import type { Database } from "bun:sqlite"; import { createHash } from "node:crypto"; +import { cosineSimilarityPairs } from "@oh-my-pi/pi-natives"; import { logger } from "@oh-my-pi/pi-utils"; import * as embeddings from "./embeddings"; import { cosineSimilarity } from "./vector-math"; @@ -166,17 +167,23 @@ export async function clusterBySimilarity(items: readonly ShmrItem[], threshold: if (items.length === 0) return []; const vectors = await resolveItemVectors(items); const adjacency: number[][] = Array.from({ length: items.length }, () => []); - for (let i = 0; i < items.length; i++) { - const leftEmbedding = vectors[i]; - if (leftEmbedding === undefined) continue; - for (let j = i + 1; j < items.length; j++) { - const rightEmbedding = vectors[j]; - if (rightEmbedding === undefined) continue; - if (cosineSimilarity(leftEmbedding, rightEmbedding) >= threshold) { - adjacency[i]?.push(j); - adjacency[j]?.push(i); - } - } + // Native batch kernel: one N-API crossing evaluates all O(n^2) pairs. + // Vectors are zero-padded to a shared dim, matching the TS `?? 0` + // missing-element semantics of cosineSimilarity. + let dim = 0; + for (const vector of vectors) if (vector.length > dim) dim = vector.length; + const flat = new Float64Array(vectors.length * dim); + for (let i = 0; i < vectors.length; i++) { + const vector = vectors[i]; + if (vector === undefined) continue; + for (let col = 0; col < vector.length; col++) flat[i * dim + col] = vector[col] ?? 0; + } + const pairs = cosineSimilarityPairs(flat, vectors.length, dim, threshold); + for (let p = 0; p < pairs.length; p += 2) { + const i = pairs[p] ?? 0; + const j = pairs[p + 1] ?? 0; + adjacency[i]?.push(j); + adjacency[j]?.push(i); } const visited = new Set(); const clusters: ShmrItem[][] = []; diff --git a/packages/mnemopi/src/core/vector-index.ts b/packages/mnemopi/src/core/vector-index.ts index 3092cba52..f3952635c 100644 --- a/packages/mnemopi/src/core/vector-index.ts +++ b/packages/mnemopi/src/core/vector-index.ts @@ -1,3 +1,5 @@ +import { vectorIndexTopK } from "@oh-my-pi/pi-natives"; + export interface ExactVectorSearchHit { id: TId; score: number; @@ -66,19 +68,14 @@ export function searchExactVectorIndex( queryNormSq += value * value; } if (queryNormSq <= 0) return []; - const queryNorm = Math.sqrt(queryNormSq); - const queryDimensions = Math.min(query.length, index.dimensions); + // Native batch kernel: one N-API crossing scores every row and ranks the + // top k with the same stable ordering as the TS sort. TS guards above + // (finite query, positive norm, non-empty index) are preserved. + const topK = vectorIndexTopK(index.matrix, index.dimensions, Float64Array.from(query), k); const hits: ExactVectorSearchHit[] = []; - - for (let row = 0; row < index.count; row += 1) { - const offset = row * index.dimensions; - let score = 0; - for (let col = 0; col < queryDimensions; col += 1) { - score += index.matrix[offset + col] * ((query[col] ?? 0) / queryNorm); - } - hits.push({ id: index.ids[row] as TId, score }); + for (let i = 0; i < topK.indices.length; i += 1) { + const row = topK.indices[i] ?? 0; + hits.push({ id: index.ids[row] as TId, score: topK.scores[i] ?? 0 }); } - - hits.sort((a, b) => b.score - a.score); - return hits.slice(0, Math.min(k, hits.length)); + return hits; } diff --git a/packages/mnemopi/test/native-vector-parity.test.ts b/packages/mnemopi/test/native-vector-parity.test.ts new file mode 100644 index 000000000..490d7ab09 --- /dev/null +++ b/packages/mnemopi/test/native-vector-parity.test.ts @@ -0,0 +1,184 @@ +import { describe, expect, test } from "bun:test"; +import { + cosineSimilarityBatch, + cosineSimilarityPairs, + hammingDistanceBatch, + hammingDistanceForDimBatch, + mmrRerankIndices, + vectorIndexTopK, +} from "@oh-my-pi/pi-natives"; +import { hammingDistance, hammingDistanceForDimension } from "../src/core/binary-vectors"; +import { jaccardSimilarity } from "../src/core/mmr"; +import { cosineSimilarity } from "../src/core/vector-math"; + +/** Deterministic LCG so parity failures reproduce exactly. */ +function makeRng(seed: number): () => number { + let state = seed >>> 0; + return () => { + state = (state * 1664525 + 1013904223) >>> 0; + return state / 4294967296; + }; +} + +const REL_TOL = 1e-9; + +function expectClose(actual: number, expected: number): void { + if (expected === 0) { + expect(actual).toBe(0); + return; + } + expect(Math.abs(actual - expected) / Math.abs(expected)).toBeLessThanOrEqual(REL_TOL); +} + +describe("native vector kernel parity", () => { + test("cosineSimilarityBatch matches TS cosineSimilarity per row", () => { + const rng = makeRng(0xc051e); + const dim = 384; + const count = 200; + const query = Float64Array.from({ length: dim }, () => rng() * 2 - 1); + const candidates = new Float64Array(count * dim); + for (let i = 0; i < candidates.length; i += 1) candidates[i] = rng() * 2 - 1; + // Sprinkle non-finite values to exercise the finite_or_zero path. + candidates[3] = Number.NaN; + candidates[dim + 7] = Number.POSITIVE_INFINITY; + const scores = cosineSimilarityBatch(query, candidates, dim); + expect(scores.length).toBe(count); + for (let row = 0; row < count; row += 1) { + const expected = cosineSimilarity(query, candidates.subarray(row * dim, (row + 1) * dim)); + expectClose(scores[row] ?? Number.NaN, expected); + } + }); + + test("cosineSimilarityPairs matches the TS pairwise threshold loop", () => { + const rng = makeRng(0x9a175); + const dim = 64; + const count = 40; + const flat = new Float64Array(count * dim); + for (let i = 0; i < flat.length; i += 1) flat[i] = rng() * 2 - 1; + const threshold = 0.02; + const expected: number[] = []; + for (let i = 0; i < count; i += 1) { + for (let j = i + 1; j < count; j += 1) { + if (cosineSimilarity(flat.subarray(i * dim, (i + 1) * dim), flat.subarray(j * dim, (j + 1) * dim)) >= threshold) { + expected.push(i, j); + } + } + } + expect(Array.from(cosineSimilarityPairs(flat, count, dim, threshold))).toEqual(expected); + }); + + test("vectorIndexTopK matches the TS scoring loop and stable sort", () => { + const rng = makeRng(0x70b1); + const dims = 384; + const count = 300; + const matrix = new Float32Array(count * dims); + for (let i = 0; i < matrix.length; i += 1) matrix[i] = rng() * 2 - 1; + const query = Float64Array.from({ length: dims }, () => rng() * 2 - 1); + let normSq = 0; + for (const v of query) normSq += v * v; + const norm = Math.sqrt(normSq); + const hits: Array<{ row: number; score: number }> = []; + for (let row = 0; row < count; row += 1) { + let score = 0; + for (let col = 0; col < dims; col += 1) { + score += (matrix[row * dims + col] ?? 0) * ((query[col] ?? 0) / norm); + } + hits.push({ row, score }); + } + hits.sort((a, b) => b.score - a.score); + const limit = 25; + const result = vectorIndexTopK(matrix, dims, query, limit); + expect(Array.from(result.indices)).toEqual(hits.slice(0, limit).map(h => h.row)); + for (let i = 0; i < limit; i += 1) { + expectClose(result.scores[i] ?? Number.NaN, hits[i]?.score ?? Number.NaN); + } + }); + + test("hammingDistanceBatch is exactly equal to TS hammingDistance", () => { + const rng = makeRng(0xba7c4); + const stride = 48; // 384-dim binarized + const count = 128; + const query = Uint8Array.from({ length: stride }, () => Math.floor(rng() * 256)); + const packed = new Uint8Array(count * stride); + const lengths = new Uint32Array(count); + const vectors: Uint8Array[] = []; + for (let i = 0; i < count; i += 1) { + const len = i % 7 === 0 ? Math.floor(rng() * stride) : stride; // ragged rows + const vector = Uint8Array.from({ length: len }, () => Math.floor(rng() * 256)); + vectors.push(vector); + lengths[i] = len; + packed.set(vector, i * stride); + } + const distances = hammingDistanceBatch(query, packed, stride, lengths); + for (let i = 0; i < count; i += 1) { + expect(distances[i]).toBe(hammingDistance(query, vectors[i] ?? new Uint8Array())); + } + }); + + test("hammingDistanceForDimBatch is exactly equal to TS hammingDistanceForDimension", () => { + const rng = makeRng(0xd1235); + const stride = 48; + const count = 96; + const query = Uint8Array.from({ length: stride }, () => Math.floor(rng() * 256)); + const packed = new Uint8Array(count * stride); + const dims = new Uint32Array(count); + for (let i = 0; i < count; i += 1) { + dims[i] = Math.floor(rng() * (stride * 8 + 1)); // includes partial-byte tails and 0 + for (let b = 0; b < stride; b += 1) packed[i * stride + b] = Math.floor(rng() * 256); + } + const distances = hammingDistanceForDimBatch(query, packed, stride, dims); + for (let i = 0; i < count; i += 1) { + const row = packed.subarray(i * stride, (i + 1) * stride); + expect(distances[i]).toBe(hammingDistanceForDimension(query, row, dims[i] ?? 0)); + } + }); + + test("mmrRerankIndices selects identical index sequences to the TS loop", () => { + const rng = makeRng(0x33a11); + const words = ["alpha", "beta", "gamma", "delta", "epsilon", "zeta", "eta", "theta", "iota", "kappa"]; + const count = 60; + const contents: string[] = []; + const scores = new Float64Array(count); + for (let i = 0; i < count; i += 1) { + const n = 1 + Math.floor(rng() * 8); + contents.push( + Array.from({ length: n }, () => words[Math.floor(rng() * words.length)]) + .join(" "), + ); + scores[i] = rng(); + } + scores[5] = scores[6] = 0.5; // exercise strict-> tie keeping the earlier candidate + for (const lambda of [0.0, 0.3, 0.7, 1.0]) { + for (const topK of [1, 10, count, count + 5]) { + // TS reference selection over pre-sorted candidates (mirrors mmrRerank + // after its sort step). + const order = contents.map((_, i) => i).sort((a, b) => (scores[b] ?? 0) - (scores[a] ?? 0)); + const sortedContents = order.map(i => contents[i] ?? ""); + const sortedScores = order.map(i => scores[i] ?? 0); + const selected: number[] = [0]; + const remaining = sortedContents.map((_, i) => i).slice(1); + while (remaining.length > 0 && selected.length < topK) { + let bestIdx = 0; + let bestScore = Number.NEGATIVE_INFINITY; + for (let idx = 0; idx < remaining.length; idx += 1) { + const candidate = remaining[idx] ?? 0; + let maxSimilarity = 0; + for (const picked of selected) { + const sim = jaccardSimilarity(sortedContents[candidate] ?? "", sortedContents[picked] ?? ""); + if (sim > maxSimilarity) maxSimilarity = sim; + } + const mmrScore = lambda * (sortedScores[candidate] ?? 0) - (1 - lambda) * maxSimilarity; + if (mmrScore > bestScore) { + bestScore = mmrScore; + bestIdx = idx; + } + } + selected.push(remaining.splice(bestIdx, 1)[0] ?? 0); + } + if (selected.length < topK) selected.push(...remaining.slice(0, topK - selected.length)); + const native = mmrRerankIndices(sortedContents, Float64Array.from(sortedScores), lambda, topK); + expect(Array.from(native)).toEqual(selected.slice(0, topK)); + } + } + }); +}); diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index f38776c88..ee5f3df66 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -483,14 +483,13 @@ export declare function cosineSimilarityBatch(query: Float64Array, candidates: F /** * All pairs `(i, j)` with `i < j` whose cosine similarity meets `threshold`. * - * `vectors` is `count` vectors flattened row-major at `dim` `f32` elements - * per row (zero-padded). Element values are widened to `f64` before any - * arithmetic, exactly like JS reads from a `Float32Array`, so the similarity - * is bit-identical to the TS pairwise loop in `clusterBySimilarity`. - * Returns pairs flattened as `[i0, j0, i1, j1, ...]` in the same `(i, j)` - * visit order as the TS nested loop. + * `vectors` is `count` vectors flattened row-major at `dim` `f64` elements + * per row (zero-padded, which matches the TS `?? 0` missing-element + * semantics), so the similarity is bit-identical to the TS pairwise loop in + * `clusterBySimilarity`. Returns pairs flattened as `[i0, j0, i1, j1, ...]` + * in the same `(i, j)` visit order as the TS nested loop. */ -export declare function cosineSimilarityPairs(vectors: Float32Array, count: number, dim: number, threshold: number): Uint32Array +export declare function cosineSimilarityPairs(vectors: Float64Array, count: number, dim: number, threshold: number): Uint32Array /** * Count tokens in `input`. From 7542b31ee8dadc523df98dd4c416b6312eb8bc0c Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 01:25:08 -0700 Subject: [PATCH 03/16] perf(mnemopi): add crossing-inclusive native vector kernel benchmark dim=384 (fastembed bge-small), stride=48B binarized, counts 10/100/1k/10k, 20 warmup + 200 timed iterations per cell, run at 22a5fb3d9. Native wins at every measured size: cosine batch 1.4-2.9x, vectorIndexTopK 1.6-2x, Hamming batch 4.5-10x; no break-even crossover above count=10. --- .../mnemopi/bench/native-vectors.bench.json | 93 +++++++++++++++++++ 1 file changed, 93 insertions(+) create mode 100644 packages/mnemopi/bench/native-vectors.bench.json diff --git a/packages/mnemopi/bench/native-vectors.bench.json b/packages/mnemopi/bench/native-vectors.bench.json new file mode 100644 index 000000000..b85048da0 --- /dev/null +++ b/packages/mnemopi/bench/native-vectors.bench.json @@ -0,0 +1,93 @@ +{ + "sha": "22a5fb3d9ff9dfbe63049b6bfa36ce16755d66a8", + "date": "2026-07-22T08:24:55.683Z", + "scenario": "dim=384, stride=48B, warmup=20, iterations=200, crossing-inclusive", + "runtime": "bun 1.3.14", + "rows": [ + { + "kernel": "cosineSimilarityBatch", + "count": 10, + "ts_us": 10.12, + "native_us": 3.44, + "speedup": 2.94 + }, + { + "kernel": "cosineSimilarityBatch", + "count": 100, + "ts_us": 42.43, + "native_us": 28.18, + "speedup": 1.51 + }, + { + "kernel": "cosineSimilarityBatch", + "count": 1000, + "ts_us": 402.47, + "native_us": 279.5, + "speedup": 1.44 + }, + { + "kernel": "cosineSimilarityBatch", + "count": 10000, + "ts_us": 3977.01, + "native_us": 2821.78, + "speedup": 1.41 + }, + { + "kernel": "vectorIndexTopK", + "count": 10, + "ts_us": 5.52, + "native_us": 3.49, + "speedup": 1.58 + }, + { + "kernel": "vectorIndexTopK", + "count": 100, + "ts_us": 31.87, + "native_us": 17.73, + "speedup": 1.8 + }, + { + "kernel": "vectorIndexTopK", + "count": 1000, + "ts_us": 331.28, + "native_us": 165.43, + "speedup": 2 + }, + { + "kernel": "vectorIndexTopK", + "count": 10000, + "ts_us": 3842.08, + "native_us": 1890.8, + "speedup": 2.03 + }, + { + "kernel": "hammingDistanceBatch", + "count": 10, + "ts_us": 3.03, + "native_us": 0.67, + "speedup": 4.51 + }, + { + "kernel": "hammingDistanceBatch", + "count": 100, + "ts_us": 7.05, + "native_us": 0.71, + "speedup": 9.99 + }, + { + "kernel": "hammingDistanceBatch", + "count": 1000, + "ts_us": 28.76, + "native_us": 4.02, + "speedup": 7.15 + }, + { + "kernel": "hammingDistanceBatch", + "count": 10000, + "ts_us": 230.52, + "native_us": 39.25, + "speedup": 5.87 + } + ], + "sink": 469653256.8853176 +} From 85fae69281fa6cfccb71b92862fb3d1e44455ae8 Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 01:58:23 -0700 Subject: [PATCH 04/16] style(mnemopi): format native vector parity test --- packages/mnemopi/test/native-vector-parity.test.ts | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/packages/mnemopi/test/native-vector-parity.test.ts b/packages/mnemopi/test/native-vector-parity.test.ts index 490d7ab09..decc0cbb2 100644 --- a/packages/mnemopi/test/native-vector-parity.test.ts +++ b/packages/mnemopi/test/native-vector-parity.test.ts @@ -59,7 +59,10 @@ describe("native vector kernel parity", () => { const expected: number[] = []; for (let i = 0; i < count; i += 1) { for (let j = i + 1; j < count; j += 1) { - if (cosineSimilarity(flat.subarray(i * dim, (i + 1) * dim), flat.subarray(j * dim, (j + 1) * dim)) >= threshold) { + if ( + cosineSimilarity(flat.subarray(i * dim, (i + 1) * dim), flat.subarray(j * dim, (j + 1) * dim)) >= + threshold + ) { expected.push(i, j); } } @@ -141,10 +144,7 @@ describe("native vector kernel parity", () => { const scores = new Float64Array(count); for (let i = 0; i < count; i += 1) { const n = 1 + Math.floor(rng() * 8); - contents.push( - Array.from({ length: n }, () => words[Math.floor(rng() * words.length)]) - .join(" "), - ); + contents.push(Array.from({ length: n }, () => words[Math.floor(rng() * words.length)]).join(" ")); scores[i] = rng(); } scores[5] = scores[6] = 0.5; // exercise strict-> tie keeping the earlier candidate From f3bbc23d0baa3a9ad955638892b2561d5d0e7bc8 Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 02:28:30 -0700 Subject: [PATCH 05/16] perf(natives): vectorize dim-masked hamming whole-byte loop --- crates/pi-natives/src/vectors.rs | 18 ++++++++++++++---- 1 file changed, 14 insertions(+), 4 deletions(-) diff --git a/crates/pi-natives/src/vectors.rs b/crates/pi-natives/src/vectors.rs index 0dc6b66c7..512dd0697 100644 --- a/crates/pi-natives/src/vectors.rs +++ b/crates/pi-natives/src/vectors.rs @@ -286,11 +286,21 @@ pub fn hamming_distance_for_dim_batch( let row = &cands[i * stride..(i + 1) * stride]; let dim = dims[i] as usize; let whole_bytes = dim >> 3; + // Bytes beyond either side's data read as 0, so the XOR reduces to a + // plain popcount over whichever side still has data. Sliced loops keep + // the hot path branch-free and autovectorizable (see `hamming_one`). + let q_end = q.len().min(whole_bytes); + let row_end = row.len().min(whole_bytes); + let shared = q_end.min(row_end); let mut distance = 0u32; - for byte in 0..whole_bytes { - let a = q.get(byte).copied().unwrap_or(0); - let b = row.get(byte).copied().unwrap_or(0); - distance += (a ^ b).count_ones(); + for byte in 0..shared { + distance += (q[byte] ^ row[byte]).count_ones(); + } + for &byte in &q[shared..q_end] { + distance += byte.count_ones(); + } + for &byte in &row[shared..row_end] { + distance += byte.count_ones(); } let remaining_bits = dim & 7; if remaining_bits > 0 { From 86ef6636c83c493d75ede6572615aadcef0da60e Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 02:28:30 -0700 Subject: [PATCH 06/16] test(mnemopi): benchmark every swapped native vector kernel with adaptive iterations --- .../mnemopi/bench/native-vectors.bench.ts | 131 +++++++++++++++--- 1 file changed, 115 insertions(+), 16 deletions(-) diff --git a/packages/mnemopi/bench/native-vectors.bench.ts b/packages/mnemopi/bench/native-vectors.bench.ts index b52de396a..c8082825c 100644 --- a/packages/mnemopi/bench/native-vectors.bench.ts +++ b/packages/mnemopi/bench/native-vectors.bench.ts @@ -6,8 +6,15 @@ * Run from the repo root: `bun packages/mnemopi/bench/native-vectors.bench.ts` */ import { execSync } from "node:child_process"; -import { cosineSimilarityBatch, hammingDistanceBatch, vectorIndexTopK } from "@oh-my-pi/pi-natives"; -import { hammingDistance } from "../src/core/binary-vectors"; +import { + cosineSimilarityBatch, + cosineSimilarityPairs, + hammingDistanceBatch, + hammingDistanceForDimBatch, + vectorIndexTopK, +} from "@oh-my-pi/pi-natives"; +import { hammingDistance, hammingDistanceForDimension } from "../src/core/binary-vectors"; +import { jaccardSimilarity, mmrRerank } from "../src/core/mmr"; import { cosineSimilarity } from "../src/core/vector-math"; const DIM = 384; @@ -26,11 +33,11 @@ function makeRng(seed: number): () => number { let sink = 0; -function timeNs(fn: () => void): number { - for (let i = 0; i < WARMUP; i += 1) fn(); +function timeNs(fn: () => void, iterations = ITERATIONS, warmup = WARMUP): { ns: number; iterations: number } { + for (let i = 0; i < warmup; i += 1) fn(); const start = Bun.nanoseconds(); - for (let i = 0; i < ITERATIONS; i += 1) fn(); - return (Bun.nanoseconds() - start) / ITERATIONS; + for (let i = 0; i < iterations; i += 1) fn(); + return { ns: (Bun.nanoseconds() - start) / iterations, iterations }; } interface Row { @@ -39,6 +46,20 @@ interface Row { tsNs: number; nativeNs: number; speedup: number; + tsIterations: number; + nativeIterations: number; +} + +function pushRow(kernel: string, count: number, ts: { ns: number; iterations: number }, native: { ns: number; iterations: number }): void { + rows.push({ + kernel, + count, + tsNs: ts.ns, + nativeNs: native.ns, + speedup: ts.ns / native.ns, + tsIterations: ts.iterations, + nativeIterations: native.iterations, + }); } const rows: Row[] = []; @@ -49,15 +70,15 @@ for (const count of COUNTS) { const flat = new Float64Array(count * DIM); for (let i = 0; i < flat.length; i += 1) flat[i] = rng() * 2 - 1; - const tsNs = timeNs(() => { + const ts = timeNs(() => { for (let row = 0; row < count; row += 1) { sink += cosineSimilarity(query, flat.subarray(row * DIM, (row + 1) * DIM)); } }); - const nativeNs = timeNs(() => { + const native = timeNs(() => { sink += cosineSimilarityBatch(query, flat, DIM)[0] ?? 0; }); - rows.push({ kernel: "cosineSimilarityBatch", count, tsNs, nativeNs, speedup: tsNs / nativeNs }); + pushRow("cosineSimilarityBatch", count, ts, native); } for (const count of COUNTS) { @@ -69,7 +90,7 @@ for (const count of COUNTS) { const norm = Math.sqrt(normSq); const limit = 10; - const tsNs = timeNs(() => { + const ts = timeNs(() => { const hits: Array<{ row: number; score: number }> = []; for (let row = 0; row < count; row += 1) { let score = 0; @@ -81,10 +102,10 @@ for (const count of COUNTS) { hits.sort((a, b) => b.score - a.score); sink += hits[0]?.score ?? 0; }); - const nativeNs = timeNs(() => { + const native = timeNs(() => { sink += vectorIndexTopK(matrix, DIM, query, limit).scores[0] ?? 0; }); - rows.push({ kernel: "vectorIndexTopK", count, tsNs, nativeNs, speedup: tsNs / nativeNs }); + pushRow("vectorIndexTopK", count, ts, native); } for (const count of COUNTS) { @@ -94,20 +115,96 @@ for (const count of COUNTS) { const vectors: Uint8Array[] = []; for (let i = 0; i < count; i += 1) vectors.push(packed.subarray(i * STRIDE, (i + 1) * STRIDE)); - const tsNs = timeNs(() => { + const ts = timeNs(() => { for (let i = 0; i < count; i += 1) sink += hammingDistance(query, vectors[i] ?? new Uint8Array()); }); - const nativeNs = timeNs(() => { + const native = timeNs(() => { sink += hammingDistanceBatch(query, packed, STRIDE)[0] ?? 0; }); - rows.push({ kernel: "hammingDistanceBatch", count, tsNs, nativeNs, speedup: tsNs / nativeNs }); + pushRow("hammingDistanceBatch", count, ts, native); +} + +// cosineSimilarityPairs: O(n²) pair scan — TS baseline is the pre-native shmr +// clustering loop. Capped at 1k candidates and given adaptive iterations (the +// TS side at n=1000 is ~500k pair cosines × 384 dims per run). +for (const count of COUNTS.filter(n => n <= 1000)) { + const flat = new Float64Array(count * DIM); + for (let i = 0; i < flat.length; i += 1) flat[i] = rng() * 2 - 1; + const threshold = 0.15; + + const pairIterations = count >= 1000 ? 10 : ITERATIONS; + const pairWarmup = count >= 1000 ? 2 : WARMUP; + const ts = timeNs( + () => { + let pairs = 0; + for (let i = 0; i < count; i += 1) { + const a = flat.subarray(i * DIM, (i + 1) * DIM); + for (let j = i + 1; j < count; j += 1) { + if (cosineSimilarity(a, flat.subarray(j * DIM, (j + 1) * DIM)) >= threshold) pairs += 1; + } + } + sink += pairs; + }, + pairIterations, + pairWarmup, + ); + const native = timeNs(() => { + sink += cosineSimilarityPairs(flat, count, DIM, threshold).length; + }, pairIterations, pairWarmup); + pushRow("cosineSimilarityPairs", count, ts, native); +} + +// hammingDistanceForDimBatch: ragged/dim-masked variant used by BinaryVectorStore.search. +for (const count of COUNTS) { + const query = Uint8Array.from({ length: STRIDE }, () => Math.floor(rng() * 256)); + const packed = new Uint8Array(count * STRIDE); + for (let i = 0; i < packed.length; i += 1) packed[i] = Math.floor(rng() * 256); + const dims = Uint32Array.from({ length: count }, () => (rng() < 0.5 ? DIM : DIM / 2)); + const vectors: Uint8Array[] = []; + for (let i = 0; i < count; i += 1) vectors.push(packed.subarray(i * STRIDE, (i + 1) * STRIDE)); + + const ts = timeNs(() => { + for (let i = 0; i < count; i += 1) { + sink += hammingDistanceForDimension(query, vectors[i] ?? new Uint8Array(), dims[i] ?? DIM); + } + }); + const native = timeNs(() => { + sink += hammingDistanceForDimBatch(query, packed, STRIDE, dims)[0] ?? 0; + }); + pushRow("hammingDistanceForDimBatch", count, ts, native); +} + +// mmrRerank production paths: the TS side wraps jaccardSimilarity in a lambda, +// defeating the identity check so the exact pre-native selection loop runs; the +// native side calls mmrRerank with the default similarity, exercising the real +// fast path including its sort and wrapper overhead. +const tsJaccard = (a: string, b: string): number => jaccardSimilarity(a, b); +for (const count of COUNTS.filter(n => n <= 1000)) { + const words = ["alpha", "beta", "gamma", "delta", "epsilon", "zeta", "eta", "theta", "iota", "kappa"]; + const results: Array<{ content: string; score: number }> = []; + for (let i = 0; i < count; i += 1) { + const n = 5 + Math.floor(rng() * 20); + results.push({ + content: Array.from({ length: n }, () => words[Math.floor(rng() * words.length)]).join(" "), + score: rng(), + }); + } + const topK = 10; + + const ts = timeNs(() => { + sink += mmrRerank(results, 0.7, topK, tsJaccard).length; + }); + const native = timeNs(() => { + sink += mmrRerank(results, 0.7, topK).length; + }); + pushRow("mmrRerankIndices (via mmrRerank)", count, ts, native); } const sha = execSync("git rev-parse HEAD").toString().trim(); const report = { sha, date: new Date().toISOString(), - scenario: `dim=${DIM}, stride=${STRIDE}B, warmup=${WARMUP}, iterations=${ITERATIONS}, crossing-inclusive`, + scenario: `dim=${DIM}, stride=${STRIDE}B, warmup=${WARMUP}, iterations=${ITERATIONS} (adaptive for O(n²) rows, see per-row fields), crossing-inclusive`, runtime: `bun ${Bun.version}`, rows: rows.map(r => ({ kernel: r.kernel, @@ -115,6 +212,8 @@ const report = { ts_us: +(r.tsNs / 1000).toFixed(2), native_us: +(r.nativeNs / 1000).toFixed(2), speedup: +r.speedup.toFixed(2), + ts_iterations: r.tsIterations, + native_iterations: r.nativeIterations, })), sink, }; From 46e296637f54cbef3ea9c1156a074e92f24f7a49 Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 02:29:55 -0700 Subject: [PATCH 07/16] refactor(mnemopi): use Bun.spawnSync for bench sha capture --- packages/mnemopi/bench/native-vectors.bench.ts | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/packages/mnemopi/bench/native-vectors.bench.ts b/packages/mnemopi/bench/native-vectors.bench.ts index c8082825c..462689dcf 100644 --- a/packages/mnemopi/bench/native-vectors.bench.ts +++ b/packages/mnemopi/bench/native-vectors.bench.ts @@ -5,7 +5,6 @@ * * Run from the repo root: `bun packages/mnemopi/bench/native-vectors.bench.ts` */ -import { execSync } from "node:child_process"; import { cosineSimilarityBatch, cosineSimilarityPairs, @@ -200,7 +199,7 @@ for (const count of COUNTS.filter(n => n <= 1000)) { pushRow("mmrRerankIndices (via mmrRerank)", count, ts, native); } -const sha = execSync("git rev-parse HEAD").toString().trim(); +const sha = Bun.spawnSync(["git", "rev-parse", "HEAD"]).stdout.toString().trim(); const report = { sha, date: new Date().toISOString(), From c11083c15bab2972ef12ac2bb23047bd06113bed Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 02:30:25 -0700 Subject: [PATCH 08/16] test(mnemopi): refresh native vector bench artifact --- .../mnemopi/bench/native-vectors.bench.json | 194 ++++++++++++++---- 1 file changed, 154 insertions(+), 40 deletions(-) diff --git a/packages/mnemopi/bench/native-vectors.bench.json b/packages/mnemopi/bench/native-vectors.bench.json index b85048da0..297e72be7 100644 --- a/packages/mnemopi/bench/native-vectors.bench.json +++ b/packages/mnemopi/bench/native-vectors.bench.json @@ -1,93 +1,207 @@ { - "sha": "22a5fb3d9ff9dfbe63049b6bfa36ce16755d66a8", - "date": "2026-07-22T08:24:55.683Z", - "scenario": "dim=384, stride=48B, warmup=20, iterations=200, crossing-inclusive", + "sha": "46e296637f54cbef3ea9c1156a074e92f24f7a49", + "date": "2026-07-22T09:30:25.203Z", + "scenario": "dim=384, stride=48B, warmup=20, iterations=200 (adaptive for O(n²) rows, see per-row fields), crossing-inclusive", "runtime": "bun 1.3.14", "rows": [ { "kernel": "cosineSimilarityBatch", "count": 10, - "ts_us": 10.12, - "native_us": 3.44, - "speedup": 2.94 + "ts_us": 11.43, + "native_us": 3.54, + "speedup": 3.23, + "ts_iterations": 200, + "native_iterations": 200 }, { "kernel": "cosineSimilarityBatch", "count": 100, - "ts_us": 42.43, - "native_us": 28.18, - "speedup": 1.51 + "ts_us": 47.43, + "native_us": 28.11, + "speedup": 1.69, + "ts_iterations": 200, + "native_iterations": 200 }, { "kernel": "cosineSimilarityBatch", "count": 1000, - "ts_us": 402.47, - "native_us": 279.5, - "speedup": 1.44 + "ts_us": 453.25, + "native_us": 278.19, + "speedup": 1.63, + "ts_iterations": 200, + "native_iterations": 200 }, { "kernel": "cosineSimilarityBatch", "count": 10000, - "ts_us": 3977.01, - "native_us": 2821.78, - "speedup": 1.41 + "ts_us": 4880.56, + "native_us": 2809.26, + "speedup": 1.74, + "ts_iterations": 200, + "native_iterations": 200 }, { "kernel": "vectorIndexTopK", "count": 10, - "ts_us": 5.52, - "native_us": 3.49, - "speedup": 1.58 + "ts_us": 5.95, + "native_us": 3.38, + "speedup": 1.76, + "ts_iterations": 200, + "native_iterations": 200 }, { "kernel": "vectorIndexTopK", "count": 100, - "ts_us": 31.87, - "native_us": 17.73, - "speedup": 1.8 + "ts_us": 36.07, + "native_us": 19.65, + "speedup": 1.84, + "ts_iterations": 200, + "native_iterations": 200 }, { "kernel": "vectorIndexTopK", "count": 1000, - "ts_us": 331.28, - "native_us": 165.43, - "speedup": 2 + "ts_us": 382.6, + "native_us": 186.72, + "speedup": 2.05, + "ts_iterations": 200, + "native_iterations": 200 }, { "kernel": "vectorIndexTopK", "count": 10000, - "ts_us": 3842.08, - "native_us": 1890.8, - "speedup": 2.03 + "ts_us": 4318, + "native_us": 2014.29, + "speedup": 2.14, + "ts_iterations": 200, + "native_iterations": 200 }, { "kernel": "hammingDistanceBatch", "count": 10, - "ts_us": 3.03, - "native_us": 0.67, - "speedup": 4.51 + "ts_us": 3.17, + "native_us": 0.65, + "speedup": 4.88, + "ts_iterations": 200, + "native_iterations": 200 }, { "kernel": "hammingDistanceBatch", "count": 100, - "ts_us": 7.05, - "native_us": 0.71, - "speedup": 9.99 + "ts_us": 7.63, + "native_us": 0.8, + "speedup": 9.6, + "ts_iterations": 200, + "native_iterations": 200 }, { "kernel": "hammingDistanceBatch", "count": 1000, - "ts_us": 28.76, - "native_us": 4.02, - "speedup": 7.15 + "ts_us": 28.5, + "native_us": 4.18, + "speedup": 6.81, + "ts_iterations": 200, + "native_iterations": 200 }, { "kernel": "hammingDistanceBatch", "count": 10000, - "ts_us": 230.52, - "native_us": 39.25, - "speedup": 5.87 + "ts_us": 260.8, + "native_us": 36.34, + "speedup": 7.18, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "cosineSimilarityPairs", + "count": 10, + "ts_us": 19.57, + "native_us": 12, + "speedup": 1.63, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "cosineSimilarityPairs", + "count": 100, + "ts_us": 2158.12, + "native_us": 1293.04, + "speedup": 1.67, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "cosineSimilarityPairs", + "count": 1000, + "ts_us": 217904.4, + "native_us": 136833.13, + "speedup": 1.59, + "ts_iterations": 10, + "native_iterations": 10 + }, + { + "kernel": "hammingDistanceForDimBatch", + "count": 10, + "ts_us": 2.69, + "native_us": 0.85, + "speedup": 3.16, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "hammingDistanceForDimBatch", + "count": 100, + "ts_us": 6.03, + "native_us": 0.91, + "speedup": 6.63, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "hammingDistanceForDimBatch", + "count": 1000, + "ts_us": 26.41, + "native_us": 5.32, + "speedup": 4.96, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "hammingDistanceForDimBatch", + "count": 10000, + "ts_us": 227.07, + "native_us": 74.1, + "speedup": 3.06, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "mmrRerankIndices (via mmrRerank)", + "count": 10, + "ts_us": 352.01, + "native_us": 14.7, + "speedup": 23.94, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "mmrRerankIndices (via mmrRerank)", + "count": 100, + "ts_us": 8222.8, + "native_us": 219.59, + "speedup": 37.45, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "mmrRerankIndices (via mmrRerank)", + "count": 1000, + "ts_us": 82410.2, + "native_us": 2504.3, + "speedup": 32.91, + "ts_iterations": 200, + "native_iterations": 200 } ], - "sink": 469653256.8853176 + "sink": 823409556.8853176 } From 4de678f44d1d56495f5f09de3d725af8b68d88fa Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 02:32:35 -0700 Subject: [PATCH 09/16] chore(mnemopi): add changelog entries for native vector kernels --- packages/mnemopi/CHANGELOG.md | 4 ++++ packages/natives/CHANGELOG.md | 4 ++++ 2 files changed, 8 insertions(+) diff --git a/packages/mnemopi/CHANGELOG.md b/packages/mnemopi/CHANGELOG.md index dee554a9e..fffe635f2 100644 --- a/packages/mnemopi/CHANGELOG.md +++ b/packages/mnemopi/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Changed + +- Recall hot loops (exact vector-index search, binary vector search, SHMR similarity clustering, and default-similarity MMR rerank) now run on native batch kernels with one N-API crossing per operation. Crossing-inclusive speedups at dim=384: 1.4-3.4x cosine scoring, ~2x top-K search, 2.7-11x Hamming search, and 30-38x MMR rerank. Custom `similarityFn` MMR reranks and incremental per-row scoring stay on the TypeScript implementations. + ## [17.0.4] - 2026-07-18 ### Fixed diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index 2409ace54..d316eaf1a 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added batch vector kernels for mnemopi recall paths: `cosineSimilarityBatch`, `cosineSimilarityPairs`, `vectorIndexTopK`, `hammingDistanceBatch`, `hammingDistanceForDimBatch`, and `mmrRerankIndices`. + ## [17.0.5] - 2026-07-18 ### Added From 4ba69285dfb939e4fc802fa66c3eac76cd075b3b Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 02:43:40 -0700 Subject: [PATCH 10/16] style(natives): satisfy clippy for vector kernels without breaking float parity --- crates/pi-natives/src/vectors.rs | 22 +++++++++++++++++----- 1 file changed, 17 insertions(+), 5 deletions(-) diff --git a/crates/pi-natives/src/vectors.rs b/crates/pi-natives/src/vectors.rs index 512dd0697..8094065db 100644 --- a/crates/pi-natives/src/vectors.rs +++ b/crates/pi-natives/src/vectors.rs @@ -18,7 +18,7 @@ fn invalid(message: &str) -> Result { } #[inline] -fn finite_or_zero(value: f64) -> f64 { +const fn finite_or_zero(value: f64) -> f64 { if value.is_finite() { value } else { 0.0 } } @@ -30,6 +30,10 @@ fn finite_or_zero(value: f64) -> f64 { /// terms of the shorter side only ever add `±0.0` to `dot` and `+0.0` to its /// own norm, in the same index order as the TS loop. #[inline] +#[allow( + clippy::suboptimal_flops, + reason = "mul_add rounds differently; bit-exact with the TS loops is the contract" +)] fn cosine_one(a: &[f64], b: &[f64]) -> f64 { if a.is_empty() && b.is_empty() { return 0.0; @@ -79,7 +83,7 @@ pub fn cosine_similarity_batch( } return Ok(Float64Array::new(Vec::new())); } - if cands.len() % dim != 0 { + if !cands.len().is_multiple_of(dim) { return invalid("candidates length must be a multiple of dim"); } let q: &[f64] = &query; @@ -146,6 +150,10 @@ pub struct VectorTopK { /// (`-0.0` and `+0.0` compare equal). Callers are expected to enforce the TS /// guards first (finite query with a positive norm, non-empty matrix). #[napi] +#[allow( + clippy::suboptimal_flops, + reason = "mul_add rounds differently; bit-exact with the TS loops is the contract" +)] pub fn vector_index_top_k( matrix: Float32Array, dimensions: u32, @@ -154,7 +162,7 @@ pub fn vector_index_top_k( ) -> Result { let dims = dimensions as usize; let data: &[f32] = &matrix; - if dims == 0 || data.len() % dims != 0 { + if dims == 0 || !data.len().is_multiple_of(dims) { return invalid("matrix length must be a positive multiple of dimensions"); } let count = data.len() / dims; @@ -239,7 +247,7 @@ pub fn hamming_distance_batch( return invalid("stride must be positive when lengths is omitted"); }, None => { - if cands.len() % stride != 0 { + if !cands.len().is_multiple_of(stride) { return invalid("candidates length must be a multiple of stride"); } cands.len() / stride @@ -314,7 +322,7 @@ pub fn hamming_distance_for_dim_batch( Ok(Uint32Array::new(distances)) } -/// ECMA-262 `\s` (WhiteSpace ∪ LineTerminator), which differs from Rust's +/// ECMA-262 `\s` (`WhiteSpace` ∪ `LineTerminator`), which differs from Rust's /// `char::is_whitespace` (JS additionally includes U+FEFF). #[inline] const fn is_js_whitespace(c: char) -> bool { @@ -396,6 +404,10 @@ fn jaccard_sorted(a: &[Box], b: &[Box]) -> f64 { /// tokenize to a single non-whitespace word so Jaccard counts still agree /// unless a text mixes U+FFFD words with lone-surrogate words. #[napi] +#[allow( + clippy::suboptimal_flops, + reason = "mul_add rounds differently; bit-exact with the TS loops is the contract" +)] pub fn mmr_rerank_indices( contents: Vec, scores: Float64Array, From 29338e1508019c9425f98f158f9df807f254e4d6 Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 02:44:31 -0700 Subject: [PATCH 11/16] test(mnemopi): refresh bench artifact at final kernel revision --- packages/mnemopi/CHANGELOG.md | 2 +- .../mnemopi/bench/native-vectors.bench.json | 158 +++++++++--------- 2 files changed, 80 insertions(+), 80 deletions(-) diff --git a/packages/mnemopi/CHANGELOG.md b/packages/mnemopi/CHANGELOG.md index fffe635f2..e0cc8db53 100644 --- a/packages/mnemopi/CHANGELOG.md +++ b/packages/mnemopi/CHANGELOG.md @@ -4,7 +4,7 @@ ### Changed -- Recall hot loops (exact vector-index search, binary vector search, SHMR similarity clustering, and default-similarity MMR rerank) now run on native batch kernels with one N-API crossing per operation. Crossing-inclusive speedups at dim=384: 1.4-3.4x cosine scoring, ~2x top-K search, 2.7-11x Hamming search, and 30-38x MMR rerank. Custom `similarityFn` MMR reranks and incremental per-row scoring stay on the TypeScript implementations. +- Recall hot loops (exact vector-index search, binary vector search, SHMR similarity clustering, and default-similarity MMR rerank) now run on native batch kernels with one N-API crossing per operation. Crossing-inclusive speedups at dim=384 (see `bench/native-vectors.bench.json`): 1.6-3.2x cosine scoring, 1.8-2x top-K search, 1.6x pairwise clustering scans, 3-10x Hamming search, and 21-39x MMR rerank. Custom `similarityFn` MMR reranks and incremental per-row scoring stay on the TypeScript implementations. ## [17.0.4] - 2026-07-18 diff --git a/packages/mnemopi/bench/native-vectors.bench.json b/packages/mnemopi/bench/native-vectors.bench.json index 297e72be7..ee78ed7d5 100644 --- a/packages/mnemopi/bench/native-vectors.bench.json +++ b/packages/mnemopi/bench/native-vectors.bench.json @@ -1,204 +1,204 @@ { - "sha": "46e296637f54cbef3ea9c1156a074e92f24f7a49", - "date": "2026-07-22T09:30:25.203Z", + "sha": "4ba69285dfb939e4fc802fa66c3eac76cd075b3b", + "date": "2026-07-22T09:44:06.872Z", "scenario": "dim=384, stride=48B, warmup=20, iterations=200 (adaptive for O(n²) rows, see per-row fields), crossing-inclusive", "runtime": "bun 1.3.14", "rows": [ { "kernel": "cosineSimilarityBatch", "count": 10, - "ts_us": 11.43, - "native_us": 3.54, - "speedup": 3.23, + "ts_us": 10.61, + "native_us": 3.35, + "speedup": 3.16, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "cosineSimilarityBatch", "count": 100, - "ts_us": 47.43, - "native_us": 28.11, - "speedup": 1.69, + "ts_us": 42.48, + "native_us": 25.04, + "speedup": 1.7, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "cosineSimilarityBatch", "count": 1000, - "ts_us": 453.25, - "native_us": 278.19, - "speedup": 1.63, + "ts_us": 470.41, + "native_us": 248.07, + "speedup": 1.9, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "cosineSimilarityBatch", "count": 10000, - "ts_us": 4880.56, - "native_us": 2809.26, - "speedup": 1.74, + "ts_us": 3963.14, + "native_us": 2454.54, + "speedup": 1.61, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "vectorIndexTopK", "count": 10, - "ts_us": 5.95, - "native_us": 3.38, - "speedup": 1.76, + "ts_us": 5.59, + "native_us": 2.97, + "speedup": 1.88, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "vectorIndexTopK", "count": 100, - "ts_us": 36.07, - "native_us": 19.65, - "speedup": 1.84, + "ts_us": 31.47, + "native_us": 17.37, + "speedup": 1.81, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "vectorIndexTopK", "count": 1000, - "ts_us": 382.6, - "native_us": 186.72, - "speedup": 2.05, + "ts_us": 325.12, + "native_us": 166.88, + "speedup": 1.95, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "vectorIndexTopK", "count": 10000, - "ts_us": 4318, - "native_us": 2014.29, - "speedup": 2.14, + "ts_us": 3774.16, + "native_us": 1869.57, + "speedup": 2.02, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "hammingDistanceBatch", "count": 10, - "ts_us": 3.17, - "native_us": 0.65, - "speedup": 4.88, + "ts_us": 3.1, + "native_us": 0.59, + "speedup": 5.27, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "hammingDistanceBatch", "count": 100, - "ts_us": 7.63, - "native_us": 0.8, - "speedup": 9.6, + "ts_us": 7.61, + "native_us": 0.74, + "speedup": 10.24, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "hammingDistanceBatch", "count": 1000, - "ts_us": 28.5, - "native_us": 4.18, - "speedup": 6.81, + "ts_us": 29.27, + "native_us": 3.83, + "speedup": 7.64, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "hammingDistanceBatch", "count": 10000, - "ts_us": 260.8, - "native_us": 36.34, - "speedup": 7.18, + "ts_us": 252.75, + "native_us": 39.02, + "speedup": 6.48, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "cosineSimilarityPairs", "count": 10, - "ts_us": 19.57, - "native_us": 12, - "speedup": 1.63, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "cosineSimilarityPairs", - "count": 100, - "ts_us": 2158.12, - "native_us": 1293.04, - "speedup": 1.67, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "cosineSimilarityPairs", - "count": 1000, - "ts_us": 217904.4, - "native_us": 136833.13, + "ts_us": 18.55, + "native_us": 11.64, "speedup": 1.59, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "cosineSimilarityPairs", + "count": 100, + "ts_us": 1955.97, + "native_us": 1201.4, + "speedup": 1.63, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "cosineSimilarityPairs", + "count": 1000, + "ts_us": 197715.54, + "native_us": 122119.79, + "speedup": 1.62, "ts_iterations": 10, "native_iterations": 10 }, { "kernel": "hammingDistanceForDimBatch", "count": 10, - "ts_us": 2.69, - "native_us": 0.85, - "speedup": 3.16, + "ts_us": 2.45, + "native_us": 0.6, + "speedup": 4.12, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "hammingDistanceForDimBatch", "count": 100, - "ts_us": 6.03, - "native_us": 0.91, - "speedup": 6.63, + "ts_us": 5.83, + "native_us": 0.92, + "speedup": 6.33, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "hammingDistanceForDimBatch", "count": 1000, - "ts_us": 26.41, - "native_us": 5.32, - "speedup": 4.96, + "ts_us": 21.69, + "native_us": 5.16, + "speedup": 4.2, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "hammingDistanceForDimBatch", "count": 10000, - "ts_us": 227.07, - "native_us": 74.1, - "speedup": 3.06, + "ts_us": 202.24, + "native_us": 68.09, + "speedup": 2.97, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "mmrRerankIndices (via mmrRerank)", "count": 10, - "ts_us": 352.01, - "native_us": 14.7, - "speedup": 23.94, + "ts_us": 302.81, + "native_us": 14.38, + "speedup": 21.05, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "mmrRerankIndices (via mmrRerank)", "count": 100, - "ts_us": 8222.8, - "native_us": 219.59, - "speedup": 37.45, + "ts_us": 7775.68, + "native_us": 197.42, + "speedup": 39.39, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "mmrRerankIndices (via mmrRerank)", "count": 1000, - "ts_us": 82410.2, - "native_us": 2504.3, - "speedup": 32.91, + "ts_us": 73795.1, + "native_us": 2331.18, + "speedup": 31.66, "ts_iterations": 200, "native_iterations": 200 } From 6b9726552b400b3c6ed9e7a5e2cd625888634307 Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 03:21:13 -0700 Subject: [PATCH 12/16] chore(mnemopi): add changelog attribution links --- packages/mnemopi/CHANGELOG.md | 2 +- packages/natives/CHANGELOG.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/packages/mnemopi/CHANGELOG.md b/packages/mnemopi/CHANGELOG.md index e0cc8db53..cbbc59cd3 100644 --- a/packages/mnemopi/CHANGELOG.md +++ b/packages/mnemopi/CHANGELOG.md @@ -4,7 +4,7 @@ ### Changed -- Recall hot loops (exact vector-index search, binary vector search, SHMR similarity clustering, and default-similarity MMR rerank) now run on native batch kernels with one N-API crossing per operation. Crossing-inclusive speedups at dim=384 (see `bench/native-vectors.bench.json`): 1.6-3.2x cosine scoring, 1.8-2x top-K search, 1.6x pairwise clustering scans, 3-10x Hamming search, and 21-39x MMR rerank. Custom `similarityFn` MMR reranks and incremental per-row scoring stay on the TypeScript implementations. +- Recall hot loops (exact vector-index search, binary vector search, SHMR similarity clustering, and default-similarity MMR rerank) now run on native batch kernels with one N-API crossing per operation. Crossing-inclusive speedups at dim=384 (see `bench/native-vectors.bench.json`): 1.6-3.2x cosine scoring, 1.8-2x top-K search, 1.6x pairwise clustering scans, 3-10x Hamming search, and 21-39x MMR rerank. Custom `similarityFn` MMR reranks and incremental per-row scoring stay on the TypeScript implementations ([#6280](https://github.com/can1357/oh-my-pi/pull/6280) by [@wolfiesch](https://github.com/wolfiesch)). ## [17.0.4] - 2026-07-18 diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index d316eaf1a..a9d8b5619 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -4,7 +4,7 @@ ### Added -- Added batch vector kernels for mnemopi recall paths: `cosineSimilarityBatch`, `cosineSimilarityPairs`, `vectorIndexTopK`, `hammingDistanceBatch`, `hammingDistanceForDimBatch`, and `mmrRerankIndices`. +- Added batch vector kernels for mnemopi recall paths: `cosineSimilarityBatch`, `cosineSimilarityPairs`, `vectorIndexTopK`, `hammingDistanceBatch`, `hammingDistanceForDimBatch`, and `mmrRerankIndices` ([#6280](https://github.com/can1357/oh-my-pi/pull/6280) by [@wolfiesch](https://github.com/wolfiesch)). ## [17.0.5] - 2026-07-18 From 3c84e2c410afae163fb1a778a4f7011354cc893c Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 04:10:17 -0700 Subject: [PATCH 13/16] fix(mnemopi): clamp native limits, preserve JS lowercase semantics, wrapper-level bench --- .../mnemopi/bench/native-vectors.bench.ts | 108 +++++++++++++----- packages/mnemopi/src/core/mmr.ts | 22 +++- packages/mnemopi/src/core/vector-index.ts | 6 +- .../mnemopi/test/native-vector-parity.test.ts | 58 +++++++++- 4 files changed, 157 insertions(+), 37 deletions(-) diff --git a/packages/mnemopi/bench/native-vectors.bench.ts b/packages/mnemopi/bench/native-vectors.bench.ts index 462689dcf..c292334c1 100644 --- a/packages/mnemopi/bench/native-vectors.bench.ts +++ b/packages/mnemopi/bench/native-vectors.bench.ts @@ -10,10 +10,10 @@ import { cosineSimilarityPairs, hammingDistanceBatch, hammingDistanceForDimBatch, - vectorIndexTopK, } from "@oh-my-pi/pi-natives"; import { hammingDistance, hammingDistanceForDimension } from "../src/core/binary-vectors"; import { jaccardSimilarity, mmrRerank } from "../src/core/mmr"; +import { searchExactVectorIndex } from "../src/core/vector-index"; import { cosineSimilarity } from "../src/core/vector-math"; const DIM = 384; @@ -80,80 +80,121 @@ for (const count of COUNTS) { pushRow("cosineSimilarityBatch", count, ts, native); } +// vectorIndexTopK is measured through the public wrapper searchExactVectorIndex, +// so the native side pays the production per-call query conversion, guards, and +// hit-array construction. The TS side replicates the pre-native wrapper body. for (const count of COUNTS) { const matrix = new Float32Array(count * DIM); for (let i = 0; i < matrix.length; i += 1) matrix[i] = rng() * 2 - 1; - const query = Float64Array.from({ length: DIM }, () => rng() * 2 - 1); - let normSq = 0; - for (const v of query) normSq += v * v; - const norm = Math.sqrt(normSq); + const queryArr: number[] = Array.from({ length: DIM }, () => rng() * 2 - 1); + const ids: number[] = Array.from({ length: count }, (_v, i) => i); + const index = { ids, matrix, dimensions: DIM, count }; const limit = 10; const ts = timeNs(() => { - const hits: Array<{ row: number; score: number }> = []; + let queryNormSq = 0; + for (const value of queryArr) queryNormSq += value * value; + const queryNorm = Math.sqrt(queryNormSq); + const hits: Array<{ id: number; score: number }> = []; for (let row = 0; row < count; row += 1) { + const offset = row * DIM; let score = 0; for (let col = 0; col < DIM; col += 1) { - score += (matrix[row * DIM + col] ?? 0) * ((query[col] ?? 0) / norm); + score += (matrix[offset + col] ?? 0) * ((queryArr[col] ?? 0) / queryNorm); } - hits.push({ row, score }); + hits.push({ id: ids[row] ?? 0, score }); } hits.sort((a, b) => b.score - a.score); - sink += hits[0]?.score ?? 0; + sink += hits.slice(0, limit)[0]?.score ?? 0; }); const native = timeNs(() => { - sink += vectorIndexTopK(matrix, DIM, query, limit).scores[0] ?? 0; + sink += searchExactVectorIndex(index, queryArr, limit)[0]?.score ?? 0; }); - pushRow("vectorIndexTopK", count, ts, native); + pushRow("searchExactVectorIndex (topK wrapper)", count, ts, native); } +// hammingDistanceBatch: the native side pays FastBinarySearch.search's per-call +// packing (stride scan, packed/lengths allocation, full copy) before crossing. for (const count of COUNTS) { const query = Uint8Array.from({ length: STRIDE }, () => Math.floor(rng() * 256)); - const packed = new Uint8Array(count * STRIDE); - for (let i = 0; i < packed.length; i += 1) packed[i] = Math.floor(rng() * 256); + const backing = new Uint8Array(count * STRIDE); + for (let i = 0; i < backing.length; i += 1) backing[i] = Math.floor(rng() * 256); const vectors: Uint8Array[] = []; - for (let i = 0; i < count; i += 1) vectors.push(packed.subarray(i * STRIDE, (i + 1) * STRIDE)); + for (let i = 0; i < count; i += 1) vectors.push(backing.subarray(i * STRIDE, (i + 1) * STRIDE)); const ts = timeNs(() => { for (let i = 0; i < count; i += 1) sink += hammingDistance(query, vectors[i] ?? new Uint8Array()); }); const native = timeNs(() => { - sink += hammingDistanceBatch(query, packed, STRIDE)[0] ?? 0; + let stride = 0; + for (const vector of vectors) if (vector.length > stride) stride = vector.length; + const packed = new Uint8Array(vectors.length * stride); + const lengths = new Uint32Array(vectors.length); + for (let i = 0; i < vectors.length; i += 1) { + const vector = vectors[i] ?? new Uint8Array(); + lengths[i] = vector.length; + packed.set(vector, i * stride); + } + sink += hammingDistanceBatch(query, packed, stride, lengths)[0] ?? 0; }); - pushRow("hammingDistanceBatch", count, ts, native); + pushRow("hammingDistanceBatch (incl. packing)", count, ts, native); } -// cosineSimilarityPairs: O(n²) pair scan — TS baseline is the pre-native shmr -// clustering loop. Capped at 1k candidates and given adaptive iterations (the -// TS side at n=1000 is ~500k pair cosines × 384 dims per run). +// cosineSimilarityPairs: O(n²) pair scan. The TS baseline is the pre-native +// clusterBySimilarity loop over per-item vectors; the native side pays the +// production flatten (dim scan + Float64Array fill) before crossing. Capped at +// 1k candidates with adaptive iterations (the TS side at n=1000 is ~500k pair +// cosines × 384 dims per run). for (const count of COUNTS.filter(n => n <= 1000)) { - const flat = new Float64Array(count * DIM); - for (let i = 0; i < flat.length; i += 1) flat[i] = rng() * 2 - 1; + const vectors: number[][] = Array.from({ length: count }, () => Array.from({ length: DIM }, () => rng() * 2 - 1)); const threshold = 0.15; const pairIterations = count >= 1000 ? 10 : ITERATIONS; const pairWarmup = count >= 1000 ? 2 : WARMUP; const ts = timeNs( () => { - let pairs = 0; + // Pre-native clusterBySimilarity: adjacency unions built inline + // during the pair scan. + const adjacency: number[][] = Array.from({ length: count }, () => []); for (let i = 0; i < count; i += 1) { - const a = flat.subarray(i * DIM, (i + 1) * DIM); + const a = vectors[i] ?? []; for (let j = i + 1; j < count; j += 1) { - if (cosineSimilarity(a, flat.subarray(j * DIM, (j + 1) * DIM)) >= threshold) pairs += 1; + if (cosineSimilarity(a, vectors[j] ?? []) >= threshold) { + adjacency[i]?.push(j); + adjacency[j]?.push(i); + } } } - sink += pairs; + sink += adjacency[0]?.length ?? 0; }, pairIterations, pairWarmup, ); const native = timeNs(() => { - sink += cosineSimilarityPairs(flat, count, DIM, threshold).length; + // Current clusterBySimilarity: flatten, one crossing, then adjacency + // built from the materialized pair list. + let dim = 0; + for (const vector of vectors) if (vector.length > dim) dim = vector.length; + const flat = new Float64Array(vectors.length * dim); + for (let i = 0; i < vectors.length; i += 1) { + const vector = vectors[i]; + if (vector === undefined) continue; + for (let col = 0; col < vector.length; col += 1) flat[i * dim + col] = vector[col] ?? 0; + } + const pairs = cosineSimilarityPairs(flat, vectors.length, dim, threshold); + const adjacency: number[][] = Array.from({ length: count }, () => []); + for (let p = 0; p < pairs.length; p += 2) { + adjacency[pairs[p] ?? 0]?.push(pairs[p + 1] ?? 0); + adjacency[pairs[p + 1] ?? 0]?.push(pairs[p] ?? 0); + } + sink += adjacency[0]?.length ?? 0; }, pairIterations, pairWarmup); - pushRow("cosineSimilarityPairs", count, ts, native); + pushRow("cosineSimilarityPairs (incl. flatten+adjacency)", count, ts, native); } -// hammingDistanceForDimBatch: ragged/dim-masked variant used by BinaryVectorStore.search. +// hammingDistanceForDimBatch: ragged/dim-masked variant used by +// BinaryVectorStore.search; the native side pays the production per-call +// packed/dims allocation and copy before crossing. for (const count of COUNTS) { const query = Uint8Array.from({ length: STRIDE }, () => Math.floor(rng() * 256)); const packed = new Uint8Array(count * STRIDE); @@ -168,9 +209,16 @@ for (const count of COUNTS) { } }); const native = timeNs(() => { - sink += hammingDistanceForDimBatch(query, packed, STRIDE, dims)[0] ?? 0; + const packed2 = new Uint8Array(vectors.length * STRIDE); + const dims2 = new Uint32Array(vectors.length); + for (let i = 0; i < vectors.length; i += 1) { + const vector = vectors[i] ?? new Uint8Array(); + dims2[i] = dims[i] ?? DIM; + packed2.set(vector, i * STRIDE); + } + sink += hammingDistanceForDimBatch(query, packed2, STRIDE, dims2)[0] ?? 0; }); - pushRow("hammingDistanceForDimBatch", count, ts, native); + pushRow("hammingDistanceForDimBatch (incl. packing)", count, ts, native); } // mmrRerank production paths: the TS side wraps jaccardSimilarity in a lambda, diff --git a/packages/mnemopi/src/core/mmr.ts b/packages/mnemopi/src/core/mmr.ts index 9f706943c..aae5662cf 100644 --- a/packages/mnemopi/src/core/mmr.ts +++ b/packages/mnemopi/src/core/mmr.ts @@ -36,11 +36,25 @@ export function mmrRerank( if (first === undefined) return []; // Native batch kernel: one N-API crossing selects all indices. Only valid - // for the default Jaccard similarity; custom similarity functions stay in TS. - if (similarityFn === jaccardSimilarity) { - const contents = sortedResults.map(result => result.content ?? ""); + // for the default Jaccard similarity; custom similarity functions stay in + // TS. Lone-surrogate contents also stay in TS: N-API converts them to + // U+FFFD, which would merge distinct tokens. NaN topK stays in TS so the + // pre-native contract (loop guard is false, first result still returned) + // is preserved. + if ( + similarityFn === jaccardSimilarity && + !Number.isNaN(limit) && + sortedResults.every(result => (result.content ?? "").isWellFormed()) + ) { + // Pre-lowercase with JS semantics so contextual mappings (Final_Sigma: + // "ΟΣ" -> "ος") are applied before the context-insensitive native + // lowercase, which is idempotent on already-lowercased text. + const contents = sortedResults.map(result => (result.content ?? "").toLowerCase()); const scores = Float64Array.from(sortedResults, result => result.score ?? 0); - const picked = mmrRerankIndices(contents, scores, lambdaParam, limit); + // Clamp before the u32 N-API boundary: Infinity or >= 2**32 would + // otherwise wrap (ToUint32) and silently return nothing. + const nativeLimit = Math.min(limit, sortedResults.length); + const picked = mmrRerankIndices(contents, scores, lambdaParam, nativeLimit); const out: T[] = []; for (const index of picked) { const item = sortedResults[index]; diff --git a/packages/mnemopi/src/core/vector-index.ts b/packages/mnemopi/src/core/vector-index.ts index f3952635c..3c43607bd 100644 --- a/packages/mnemopi/src/core/vector-index.ts +++ b/packages/mnemopi/src/core/vector-index.ts @@ -70,8 +70,10 @@ export function searchExactVectorIndex( if (queryNormSq <= 0) return []; // Native batch kernel: one N-API crossing scores every row and ranks the // top k with the same stable ordering as the TS sort. TS guards above - // (finite query, positive norm, non-empty index) are preserved. - const topK = vectorIndexTopK(index.matrix, index.dimensions, Float64Array.from(query), k); + // (finite query, positive norm, non-empty index) are preserved. Clamp k to + // the row count before the u32 boundary: Infinity or >= 2**32 would + // otherwise wrap (ToUint32) and return no hits. + const topK = vectorIndexTopK(index.matrix, index.dimensions, Float64Array.from(query), Math.min(k, index.count)); const hits: ExactVectorSearchHit[] = []; for (let i = 0; i < topK.indices.length; i += 1) { const row = topK.indices[i] ?? 0; diff --git a/packages/mnemopi/test/native-vector-parity.test.ts b/packages/mnemopi/test/native-vector-parity.test.ts index decc0cbb2..da0499932 100644 --- a/packages/mnemopi/test/native-vector-parity.test.ts +++ b/packages/mnemopi/test/native-vector-parity.test.ts @@ -8,7 +8,8 @@ import { vectorIndexTopK, } from "@oh-my-pi/pi-natives"; import { hammingDistance, hammingDistanceForDimension } from "../src/core/binary-vectors"; -import { jaccardSimilarity } from "../src/core/mmr"; +import { jaccardSimilarity, mmrRerank } from "../src/core/mmr"; +import { buildExactVectorIndex, searchExactVectorIndex } from "../src/core/vector-index"; import { cosineSimilarity } from "../src/core/vector-math"; /** Deterministic LCG so parity failures reproduce exactly. */ @@ -181,4 +182,59 @@ describe("native vector kernel parity", () => { } } }); + + test("mmrRerank wrapper preserves the pre-native limit contract at u32 boundaries", () => { + const results = Array.from({ length: 8 }, (_v, i) => ({ + content: `item ${i} alpha beta`, + score: (8 - i) / 10, + })); + const all = mmrRerank(results, 0.7, results.length); + // Infinity and >= 2**32 previously walked every candidate; ToUint32 + // would have collapsed them to 0/1 without the TS-side clamp. + expect(mmrRerank(results, 0.7, Number.POSITIVE_INFINITY)).toEqual(all); + expect(mmrRerank(results, 0.7, 2 ** 32)).toEqual(all); + expect(mmrRerank(results, 0.7, 2 ** 32 + 1)).toEqual(all); + // NaN: loop guard is false but the first result is already selected. + expect(mmrRerank(results, 0.7, Number.NaN)).toEqual([results[0] as (typeof results)[number]]); + expect(mmrRerank(results, 0.7, 0)).toEqual([]); + expect(mmrRerank(results, 0.7, -3)).toEqual([]); + }); + + test("mmrRerank matches the TS path on contextual-lowercase and lone-surrogate content", () => { + // Force the TS selection loop by defeating the identity check. + const tsJaccard = (a: string, b: string): number => jaccardSimilarity(a, b); + // Final_Sigma: JS lowercases "ΟΣ" to "ος"; a context-insensitive + // lowercase would produce "οσ" and score these words as distinct. + const sigma = [ + { content: "ΟΣ", score: 0.9 }, + { content: "ος", score: 0.8 }, + { content: "other words entirely", score: 0.7 }, + ]; + for (const lambda of [0, 0.3, 0.7]) { + expect(mmrRerank(sigma, lambda, 2)).toEqual(mmrRerank(sigma, lambda, 2, tsJaccard)); + } + // Lone surrogates route to the TS path: N-API would convert them to + // U+FFFD and merge the first two tokens. + const surrogate = [ + { content: "\ud800", score: 0.9 }, + { content: "\ufffd", score: 0.8 }, + { content: "other", score: 0.7 }, + ]; + expect(mmrRerank(surrogate, 0, 2)).toEqual(mmrRerank(surrogate, 0, 2, tsJaccard)); + }); + + test("searchExactVectorIndex preserves the pre-native limit contract at u32 boundaries", () => { + const rows = Array.from({ length: 6 }, (_v, i) => ({ + id: i, + vector: Array.from({ length: 8 }, (_x, j) => Math.sin(i * 8 + j)), + })); + const index = buildExactVectorIndex(rows); + const query = Array.from({ length: 8 }, (_x, j) => Math.cos(j)); + const all = searchExactVectorIndex(index, query, index.count); + expect(all.length).toBe(index.count); + expect(searchExactVectorIndex(index, query, Number.POSITIVE_INFINITY)).toEqual(all); + expect(searchExactVectorIndex(index, query, 2 ** 32)).toEqual(all); + expect(searchExactVectorIndex(index, query, Number.NaN)).toEqual([]); + expect(searchExactVectorIndex(index, query, 0)).toEqual([]); + }); }); From 4ffe1822d4a966e772dadb67b6319a652757d234 Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 04:45:13 -0700 Subject: [PATCH 14/16] refactor(mnemopi): keep only production-backed native kernels after wrapper-level benchmarking --- crates/pi-natives/src/vectors.rs | 159 +--------- .../mnemopi/bench/native-vectors.bench.json | 288 +++++++++--------- .../mnemopi/bench/native-vectors.bench.ts | 75 ----- packages/mnemopi/src/core/binary-vectors.ts | 44 +-- .../mnemopi/test/native-vector-parity.test.ts | 67 +--- packages/natives/native/index.d.ts | 32 -- packages/natives/native/index.js | 3 - 7 files changed, 154 insertions(+), 514 deletions(-) diff --git a/crates/pi-natives/src/vectors.rs b/crates/pi-natives/src/vectors.rs index 8094065db..5dbbc127c 100644 --- a/crates/pi-natives/src/vectors.rs +++ b/crates/pi-natives/src/vectors.rs @@ -9,7 +9,7 @@ use napi::{ Error, Result, Status, - bindgen_prelude::{Float32Array, Float64Array, Uint8Array, Uint32Array}, + bindgen_prelude::{Float32Array, Float64Array, Uint32Array}, }; use napi_derive::napi; @@ -63,37 +63,6 @@ fn cosine_one(a: &[f64], b: &[f64]) -> f64 { dot / (norm_a.sqrt() * norm_b.sqrt()) } -/// Cosine similarity of `query` against a batch of candidate vectors. -/// -/// `candidates` is `n` vectors flattened row-major at `dim` elements per row -/// (callers zero-pad shorter vectors, which matches the TS `?? 0` missing -/// element semantics). Returns one score per candidate, bit-identical to -/// calling the TS `cosineSimilarity(query, candidate)` per row. -#[napi] -pub fn cosine_similarity_batch( - query: Float64Array, - candidates: Float64Array, - dim: u32, -) -> Result { - let dim = dim as usize; - let cands: &[f64] = &candidates; - if dim == 0 { - if !cands.is_empty() { - return invalid("candidates must be empty when dim is 0"); - } - return Ok(Float64Array::new(Vec::new())); - } - if !cands.len().is_multiple_of(dim) { - return invalid("candidates length must be a multiple of dim"); - } - let q: &[f64] = &query; - let scores: Vec = cands - .chunks_exact(dim) - .map(|row| cosine_one(q, row)) - .collect(); - Ok(Float64Array::new(scores)) -} - /// All pairs `(i, j)` with `i < j` whose cosine similarity meets `threshold`. /// /// `vectors` is `count` vectors flattened row-major at `dim` `f64` elements @@ -205,123 +174,6 @@ pub fn vector_index_top_k( Ok(VectorTopK { indices: Uint32Array::new(indices), scores: Float64Array::new(scores) }) } -#[inline] -fn hamming_one(query: &[u8], row: &[u8]) -> u32 { - let shared = query.len().min(row.len()); - let mut distance = 0u32; - for i in 0..shared { - distance += (query[i] ^ row[i]).count_ones(); - } - for &byte in &query[shared..] { - distance += byte.count_ones(); - } - for &byte in &row[shared..] { - distance += byte.count_ones(); - } - distance -} - -/// Hamming distance of `query` against a batch of packed binary vectors. -/// -/// `candidates` is flattened row-major at `stride` bytes per row. -/// `lengths[i]` gives the meaningful byte length of row `i` (clamped to -/// `stride`); omit it when every row is exactly `stride` bytes. Semantics -/// match the TS `hammingDistance` exactly, including popcounting the -/// unmatched tail of whichever side is longer. -#[napi] -pub fn hamming_distance_batch( - query: Uint8Array, - candidates: Uint8Array, - stride: u32, - lengths: Option, -) -> Result { - let stride = stride as usize; - let cands: &[u8] = &candidates; - let q: &[u8] = &query; - let count = match &lengths { - Some(lens) => lens.len(), - None if stride == 0 => { - if cands.is_empty() { - return Ok(Uint32Array::new(Vec::new())); - } - return invalid("stride must be positive when lengths is omitted"); - }, - None => { - if !cands.len().is_multiple_of(stride) { - return invalid("candidates length must be a multiple of stride"); - } - cands.len() / stride - }, - }; - if count * stride > cands.len() { - return invalid("candidates shorter than count * stride"); - } - let mut distances: Vec = Vec::with_capacity(count); - for i in 0..count { - let len = match &lengths { - Some(lens) => (lens[i] as usize).min(stride), - None => stride, - }; - let row = &cands[i * stride..i * stride + len]; - distances.push(hamming_one(q, row)); - } - Ok(Uint32Array::new(distances)) -} - -/// Dimension-masked Hamming distance of `query` against a batch of packed -/// binary vectors. -/// -/// `candidates` is flattened row-major at `stride` bytes per row and -/// `dims[i]` is the bit dimension compared for row `i`. Bytes beyond either -/// side's data read as `0`, and a trailing partial byte is masked to the top -/// `dims[i] % 8` bits — exactly the TS `hammingDistanceForDimension`. -#[napi] -pub fn hamming_distance_for_dim_batch( - query: Uint8Array, - candidates: Uint8Array, - stride: u32, - dims: Uint32Array, -) -> Result { - let stride = stride as usize; - let cands: &[u8] = &candidates; - let q: &[u8] = &query; - let count = dims.len(); - if count * stride > cands.len() { - return invalid("candidates shorter than dims.length * stride"); - } - let mut distances: Vec = Vec::with_capacity(count); - for i in 0..count { - let row = &cands[i * stride..(i + 1) * stride]; - let dim = dims[i] as usize; - let whole_bytes = dim >> 3; - // Bytes beyond either side's data read as 0, so the XOR reduces to a - // plain popcount over whichever side still has data. Sliced loops keep - // the hot path branch-free and autovectorizable (see `hamming_one`). - let q_end = q.len().min(whole_bytes); - let row_end = row.len().min(whole_bytes); - let shared = q_end.min(row_end); - let mut distance = 0u32; - for byte in 0..shared { - distance += (q[byte] ^ row[byte]).count_ones(); - } - for &byte in &q[shared..q_end] { - distance += byte.count_ones(); - } - for &byte in &row[shared..row_end] { - distance += byte.count_ones(); - } - let remaining_bits = dim & 7; - if remaining_bits > 0 { - let mask = (0xffu16 << (8 - remaining_bits)) as u8; - let a = q.get(whole_bytes).copied().unwrap_or(0); - let b = row.get(whole_bytes).copied().unwrap_or(0); - distance += ((a ^ b) & mask).count_ones(); - } - distances.push(distance); - } - Ok(Uint32Array::new(distances)) -} - /// ECMA-262 `\s` (`WhiteSpace` ∪ `LineTerminator`), which differs from Rust's /// `char::is_whitespace` (JS additionally includes U+FEFF). #[inline] @@ -456,7 +308,7 @@ pub fn mmr_rerank_indices( #[cfg(test)] mod tests { - use super::{cosine_one, hamming_one, is_js_whitespace, jaccard_sorted, word_set}; + use super::{cosine_one, is_js_whitespace, jaccard_sorted, word_set}; #[test] fn cosine_matches_reference_semantics() { @@ -470,13 +322,6 @@ mod tests { assert!((sim - expect).abs() < 1e-12); } - #[test] - fn hamming_counts_unmatched_tails() { - assert_eq!(hamming_one(&[0xff, 0x0f], &[0x0f]), 4 + 4); - assert_eq!(hamming_one(&[], &[0xff]), 8); - assert_eq!(hamming_one(&[0b1010], &[0b0101]), 4); - } - #[test] fn word_set_matches_js_tokenizer() { let set = word_set("Hello\u{00a0}WORLD hello\u{feff}world"); diff --git a/packages/mnemopi/bench/native-vectors.bench.json b/packages/mnemopi/bench/native-vectors.bench.json index ee78ed7d5..7d781530e 100644 --- a/packages/mnemopi/bench/native-vectors.bench.json +++ b/packages/mnemopi/bench/native-vectors.bench.json @@ -1,207 +1,207 @@ { - "sha": "4ba69285dfb939e4fc802fa66c3eac76cd075b3b", - "date": "2026-07-22T09:44:06.872Z", + "sha": "3c84e2c410afae163fb1a778a4f7011354cc893c", + "date": "2026-07-22T11:33:29.804Z", "scenario": "dim=384, stride=48B, warmup=20, iterations=200 (adaptive for O(n²) rows, see per-row fields), crossing-inclusive", "runtime": "bun 1.3.14", "rows": [ { "kernel": "cosineSimilarityBatch", "count": 10, - "ts_us": 10.61, - "native_us": 3.35, - "speedup": 3.16, + "ts_us": 9.85, + "native_us": 3.2, + "speedup": 3.08, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "cosineSimilarityBatch", "count": 100, - "ts_us": 42.48, - "native_us": 25.04, - "speedup": 1.7, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "cosineSimilarityBatch", - "count": 1000, - "ts_us": 470.41, - "native_us": 248.07, - "speedup": 1.9, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "cosineSimilarityBatch", - "count": 10000, - "ts_us": 3963.14, - "native_us": 2454.54, - "speedup": 1.61, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "vectorIndexTopK", - "count": 10, - "ts_us": 5.59, - "native_us": 2.97, - "speedup": 1.88, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "vectorIndexTopK", - "count": 100, - "ts_us": 31.47, - "native_us": 17.37, - "speedup": 1.81, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "vectorIndexTopK", - "count": 1000, - "ts_us": 325.12, - "native_us": 166.88, - "speedup": 1.95, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "vectorIndexTopK", - "count": 10000, - "ts_us": 3774.16, - "native_us": 1869.57, - "speedup": 2.02, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "hammingDistanceBatch", - "count": 10, - "ts_us": 3.1, - "native_us": 0.59, - "speedup": 5.27, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "hammingDistanceBatch", - "count": 100, - "ts_us": 7.61, - "native_us": 0.74, - "speedup": 10.24, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "hammingDistanceBatch", - "count": 1000, - "ts_us": 29.27, - "native_us": 3.83, - "speedup": 7.64, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "hammingDistanceBatch", - "count": 10000, - "ts_us": 252.75, - "native_us": 39.02, - "speedup": 6.48, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "cosineSimilarityPairs", - "count": 10, - "ts_us": 18.55, - "native_us": 11.64, - "speedup": 1.59, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "cosineSimilarityPairs", - "count": 100, - "ts_us": 1955.97, - "native_us": 1201.4, + "ts_us": 41.15, + "native_us": 25.18, "speedup": 1.63, "ts_iterations": 200, "native_iterations": 200 }, { - "kernel": "cosineSimilarityPairs", + "kernel": "cosineSimilarityBatch", "count": 1000, - "ts_us": 197715.54, - "native_us": 122119.79, + "ts_us": 398.55, + "native_us": 242.88, + "speedup": 1.64, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "cosineSimilarityBatch", + "count": 10000, + "ts_us": 3936.21, + "native_us": 2431.53, "speedup": 1.62, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "searchExactVectorIndex (topK wrapper)", + "count": 10, + "ts_us": 5.82, + "native_us": 6.44, + "speedup": 0.9, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "searchExactVectorIndex (topK wrapper)", + "count": 100, + "ts_us": 36.99, + "native_us": 19.48, + "speedup": 1.9, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "searchExactVectorIndex (topK wrapper)", + "count": 1000, + "ts_us": 310.52, + "native_us": 165.81, + "speedup": 1.87, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "searchExactVectorIndex (topK wrapper)", + "count": 10000, + "ts_us": 3513.57, + "native_us": 1884.54, + "speedup": 1.86, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "hammingDistanceBatch (incl. packing)", + "count": 10, + "ts_us": 3.07, + "native_us": 1.39, + "speedup": 2.21, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "hammingDistanceBatch (incl. packing)", + "count": 100, + "ts_us": 7.09, + "native_us": 3.97, + "speedup": 1.79, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "hammingDistanceBatch (incl. packing)", + "count": 1000, + "ts_us": 34.25, + "native_us": 26.11, + "speedup": 1.31, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "hammingDistanceBatch (incl. packing)", + "count": 10000, + "ts_us": 221.43, + "native_us": 207.45, + "speedup": 1.07, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "cosineSimilarityPairs (incl. flatten+adjacency)", + "count": 10, + "ts_us": 246.34, + "native_us": 22.08, + "speedup": 11.16, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "cosineSimilarityPairs (incl. flatten+adjacency)", + "count": 100, + "ts_us": 5767.19, + "native_us": 1381.02, + "speedup": 4.18, + "ts_iterations": 200, + "native_iterations": 200 + }, + { + "kernel": "cosineSimilarityPairs (incl. flatten+adjacency)", + "count": 1000, + "ts_us": 530658.9, + "native_us": 131050.53, + "speedup": 4.05, "ts_iterations": 10, "native_iterations": 10 }, { - "kernel": "hammingDistanceForDimBatch", + "kernel": "hammingDistanceForDimBatch (incl. packing)", "count": 10, - "ts_us": 2.45, - "native_us": 0.6, - "speedup": 4.12, + "ts_us": 2.52, + "native_us": 1.95, + "speedup": 1.29, "ts_iterations": 200, "native_iterations": 200 }, { - "kernel": "hammingDistanceForDimBatch", + "kernel": "hammingDistanceForDimBatch (incl. packing)", "count": 100, - "ts_us": 5.83, - "native_us": 0.92, - "speedup": 6.33, + "ts_us": 5.88, + "native_us": 4.16, + "speedup": 1.41, "ts_iterations": 200, "native_iterations": 200 }, { - "kernel": "hammingDistanceForDimBatch", + "kernel": "hammingDistanceForDimBatch (incl. packing)", "count": 1000, - "ts_us": 21.69, - "native_us": 5.16, - "speedup": 4.2, + "ts_us": 27.69, + "native_us": 28.69, + "speedup": 0.97, "ts_iterations": 200, "native_iterations": 200 }, { - "kernel": "hammingDistanceForDimBatch", + "kernel": "hammingDistanceForDimBatch (incl. packing)", "count": 10000, - "ts_us": 202.24, - "native_us": 68.09, - "speedup": 2.97, + "ts_us": 231.52, + "native_us": 235.98, + "speedup": 0.98, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "mmrRerankIndices (via mmrRerank)", "count": 10, - "ts_us": 302.81, - "native_us": 14.38, - "speedup": 21.05, + "ts_us": 331.56, + "native_us": 15.1, + "speedup": 21.95, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "mmrRerankIndices (via mmrRerank)", "count": 100, - "ts_us": 7775.68, - "native_us": 197.42, - "speedup": 39.39, + "ts_us": 7965.61, + "native_us": 220.67, + "speedup": 36.1, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "mmrRerankIndices (via mmrRerank)", "count": 1000, - "ts_us": 73795.1, - "native_us": 2331.18, - "speedup": 31.66, + "ts_us": 78042.9, + "native_us": 2639.47, + "speedup": 29.57, "ts_iterations": 200, "native_iterations": 200 } ], - "sink": 823409556.8853176 + "sink": 823377224.8853176 } diff --git a/packages/mnemopi/bench/native-vectors.bench.ts b/packages/mnemopi/bench/native-vectors.bench.ts index c292334c1..c6ae57cf7 100644 --- a/packages/mnemopi/bench/native-vectors.bench.ts +++ b/packages/mnemopi/bench/native-vectors.bench.ts @@ -6,12 +6,8 @@ * Run from the repo root: `bun packages/mnemopi/bench/native-vectors.bench.ts` */ import { - cosineSimilarityBatch, cosineSimilarityPairs, - hammingDistanceBatch, - hammingDistanceForDimBatch, } from "@oh-my-pi/pi-natives"; -import { hammingDistance, hammingDistanceForDimension } from "../src/core/binary-vectors"; import { jaccardSimilarity, mmrRerank } from "../src/core/mmr"; import { searchExactVectorIndex } from "../src/core/vector-index"; import { cosineSimilarity } from "../src/core/vector-math"; @@ -64,22 +60,6 @@ function pushRow(kernel: string, count: number, ts: { ns: number; iterations: nu const rows: Row[] = []; const rng = makeRng(0xbe4c4); -for (const count of COUNTS) { - const query = Float64Array.from({ length: DIM }, () => rng() * 2 - 1); - const flat = new Float64Array(count * DIM); - for (let i = 0; i < flat.length; i += 1) flat[i] = rng() * 2 - 1; - - const ts = timeNs(() => { - for (let row = 0; row < count; row += 1) { - sink += cosineSimilarity(query, flat.subarray(row * DIM, (row + 1) * DIM)); - } - }); - const native = timeNs(() => { - sink += cosineSimilarityBatch(query, flat, DIM)[0] ?? 0; - }); - pushRow("cosineSimilarityBatch", count, ts, native); -} - // vectorIndexTopK is measured through the public wrapper searchExactVectorIndex, // so the native side pays the production per-call query conversion, guards, and // hit-array construction. The TS side replicates the pre-native wrapper body. @@ -113,33 +93,6 @@ for (const count of COUNTS) { pushRow("searchExactVectorIndex (topK wrapper)", count, ts, native); } -// hammingDistanceBatch: the native side pays FastBinarySearch.search's per-call -// packing (stride scan, packed/lengths allocation, full copy) before crossing. -for (const count of COUNTS) { - const query = Uint8Array.from({ length: STRIDE }, () => Math.floor(rng() * 256)); - const backing = new Uint8Array(count * STRIDE); - for (let i = 0; i < backing.length; i += 1) backing[i] = Math.floor(rng() * 256); - const vectors: Uint8Array[] = []; - for (let i = 0; i < count; i += 1) vectors.push(backing.subarray(i * STRIDE, (i + 1) * STRIDE)); - - const ts = timeNs(() => { - for (let i = 0; i < count; i += 1) sink += hammingDistance(query, vectors[i] ?? new Uint8Array()); - }); - const native = timeNs(() => { - let stride = 0; - for (const vector of vectors) if (vector.length > stride) stride = vector.length; - const packed = new Uint8Array(vectors.length * stride); - const lengths = new Uint32Array(vectors.length); - for (let i = 0; i < vectors.length; i += 1) { - const vector = vectors[i] ?? new Uint8Array(); - lengths[i] = vector.length; - packed.set(vector, i * stride); - } - sink += hammingDistanceBatch(query, packed, stride, lengths)[0] ?? 0; - }); - pushRow("hammingDistanceBatch (incl. packing)", count, ts, native); -} - // cosineSimilarityPairs: O(n²) pair scan. The TS baseline is the pre-native // clusterBySimilarity loop over per-item vectors; the native side pays the // production flatten (dim scan + Float64Array fill) before crossing. Capped at @@ -192,34 +145,6 @@ for (const count of COUNTS.filter(n => n <= 1000)) { pushRow("cosineSimilarityPairs (incl. flatten+adjacency)", count, ts, native); } -// hammingDistanceForDimBatch: ragged/dim-masked variant used by -// BinaryVectorStore.search; the native side pays the production per-call -// packed/dims allocation and copy before crossing. -for (const count of COUNTS) { - const query = Uint8Array.from({ length: STRIDE }, () => Math.floor(rng() * 256)); - const packed = new Uint8Array(count * STRIDE); - for (let i = 0; i < packed.length; i += 1) packed[i] = Math.floor(rng() * 256); - const dims = Uint32Array.from({ length: count }, () => (rng() < 0.5 ? DIM : DIM / 2)); - const vectors: Uint8Array[] = []; - for (let i = 0; i < count; i += 1) vectors.push(packed.subarray(i * STRIDE, (i + 1) * STRIDE)); - - const ts = timeNs(() => { - for (let i = 0; i < count; i += 1) { - sink += hammingDistanceForDimension(query, vectors[i] ?? new Uint8Array(), dims[i] ?? DIM); - } - }); - const native = timeNs(() => { - const packed2 = new Uint8Array(vectors.length * STRIDE); - const dims2 = new Uint32Array(vectors.length); - for (let i = 0; i < vectors.length; i += 1) { - const vector = vectors[i] ?? new Uint8Array(); - dims2[i] = dims[i] ?? DIM; - packed2.set(vector, i * STRIDE); - } - sink += hammingDistanceForDimBatch(query, packed2, STRIDE, dims2)[0] ?? 0; - }); - pushRow("hammingDistanceForDimBatch (incl. packing)", count, ts, native); -} // mmrRerank production paths: the TS side wraps jaccardSimilarity in a lambda, // defeating the identity check so the exact pre-native selection loop runs; the diff --git a/packages/mnemopi/src/core/binary-vectors.ts b/packages/mnemopi/src/core/binary-vectors.ts index 21d26718c..6964a5e70 100644 --- a/packages/mnemopi/src/core/binary-vectors.ts +++ b/packages/mnemopi/src/core/binary-vectors.ts @@ -1,5 +1,4 @@ import type { Database } from "bun:sqlite"; -import { hammingDistanceBatch, hammingDistanceForDimBatch } from "@oh-my-pi/pi-natives"; import { embeddingDim, type VecType } from "../config"; import { closeQuietly, type DatabasePath, openDatabase } from "../db"; @@ -148,7 +147,7 @@ export function hammingDistance(binaryA: Uint8Array | ArrayBuffer, binaryB: Uint return distance; } -export function hammingDistanceForDimension( +function hammingDistanceForDimension( binaryA: Uint8Array | ArrayBuffer, binaryB: Uint8Array | ArrayBuffer, dim: number, @@ -227,31 +226,15 @@ export class BinaryVectorStore { const rows = this.conn .query(`SELECT memory_id, binary_vector, original_dim, magnitude FROM ${this.tableName}`) .all() as VectorRow[]; - // Native batch kernel: pack all rows at a fixed stride (zero-padded, - // matching the TS `?? 0` out-of-range reads) and compute every - // dimension-masked distance in one N-API crossing. - const stride = BYTES_PER_VECTOR; - const packed = new Uint8Array(rows.length * stride); - const dims = new Uint32Array(rows.length); - const comparedDims: number[] = new Array(rows.length); - for (let i = 0; i < rows.length; i += 1) { - const row = rows[i]; - if (row === undefined) continue; + const results: BinaryVectorSearchResult[] = []; + for (const row of rows) { const storedDim = Math.max(0, Math.min(EMBEDDING_DIM, Math.trunc(toFiniteNumber(row.original_dim)))); const comparedDim = Math.min(queryDim, storedDim); - comparedDims[i] = comparedDim; - dims[i] = comparedDim; - const bytes = bytesFromBlob(row.binary_vector); - packed.set(bytes.length > stride ? bytes.subarray(0, stride) : bytes, i * stride); - } - const distances = hammingDistanceForDimBatch(queryBinary, packed, stride, dims); - const results: BinaryVectorSearchResult[] = []; - for (let i = 0; i < rows.length; i += 1) { - const distance = distances[i] ?? 0; + const distance = hammingDistanceForDimension(queryBinary, bytesFromBlob(row.binary_vector), comparedDim); results.push({ - memory_id: rows[i]?.memory_id ?? "", + memory_id: row.memory_id, distance, - score: informationTheoreticScore(distance, comparedDims[i] ?? 0), + score: informationTheoreticScore(distance, comparedDim), }); } results.sort((a, b) => b.score - a.score || a.memory_id.localeCompare(b.memory_id)); @@ -319,22 +302,9 @@ export class FastBinarySearch { search(queryBinary: Uint8Array | ArrayBuffer, topK = 10): BinaryVectorSearchResult[] { const query = queryBinary instanceof Uint8Array ? queryBinary : new Uint8Array(queryBinary); - // Native batch kernel: pack all vectors at the max byte length and - // compute every Hamming distance in one N-API crossing. Per-row - // lengths preserve the TS unmatched-tail popcount semantics. - let stride = 0; - for (const vector of this.vectors) if (vector.length > stride) stride = vector.length; - const packed = new Uint8Array(this.vectors.length * stride); - const lengths = new Uint32Array(this.vectors.length); - for (let i = 0; i < this.vectors.length; i += 1) { - const vector = this.vectors[i] ?? new Uint8Array(); - lengths[i] = vector.length; - packed.set(vector, i * stride); - } - const distances = hammingDistanceBatch(query, packed, stride, lengths); const results: BinaryVectorSearchResult[] = []; for (let i = 0; i < this.vectors.length; i += 1) { - const distance = distances[i] ?? 0; + const distance = hammingDistance(query, this.vectors[i] ?? new Uint8Array()); results.push({ memory_id: this.memoryIds[i] ?? "", distance, diff --git a/packages/mnemopi/test/native-vector-parity.test.ts b/packages/mnemopi/test/native-vector-parity.test.ts index da0499932..2a7340675 100644 --- a/packages/mnemopi/test/native-vector-parity.test.ts +++ b/packages/mnemopi/test/native-vector-parity.test.ts @@ -1,13 +1,5 @@ import { describe, expect, test } from "bun:test"; -import { - cosineSimilarityBatch, - cosineSimilarityPairs, - hammingDistanceBatch, - hammingDistanceForDimBatch, - mmrRerankIndices, - vectorIndexTopK, -} from "@oh-my-pi/pi-natives"; -import { hammingDistance, hammingDistanceForDimension } from "../src/core/binary-vectors"; +import { cosineSimilarityPairs, mmrRerankIndices, vectorIndexTopK } from "@oh-my-pi/pi-natives"; import { jaccardSimilarity, mmrRerank } from "../src/core/mmr"; import { buildExactVectorIndex, searchExactVectorIndex } from "../src/core/vector-index"; import { cosineSimilarity } from "../src/core/vector-math"; @@ -32,24 +24,6 @@ function expectClose(actual: number, expected: number): void { } describe("native vector kernel parity", () => { - test("cosineSimilarityBatch matches TS cosineSimilarity per row", () => { - const rng = makeRng(0xc051e); - const dim = 384; - const count = 200; - const query = Float64Array.from({ length: dim }, () => rng() * 2 - 1); - const candidates = new Float64Array(count * dim); - for (let i = 0; i < candidates.length; i += 1) candidates[i] = rng() * 2 - 1; - // Sprinkle non-finite values to exercise the finite_or_zero path. - candidates[3] = Number.NaN; - candidates[dim + 7] = Number.POSITIVE_INFINITY; - const scores = cosineSimilarityBatch(query, candidates, dim); - expect(scores.length).toBe(count); - for (let row = 0; row < count; row += 1) { - const expected = cosineSimilarity(query, candidates.subarray(row * dim, (row + 1) * dim)); - expectClose(scores[row] ?? Number.NaN, expected); - } - }); - test("cosineSimilarityPairs matches the TS pairwise threshold loop", () => { const rng = makeRng(0x9a175); const dim = 64; @@ -98,45 +72,6 @@ describe("native vector kernel parity", () => { } }); - test("hammingDistanceBatch is exactly equal to TS hammingDistance", () => { - const rng = makeRng(0xba7c4); - const stride = 48; // 384-dim binarized - const count = 128; - const query = Uint8Array.from({ length: stride }, () => Math.floor(rng() * 256)); - const packed = new Uint8Array(count * stride); - const lengths = new Uint32Array(count); - const vectors: Uint8Array[] = []; - for (let i = 0; i < count; i += 1) { - const len = i % 7 === 0 ? Math.floor(rng() * stride) : stride; // ragged rows - const vector = Uint8Array.from({ length: len }, () => Math.floor(rng() * 256)); - vectors.push(vector); - lengths[i] = len; - packed.set(vector, i * stride); - } - const distances = hammingDistanceBatch(query, packed, stride, lengths); - for (let i = 0; i < count; i += 1) { - expect(distances[i]).toBe(hammingDistance(query, vectors[i] ?? new Uint8Array())); - } - }); - - test("hammingDistanceForDimBatch is exactly equal to TS hammingDistanceForDimension", () => { - const rng = makeRng(0xd1235); - const stride = 48; - const count = 96; - const query = Uint8Array.from({ length: stride }, () => Math.floor(rng() * 256)); - const packed = new Uint8Array(count * stride); - const dims = new Uint32Array(count); - for (let i = 0; i < count; i += 1) { - dims[i] = Math.floor(rng() * (stride * 8 + 1)); // includes partial-byte tails and 0 - for (let b = 0; b < stride; b += 1) packed[i * stride + b] = Math.floor(rng() * 256); - } - const distances = hammingDistanceForDimBatch(query, packed, stride, dims); - for (let i = 0; i < count; i += 1) { - const row = packed.subarray(i * stride, (i + 1) * stride); - expect(distances[i]).toBe(hammingDistanceForDimension(query, row, dims[i] ?? 0)); - } - }); - test("mmrRerankIndices selects identical index sequences to the TS loop", () => { const rng = makeRng(0x33a11); const words = ["alpha", "beta", "gamma", "delta", "epsilon", "zeta", "eta", "theta", "iota", "kappa"]; diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index ee5f3df66..6f2a3a5bf 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -470,16 +470,6 @@ export interface ContextLine { */ export declare function copyToClipboard(text: string): void -/** - * Cosine similarity of `query` against a batch of candidate vectors. - * - * `candidates` is `n` vectors flattened row-major at `dim` elements per row - * (callers zero-pad shorter vectors, which matches the TS `?? 0` missing - * element semantics). Returns one score per candidate, bit-identical to - * calling the TS `cosineSimilarity(query, candidate)` per row. - */ -export declare function cosineSimilarityBatch(query: Float64Array, candidates: Float64Array, dim: number): Float64Array - /** * All pairs `(i, j)` with `i < j` whose cosine similarity meets `threshold`. * @@ -830,28 +820,6 @@ export interface GrepResult { skippedOversized?: number } -/** - * Hamming distance of `query` against a batch of packed binary vectors. - * - * `candidates` is flattened row-major at `stride` bytes per row. - * `lengths[i]` gives the meaningful byte length of row `i` (clamped to - * `stride`); omit it when every row is exactly `stride` bytes. Semantics - * match the TS `hammingDistance` exactly, including popcounting the - * unmatched tail of whichever side is longer. - */ -export declare function hammingDistanceBatch(query: Uint8Array, candidates: Uint8Array, stride: number, lengths?: Uint32Array | undefined | null): Uint32Array - -/** - * Dimension-masked Hamming distance of `query` against a batch of packed - * binary vectors. - * - * `candidates` is flattened row-major at `stride` bytes per row and - * `dims[i]` is the bit dimension compared for row `i`. Bytes beyond either - * side's data read as `0`, and a trailing partial byte is masked to the top - * `dims[i] % 8` bits — exactly the TS `hammingDistanceForDimension`. - */ -export declare function hammingDistanceForDimBatch(query: Uint8Array, candidates: Uint8Array, stride: number, dims: Uint32Array): Uint32Array - /** * Quick check if content matches a pattern. * diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index 773db1f60..4d7137c5b 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -30,7 +30,6 @@ export const astGrep = nativeBindings.astGrep; export const astMatch = nativeBindings.astMatch; export const blockRangeAt = nativeBindings.blockRangeAt; export const copyToClipboard = nativeBindings.copyToClipboard; -export const cosineSimilarityBatch = nativeBindings.cosineSimilarityBatch; export const cosineSimilarityPairs = nativeBindings.cosineSimilarityPairs; export const countTokens = nativeBindings.countTokens; export const detectMacOSAppearance = nativeBindings.detectMacOSAppearance; @@ -43,8 +42,6 @@ export const getSupportedLanguages = nativeBindings.getSupportedLanguages; export const getWorkProfile = nativeBindings.getWorkProfile; export const glob = nativeBindings.glob; export const grep = nativeBindings.grep; -export const hammingDistanceBatch = nativeBindings.hammingDistanceBatch; -export const hammingDistanceForDimBatch = nativeBindings.hammingDistanceForDimBatch; export const hasMatch = nativeBindings.hasMatch; export const highlightCode = nativeBindings.highlightCode; export const htmlToMarkdown = nativeBindings.htmlToMarkdown; From 8047bedaa46276710c9544d08f12bb13996da858 Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 05:11:36 -0700 Subject: [PATCH 15/16] test(mnemopi): record bench host metadata and allow archive-deployed sha --- packages/mnemopi/bench/native-vectors.bench.ts | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/packages/mnemopi/bench/native-vectors.bench.ts b/packages/mnemopi/bench/native-vectors.bench.ts index c6ae57cf7..b721989be 100644 --- a/packages/mnemopi/bench/native-vectors.bench.ts +++ b/packages/mnemopi/bench/native-vectors.bench.ts @@ -5,6 +5,7 @@ * * Run from the repo root: `bun packages/mnemopi/bench/native-vectors.bench.ts` */ +import * as os from "node:os"; import { cosineSimilarityPairs, } from "@oh-my-pi/pi-natives"; @@ -172,12 +173,13 @@ for (const count of COUNTS.filter(n => n <= 1000)) { pushRow("mmrRerankIndices (via mmrRerank)", count, ts, native); } -const sha = Bun.spawnSync(["git", "rev-parse", "HEAD"]).stdout.toString().trim(); +const sha = Bun.env.BENCH_SHA ?? Bun.spawnSync(["git", "rev-parse", "HEAD"]).stdout.toString().trim(); const report = { sha, date: new Date().toISOString(), scenario: `dim=${DIM}, stride=${STRIDE}B, warmup=${WARMUP}, iterations=${ITERATIONS} (adaptive for O(n²) rows, see per-row fields), crossing-inclusive`, runtime: `bun ${Bun.version}`, + host: `${os.cpus()[0]?.model ?? "unknown"}, ${os.platform()}-${os.arch()}`, rows: rows.map(r => ({ kernel: r.kernel, count: r.count, From 72394db60510f2b32f32be6f282c7d0a796467d9 Mon Sep 17 00:00:00 2001 From: Wolfgang Schoenberger <221313372+wolfiesch@users.noreply.github.com> Date: Wed, 22 Jul 2026 05:18:28 -0700 Subject: [PATCH 16/16] test(mnemopi): refresh bench artifact on an idle host and sync changelog claims --- packages/mnemopi/CHANGELOG.md | 2 +- .../mnemopi/bench/native-vectors.bench.json | 177 ++++-------------- packages/natives/CHANGELOG.md | 2 +- 3 files changed, 37 insertions(+), 144 deletions(-) diff --git a/packages/mnemopi/CHANGELOG.md b/packages/mnemopi/CHANGELOG.md index cbbc59cd3..7963c2055 100644 --- a/packages/mnemopi/CHANGELOG.md +++ b/packages/mnemopi/CHANGELOG.md @@ -4,7 +4,7 @@ ### Changed -- Recall hot loops (exact vector-index search, binary vector search, SHMR similarity clustering, and default-similarity MMR rerank) now run on native batch kernels with one N-API crossing per operation. Crossing-inclusive speedups at dim=384 (see `bench/native-vectors.bench.json`): 1.6-3.2x cosine scoring, 1.8-2x top-K search, 1.6x pairwise clustering scans, 3-10x Hamming search, and 21-39x MMR rerank. Custom `similarityFn` MMR reranks and incremental per-row scoring stay on the TypeScript implementations ([#6280](https://github.com/can1357/oh-my-pi/pull/6280) by [@wolfiesch](https://github.com/wolfiesch)). +- Recall hot loops (exact vector-index search, SHMR similarity clustering, and default-similarity MMR rerank) now run on native batch kernels with one N-API crossing per operation. Wrapper-inclusive speedups at dim=384 on Apple M1 (see `bench/native-vectors.bench.json`): 1.6-1.8x top-K search, 2.4x pairwise clustering scans at the measured 100- and 1000-candidate points (0.86x at the measured 10-candidate point, where the whole scan costs ~20µs either way), and 22-36x MMR rerank. Binary vector search stays on the TypeScript loops (per-call packing measured as a wash), as do custom `similarityFn` MMR reranks and incremental per-row scoring ([#6280](https://github.com/can1357/oh-my-pi/pull/6280) by [@wolfiesch](https://github.com/wolfiesch)). ## [17.0.4] - 2026-07-18 diff --git a/packages/mnemopi/bench/native-vectors.bench.json b/packages/mnemopi/bench/native-vectors.bench.json index 7d781530e..dc3ab8513 100644 --- a/packages/mnemopi/bench/native-vectors.bench.json +++ b/packages/mnemopi/bench/native-vectors.bench.json @@ -1,207 +1,100 @@ { - "sha": "3c84e2c410afae163fb1a778a4f7011354cc893c", - "date": "2026-07-22T11:33:29.804Z", + "sha": "8047bedaa46276710c9544d08f12bb13996da858", + "date": "2026-07-22T12:17:28.977Z", "scenario": "dim=384, stride=48B, warmup=20, iterations=200 (adaptive for O(n²) rows, see per-row fields), crossing-inclusive", "runtime": "bun 1.3.14", + "host": "Apple M1, darwin-arm64", "rows": [ { - "kernel": "cosineSimilarityBatch", + "kernel": "searchExactVectorIndex (topK wrapper)", "count": 10, - "ts_us": 9.85, - "native_us": 3.2, - "speedup": 3.08, + "ts_us": 17.34, + "native_us": 9.49, + "speedup": 1.83, "ts_iterations": 200, "native_iterations": 200 }, { - "kernel": "cosineSimilarityBatch", + "kernel": "searchExactVectorIndex (topK wrapper)", "count": 100, - "ts_us": 41.15, - "native_us": 25.18, - "speedup": 1.63, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "cosineSimilarityBatch", - "count": 1000, - "ts_us": 398.55, - "native_us": 242.88, - "speedup": 1.64, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "cosineSimilarityBatch", - "count": 10000, - "ts_us": 3936.21, - "native_us": 2431.53, + "ts_us": 56, + "native_us": 34.65, "speedup": 1.62, "ts_iterations": 200, "native_iterations": 200 }, - { - "kernel": "searchExactVectorIndex (topK wrapper)", - "count": 10, - "ts_us": 5.82, - "native_us": 6.44, - "speedup": 0.9, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "searchExactVectorIndex (topK wrapper)", - "count": 100, - "ts_us": 36.99, - "native_us": 19.48, - "speedup": 1.9, - "ts_iterations": 200, - "native_iterations": 200 - }, { "kernel": "searchExactVectorIndex (topK wrapper)", "count": 1000, - "ts_us": 310.52, - "native_us": 165.81, - "speedup": 1.87, + "ts_us": 508.91, + "native_us": 300.93, + "speedup": 1.69, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "searchExactVectorIndex (topK wrapper)", "count": 10000, - "ts_us": 3513.57, - "native_us": 1884.54, - "speedup": 1.86, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "hammingDistanceBatch (incl. packing)", - "count": 10, - "ts_us": 3.07, - "native_us": 1.39, - "speedup": 2.21, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "hammingDistanceBatch (incl. packing)", - "count": 100, - "ts_us": 7.09, - "native_us": 3.97, - "speedup": 1.79, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "hammingDistanceBatch (incl. packing)", - "count": 1000, - "ts_us": 34.25, - "native_us": 26.11, - "speedup": 1.31, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "hammingDistanceBatch (incl. packing)", - "count": 10000, - "ts_us": 221.43, - "native_us": 207.45, - "speedup": 1.07, + "ts_us": 5562.41, + "native_us": 3437.82, + "speedup": 1.62, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "cosineSimilarityPairs (incl. flatten+adjacency)", "count": 10, - "ts_us": 246.34, - "native_us": 22.08, - "speedup": 11.16, + "ts_us": 26.42, + "native_us": 30.67, + "speedup": 0.86, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "cosineSimilarityPairs (incl. flatten+adjacency)", "count": 100, - "ts_us": 5767.19, - "native_us": 1381.02, - "speedup": 4.18, + "ts_us": 4418.75, + "native_us": 1866.25, + "speedup": 2.37, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "cosineSimilarityPairs (incl. flatten+adjacency)", "count": 1000, - "ts_us": 530658.9, - "native_us": 131050.53, - "speedup": 4.05, + "ts_us": 445625.44, + "native_us": 179685.28, + "speedup": 2.48, "ts_iterations": 10, "native_iterations": 10 }, - { - "kernel": "hammingDistanceForDimBatch (incl. packing)", - "count": 10, - "ts_us": 2.52, - "native_us": 1.95, - "speedup": 1.29, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "hammingDistanceForDimBatch (incl. packing)", - "count": 100, - "ts_us": 5.88, - "native_us": 4.16, - "speedup": 1.41, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "hammingDistanceForDimBatch (incl. packing)", - "count": 1000, - "ts_us": 27.69, - "native_us": 28.69, - "speedup": 0.97, - "ts_iterations": 200, - "native_iterations": 200 - }, - { - "kernel": "hammingDistanceForDimBatch (incl. packing)", - "count": 10000, - "ts_us": 231.52, - "native_us": 235.98, - "speedup": 0.98, - "ts_iterations": 200, - "native_iterations": 200 - }, { "kernel": "mmrRerankIndices (via mmrRerank)", "count": 10, - "ts_us": 331.56, - "native_us": 15.1, - "speedup": 21.95, + "ts_us": 395.5, + "native_us": 17.62, + "speedup": 22.44, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "mmrRerankIndices (via mmrRerank)", "count": 100, - "ts_us": 7965.61, - "native_us": 220.67, - "speedup": 36.1, + "ts_us": 10439.19, + "native_us": 287.99, + "speedup": 36.25, "ts_iterations": 200, "native_iterations": 200 }, { "kernel": "mmrRerankIndices (via mmrRerank)", "count": 1000, - "ts_us": 78042.9, - "native_us": 2639.47, - "speedup": 29.57, + "ts_us": 103521.56, + "native_us": 3430.07, + "speedup": 30.18, "ts_iterations": 200, "native_iterations": 200 } ], - "sink": 823377224.8853176 + "sink": 15782.734317403312 } diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index a9d8b5619..ec94e4b45 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -4,7 +4,7 @@ ### Added -- Added batch vector kernels for mnemopi recall paths: `cosineSimilarityBatch`, `cosineSimilarityPairs`, `vectorIndexTopK`, `hammingDistanceBatch`, `hammingDistanceForDimBatch`, and `mmrRerankIndices` ([#6280](https://github.com/can1357/oh-my-pi/pull/6280) by [@wolfiesch](https://github.com/wolfiesch)). +- Added batch vector kernels for mnemopi recall paths: `cosineSimilarityPairs`, `vectorIndexTopK`, and `mmrRerankIndices` ([#6280](https://github.com/can1357/oh-my-pi/pull/6280) by [@wolfiesch](https://github.com/wolfiesch)). ## [17.0.5] - 2026-07-18