chore: clippy
This commit is contained in:
+36
-33
@@ -413,58 +413,61 @@ impl<const N: usize> ToNapiValue for InlineStr<N> {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn arena() -> Arena {
|
||||
Arena {
|
||||
buf: UnsafeCell::new([0; SCRATCH_LEN / 2]),
|
||||
offset: Cell::new(0),
|
||||
live: Cell::new(0),
|
||||
}
|
||||
fn with_arena(test: impl FnOnce(&Arena)) {
|
||||
ARENA.with(test);
|
||||
}
|
||||
|
||||
/// Live guards own disjoint ranges; overlap would alias the derefs (UB).
|
||||
#[test]
|
||||
fn commits_never_overlap_live_ranges() {
|
||||
let a = arena();
|
||||
let (s1, _) = a.tail(1);
|
||||
a.commit(s1, 100);
|
||||
let (s2, _) = a.tail(2);
|
||||
assert!(s2 >= s1 + 100);
|
||||
a.commit(s2, 50);
|
||||
let (s3, _) = a.tail(1);
|
||||
assert!(s3 >= s2 + 50);
|
||||
with_arena(|a| {
|
||||
let (s1, _) = a.tail(1);
|
||||
a.commit(s1, 100);
|
||||
let (s2, _) = a.tail(2);
|
||||
assert!(s2 >= s1 + 100);
|
||||
a.commit(s2, 50);
|
||||
let (s3, _) = a.tail(1);
|
||||
assert!(s3 >= s2 + 50);
|
||||
a.release(s2, s2 + 50);
|
||||
a.release(s1, s1 + 100);
|
||||
});
|
||||
}
|
||||
|
||||
/// LIFO drops recycle immediately; the next fill reuses the range.
|
||||
#[test]
|
||||
fn lifo_release_rolls_back() {
|
||||
let a = arena();
|
||||
a.commit(0, 100);
|
||||
a.commit(100, 50);
|
||||
a.release(100, 150);
|
||||
assert_eq!(a.tail(1).0, 100);
|
||||
a.release(0, 100);
|
||||
assert_eq!(a.tail(1).0, 0);
|
||||
with_arena(|a| {
|
||||
a.commit(0, 100);
|
||||
a.commit(100, 50);
|
||||
a.release(100, 150);
|
||||
assert_eq!(a.tail(1).0, 100);
|
||||
a.release(0, 100);
|
||||
assert_eq!(a.tail(1).0, 0);
|
||||
});
|
||||
}
|
||||
|
||||
/// Non-LIFO drops strand bytes only until the last guard goes away.
|
||||
#[test]
|
||||
fn arena_resets_when_last_guard_drops() {
|
||||
let a = arena();
|
||||
a.commit(0, 100);
|
||||
a.commit(100, 50);
|
||||
a.release(0, 100);
|
||||
assert_eq!(a.tail(1).0, 150, "inner range stays stranded while a guard is live");
|
||||
a.release(100, 150);
|
||||
assert_eq!(a.tail(1).0, 0);
|
||||
with_arena(|a| {
|
||||
a.commit(0, 100);
|
||||
a.commit(100, 50);
|
||||
a.release(0, 100);
|
||||
assert_eq!(a.tail(1).0, 150, "inner range stays stranded while a guard is live");
|
||||
a.release(100, 150);
|
||||
assert_eq!(a.tail(1).0, 0);
|
||||
});
|
||||
}
|
||||
|
||||
/// A utf16 fill after an odd utf8 commit must get a 2-aligned range.
|
||||
#[test]
|
||||
fn utf16_tail_is_aligned() {
|
||||
let a = arena();
|
||||
a.commit(0, 7);
|
||||
let (start, len) = a.tail(2);
|
||||
assert_eq!(start, 8);
|
||||
assert_eq!(len, SCRATCH_LEN - 8);
|
||||
with_arena(|a| {
|
||||
a.commit(0, 7);
|
||||
let (start, len) = a.tail(2);
|
||||
assert_eq!(start, 8);
|
||||
assert_eq!(len, SCRATCH_LEN - 8);
|
||||
a.release(0, 7);
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
//! decompression in [`tables`](crate::utok::tables).
|
||||
//!
|
||||
//! Input is encoding-generic: [`BpeEncoding::count`]/[`encode`]
|
||||
//! (BpeEncoding::encode) take `&[U: Unit]`. Pre-tokenization scans the
|
||||
//! (`BpeEncoding::encode`) take `&[U: Unit]`. Pre-tokenization scans the
|
||||
//! units natively; each piece is then UTF-8-encoded into a reused buffer
|
||||
//! for the byte-keyed rank table (`str` input skips that copy entirely,
|
||||
//! non-UTF-8 flavors narrow ASCII runs 1:1). Steady state performs no
|
||||
@@ -31,7 +31,7 @@ use crate::utok::{
|
||||
};
|
||||
|
||||
/// Firefox/rustc Fx hash: multiplicative word-at-a-time mixing. Rank
|
||||
/// lookups hash short byte keys on every merge step; SipHash is the
|
||||
/// lookups hash short byte keys on every merge step; `SipHash` is the
|
||||
/// dominant cost there (~30% end-to-end at the default hasher).
|
||||
#[derive(Default)]
|
||||
struct FxHasher(u64);
|
||||
@@ -102,7 +102,7 @@ fn pack(key: &[u8]) -> Option<u128> {
|
||||
/// - other ≤15 bytes — [`pack`]ed `u128` keys in an Fx map: KV inline in the
|
||||
/// table, no `Box` pointer chase, no byte-wise compare.
|
||||
/// - >15 bytes — plain byte-keyed Fx map (~3% of vocab; spans this long are
|
||||
/// almost always misses).
|
||||
/// > almost always misses).
|
||||
pub struct RankTable {
|
||||
/// Rank of 2-byte token `[a, b]` at `a << 8 | b`; `u32::MAX` where
|
||||
/// absent (ranks are vocab indices, far below the sentinel).
|
||||
@@ -190,7 +190,7 @@ impl RankTable {
|
||||
self
|
||||
.rank(&piece[start..end])
|
||||
.expect("utoken: unreachable merge state"),
|
||||
)
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -270,7 +270,7 @@ pub struct BpeEncoding {
|
||||
/// (GLM-5). The engine already short-circuits whole-piece hits, which
|
||||
/// is proven equivalent for GLM-5 (see GLM tests); flag kept for
|
||||
/// documentation and any future divergence.
|
||||
#[allow(dead_code)]
|
||||
#[allow(dead_code, reason = "retained to document the GLM-5 tokenizer behavior")]
|
||||
pub ignore_merges: bool,
|
||||
}
|
||||
|
||||
@@ -301,8 +301,12 @@ impl BpeEncoding {
|
||||
return self.scan(bytes, f);
|
||||
}
|
||||
// Non-UTF-8 flavors: owned UTF-8 needed only when NFC actually has
|
||||
// work to do, or while the family is still on the regex splitter.
|
||||
if (self.nfc && !nfc_quick(units)) || self.splitter.is_regex() {
|
||||
// work to do, or while the test-only regex oracle is active.
|
||||
#[cfg(test)]
|
||||
let regex_splitter = self.splitter.is_regex();
|
||||
#[cfg(not(test))]
|
||||
let regex_splitter = false;
|
||||
if (self.nfc && !nfc_quick(units)) || regex_splitter {
|
||||
let s = decode_lossy(units);
|
||||
let s = match pretoken::nfc(&s) {
|
||||
Cow::Owned(o) if self.nfc => o,
|
||||
@@ -310,7 +314,7 @@ impl BpeEncoding {
|
||||
};
|
||||
return self.scan(s.as_bytes(), f);
|
||||
}
|
||||
self.scan(units, f)
|
||||
self.scan(units, f);
|
||||
}
|
||||
|
||||
fn scan<U: Unit>(&self, units: &[U], f: &mut impl FnMut(&RankTable, &[u8])) {
|
||||
@@ -334,18 +338,15 @@ fn piece_bytes<'a, U: Unit>(piece: &'a [U], buf: &'a mut Vec<u8>) -> &'a [u8] {
|
||||
// ASCII runs narrow 1:1 without the decode/encode round-trip
|
||||
// (dominant for code/English u16 input, cf. xutf's ASCII kernels;
|
||||
// the trivial loop autovectorizes).
|
||||
match piece[i].ascii() {
|
||||
Some(b) => {
|
||||
buf.push(b);
|
||||
i += 1;
|
||||
},
|
||||
None => {
|
||||
let (c, n) = U::decode(piece, i);
|
||||
i += n;
|
||||
let mut tmp = [0u8; 4];
|
||||
buf.extend_from_slice(c.encode_utf8(&mut tmp).as_bytes());
|
||||
},
|
||||
}
|
||||
if let Some(b) = piece[i].ascii() {
|
||||
buf.push(b);
|
||||
i += 1;
|
||||
} else {
|
||||
let (c, n) = U::decode(piece, i);
|
||||
i += n;
|
||||
let mut tmp = [0u8; 4];
|
||||
buf.extend_from_slice(c.encode_utf8(&mut tmp).as_bytes());
|
||||
}
|
||||
}
|
||||
buf
|
||||
}
|
||||
|
||||
@@ -156,7 +156,7 @@ pub fn content_token_count<U: Unit>(units: &[U], family: Family) -> u32 {
|
||||
/// Reconstructed `count_tokens` value for `units` as a single user message:
|
||||
/// content tiling plus the measured message frame (ctok's `token_count`).
|
||||
// Exercised by the fixture tests; `lib.rs` only routes content counts.
|
||||
#[cfg_attr(not(test), allow(dead_code))]
|
||||
#[cfg_attr(not(test), allow(dead_code, reason = "used only by fixture tests"))]
|
||||
pub fn message_token_count<U: Unit>(units: &[U], family: Family) -> u32 {
|
||||
content_token_count(units, family) + family.params().message_overhead
|
||||
}
|
||||
@@ -223,7 +223,7 @@ mod tests {
|
||||
fn content_count_is_message_minus_frame() {
|
||||
// The public split every consumer relies on: summing fragments must
|
||||
// never include per-message frame overhead.
|
||||
assert_eq!(content_token_count("".as_bytes(), Family::V5), 0);
|
||||
assert_eq!(content_token_count(b"", Family::V5), 0);
|
||||
for (family, overhead) in
|
||||
[(Family::V3, 7), (Family::V47, 11), (Family::V5, 6), (Family::V5Sonnet, 6)]
|
||||
{
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
//! | `Cl100kBase` | GPT-3.5 / GPT-4 | byte-level BPE |
|
||||
//! | `ClaudeV3` / `ClaudeV47` / `ClaudeV5` / `ClaudeV5Sonnet` | Claude generations | ctok count reconstruction (count-only) |
|
||||
//! | `Qwen3` | Qwen 3.5 / 3.6 / 3.8 (248k vocab) | byte-level BPE + NFC |
|
||||
//! | `DeepSeekV3` | DeepSeek V3 … V4 | byte-level BPE, 3-stage split chain |
|
||||
//! | `DeepSeekV3` | `DeepSeek` V3 … V4 | byte-level BPE, 3-stage split chain |
|
||||
//! | `KimiK2` | Kimi K2 … K3 | byte-level BPE (tiktoken ranks) |
|
||||
//! | `Glm5` | GLM-5.x (exact), GLM-4.x (near-exact) | byte-level BPE, `ignore_merges` |
|
||||
//!
|
||||
@@ -41,9 +41,9 @@ pub use self::{
|
||||
/// process-wide tables (first use pays one zstd decode of ~0.5–1 MB).
|
||||
#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)]
|
||||
pub enum Encoding {
|
||||
/// GPT-4o / o1 / GPT-5 (OpenAI default).
|
||||
/// GPT-4o / o1 / GPT-5 (`OpenAI` default).
|
||||
O200kBase,
|
||||
/// GPT-3.5 / GPT-4 / older OpenAI.
|
||||
/// GPT-3.5 / GPT-4 / older `OpenAI`.
|
||||
Cl100kBase,
|
||||
/// Claude 3 through Opus 4.6 (and every non-opus Claude < 5).
|
||||
ClaudeV3,
|
||||
@@ -55,7 +55,7 @@ pub enum Encoding {
|
||||
ClaudeV5Sonnet,
|
||||
/// Qwen 3.5 / 3.6 / 3.8 (248,044-token vocabulary, NFC input).
|
||||
Qwen3,
|
||||
/// DeepSeek V3 / V3.1 / V3.2 / R1 / V4 (identical base BPE).
|
||||
/// `DeepSeek` V3 / V3.1 / V3.2 / R1 / V4 (identical base BPE).
|
||||
DeepSeekV3,
|
||||
/// Kimi K2 / K2.5 / K3 (163,584-token base vocabulary).
|
||||
KimiK2,
|
||||
|
||||
@@ -18,7 +18,7 @@ pub enum Splitter {
|
||||
O200k,
|
||||
/// tiktoken `cl100k_base` scanner.
|
||||
Cl100k,
|
||||
/// DeepSeek V3..V4 three-stage chain scanner.
|
||||
/// `DeepSeek` V3..V4 three-stage chain scanner.
|
||||
DeepSeek,
|
||||
/// Kimi K2/K3 scanner.
|
||||
Kimi,
|
||||
@@ -40,14 +40,10 @@ impl Splitter {
|
||||
}
|
||||
|
||||
/// Whether this splitter needs `&str` input (the engine transcodes
|
||||
/// non-UTF-8 flavors before calling in). Always false at runtime; only
|
||||
/// the test-only regex oracle transcodes.
|
||||
pub fn is_regex(&self) -> bool {
|
||||
#[cfg(test)]
|
||||
if matches!(self, Self::Regex(_)) {
|
||||
return true;
|
||||
}
|
||||
false
|
||||
/// non-UTF-8 flavors before calling in).
|
||||
#[cfg(test)]
|
||||
pub const fn is_regex(&self) -> bool {
|
||||
matches!(self, Self::Regex(_))
|
||||
}
|
||||
|
||||
/// Feed every piece of `units` to `f`, in order, covering the input
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//! DeepSeek V3..V4 pre-tokenizer: hand-written codepoint scanner over `&[U]`.
|
||||
//! `DeepSeek` V3..V4 pre-tokenizer: hand-written codepoint scanner over `&[U]`.
|
||||
//!
|
||||
//! Faithful port of the three-stage HF `Split(Isolated)` chain (see
|
||||
//! `DEEPSEEK_STAGE_*` in `tables.rs`). Every stage applies to each piece
|
||||
@@ -10,7 +10,7 @@
|
||||
//! 2. `[一-龥-ゟ゠-ヿ]+` — Han U+4E00–9FA5 plus the contiguous
|
||||
//! hiragana/katakana blocks U+3040–30FF.
|
||||
//! 3. The main pattern:
|
||||
//! - `[!"#$%&'()*+,\-./:;<=>?@\[\\\]^_`{|}~][A-Za-z]+` (one ASCII
|
||||
//! - ``[!"#$%&'()*+,\-./:;<=>?@\[\\\]^_`{|}~][A-Za-z]+`` (one ASCII
|
||||
//! punctuation codepoint glued to ASCII letters, e.g. `.NET`, `(foo`)
|
||||
//! - `[^\r\n\p{L}\p{P}\p{S}]?[\p{L}\p{M}]+`
|
||||
//! - ` ?[\p{P}\p{S}]+[\r\n]*`
|
||||
@@ -72,7 +72,7 @@ fn isolated<U: Unit>(
|
||||
|
||||
/// `[一-龥-ゟ゠-ヿ]`.
|
||||
#[inline]
|
||||
fn is_cjk(c: char) -> bool {
|
||||
const fn is_cjk(c: char) -> bool {
|
||||
matches!(c as u32, 0x3040..=0x30FF | 0x4E00..=0x9FA5)
|
||||
}
|
||||
|
||||
@@ -128,11 +128,10 @@ fn main_end<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||
}
|
||||
if c != '\r' && c != '\n' && !is_ps(c) {
|
||||
// Not L (checked above), not P/S, not CR/LF: prefix-eligible.
|
||||
if let Some((c2, n2)) = decode_at(units, pos + n) {
|
||||
if is_lm(c2) {
|
||||
if let Some((c2, n2)) = decode_at(units, pos + n)
|
||||
&& is_lm(c2) {
|
||||
return Some(lm_run_end(units, pos + n + n2));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ` ?[\p{P}\p{S}]+[\r\n]*`
|
||||
|
||||
@@ -64,7 +64,7 @@ pub mod cls {
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn is_ws(c: char) -> bool {
|
||||
pub const fn is_ws(c: char) -> bool {
|
||||
c.is_whitespace()
|
||||
}
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
//! Qwen3 (3.5/3.6/3.8) split pattern as a codepoint scanner.
|
||||
//!
|
||||
//! Reference (qwen3.8.tokenizer.json pre_tokenizer, = families.json qwen3):
|
||||
//! Reference (qwen3.8.tokenizer.json `pre_tokenizer`, = families.json qwen3):
|
||||
//! ```text
|
||||
//! (?i:'s|'t|'re|'ve|'m|'ll|'d)
|
||||
//! |[^\r\n\p{L}\p{N}]?[\p{L}\p{M}]+
|
||||
|
||||
@@ -132,6 +132,19 @@ static GLM5: LazyLock<BpeEncoding> = LazyLock::new(|| BpeEncoding {
|
||||
ignore_merges: true,
|
||||
});
|
||||
|
||||
/// Resolve the BPE encoding for a non-Claude family.
|
||||
pub(crate) fn bpe_for(enc: Encoding) -> &'static BpeEncoding {
|
||||
match enc {
|
||||
Encoding::O200kBase => &O200K_BASE,
|
||||
Encoding::Cl100kBase => &CL100K_BASE,
|
||||
Encoding::Qwen3 => &QWEN3,
|
||||
Encoding::DeepSeekV3 => &DEEPSEEK3,
|
||||
Encoding::KimiK2 => &KIMI_K2,
|
||||
Encoding::Glm5 => &GLM5,
|
||||
_ => unreachable!("claude families never reach bpe_for"),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod glm_scan_tests {
|
||||
//! Differential justifying the cl100k-scanner alias: piece boundaries
|
||||
@@ -212,16 +225,3 @@ mod glm_scan_tests {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Resolve the BPE encoding for a non-Claude family.
|
||||
pub(crate) fn bpe_for(enc: Encoding) -> &'static BpeEncoding {
|
||||
match enc {
|
||||
Encoding::O200kBase => &O200K_BASE,
|
||||
Encoding::Cl100kBase => &CL100K_BASE,
|
||||
Encoding::Qwen3 => &QWEN3,
|
||||
Encoding::DeepSeekV3 => &DEEPSEEK3,
|
||||
Encoding::KimiK2 => &KIMI_K2,
|
||||
Encoding::Glm5 => &GLM5,
|
||||
_ => unreachable!("claude families never reach bpe_for"),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
//! Golden-fixture test: DeepSeekV3 vs reference HF tokenizers encode
|
||||
//! (add_special_tokens=False) over cache/deepseek-v4.tokenizer.json.
|
||||
//! Golden-fixture test: `DeepSeekV3` vs reference HF tokenizers encode
|
||||
//! (`add_special_tokens=False`) over cache/deepseek-v4.tokenizer.json.
|
||||
//! Base BPE is identical V3..V4 (verified upstream: same vocab + merges
|
||||
//! hash), so one encoding covers the whole family.
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
//! Golden-fixture test: Glm5 vs reference HF tokenizers
|
||||
//! (add_special_tokens=False).
|
||||
//! (`add_special_tokens=False`).
|
||||
//!
|
||||
//! GLM-4.x note: GLM-5 is an ID-preserving superset of GLM-4.x (~3.5k extra
|
||||
//! merges), so GLM-4.x counts are near-exact under this table — no separate
|
||||
@@ -36,10 +36,10 @@ fn glm5_matches_reference() {
|
||||
}
|
||||
}
|
||||
|
||||
/// Directed ignore_merges semantics: ' 参考' exists in vocab (99855) but
|
||||
/// Directed `ignore_merges` semantics: ' 参考' exists in vocab (99855) but
|
||||
/// HF's greedy merge order never forms it from bytes; normal merge-loop BPE
|
||||
/// yields [26767, 224, 98580] (verified with the Python reference with
|
||||
/// ignore_merges=false). The whole-piece short-circuit must win.
|
||||
/// `ignore_merges=false`). The whole-piece short-circuit must win.
|
||||
#[test]
|
||||
fn glm5_ignore_merges_whole_piece_wins() {
|
||||
assert_eq!(Encoding::Glm5.encode(" 参考").unwrap(), vec![99855]);
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//! Golden-fixture test: KimiK2 vs reference tiktoken encode_ordinary.
|
||||
//! Golden-fixture test: `KimiK2` vs reference tiktoken `encode_ordinary`.
|
||||
|
||||
use crate::utok::Encoding;
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//! OpenAI family tests: golden fixtures from Python tiktoken, plus a
|
||||
//! `OpenAI` family tests: golden fixtures from Python tiktoken, plus a
|
||||
//! differential test against tiktoken-rs over the corpus and seeded
|
||||
//! randomized strings.
|
||||
|
||||
@@ -8,7 +8,7 @@ use crate::utok::Encoding;
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct Fixture {
|
||||
#[allow(dead_code)]
|
||||
#[allow(dead_code, reason = "fixture provenance is deserialized but not asserted")]
|
||||
generator: String,
|
||||
cases: Vec<Case>,
|
||||
}
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
//! Golden-fixture test: Qwen3 vs reference HF tokenizers encode
|
||||
//! (add_special_tokens=false), including NFC normalization and the
|
||||
//! (`add_special_tokens=false`), including NFC normalization and the
|
||||
//! dead-rank (merge-unreachable vocab entry) regressions.
|
||||
|
||||
use crate::utok::Encoding;
|
||||
|
||||
@@ -119,7 +119,7 @@ impl Unit for u32 {
|
||||
|
||||
#[inline]
|
||||
fn encode(cp: char, out: &mut [Self]) -> usize {
|
||||
out[0] = cp as u32;
|
||||
out[0] = cp as Self;
|
||||
1
|
||||
}
|
||||
|
||||
@@ -203,7 +203,7 @@ pub struct Cursor<'a, U: Unit> {
|
||||
|
||||
impl<'a, U: Unit> Cursor<'a, U> {
|
||||
#[inline]
|
||||
pub fn new(units: &'a [U]) -> Self {
|
||||
pub const fn new(units: &'a [U]) -> Self {
|
||||
Self { units, pos: 0 }
|
||||
}
|
||||
|
||||
@@ -221,7 +221,7 @@ impl<'a, U: Unit> Cursor<'a, U> {
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn advance(&mut self, n: usize) {
|
||||
pub const fn advance(&mut self, n: usize) {
|
||||
self.pos += n;
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user