chore: clippy

This commit is contained in:
can1357
2026-08-20 04:05:49 +02:00
parent 2f54e760c1
commit 90ac96242f
16 changed files with 100 additions and 103 deletions
+36 -33
View File
@@ -413,58 +413,61 @@ impl<const N: usize> ToNapiValue for InlineStr<N> {
mod tests {
use super::*;
fn arena() -> Arena {
Arena {
buf: UnsafeCell::new([0; SCRATCH_LEN / 2]),
offset: Cell::new(0),
live: Cell::new(0),
}
fn with_arena(test: impl FnOnce(&Arena)) {
ARENA.with(test);
}
/// Live guards own disjoint ranges; overlap would alias the derefs (UB).
#[test]
fn commits_never_overlap_live_ranges() {
let a = arena();
let (s1, _) = a.tail(1);
a.commit(s1, 100);
let (s2, _) = a.tail(2);
assert!(s2 >= s1 + 100);
a.commit(s2, 50);
let (s3, _) = a.tail(1);
assert!(s3 >= s2 + 50);
with_arena(|a| {
let (s1, _) = a.tail(1);
a.commit(s1, 100);
let (s2, _) = a.tail(2);
assert!(s2 >= s1 + 100);
a.commit(s2, 50);
let (s3, _) = a.tail(1);
assert!(s3 >= s2 + 50);
a.release(s2, s2 + 50);
a.release(s1, s1 + 100);
});
}
/// LIFO drops recycle immediately; the next fill reuses the range.
#[test]
fn lifo_release_rolls_back() {
let a = arena();
a.commit(0, 100);
a.commit(100, 50);
a.release(100, 150);
assert_eq!(a.tail(1).0, 100);
a.release(0, 100);
assert_eq!(a.tail(1).0, 0);
with_arena(|a| {
a.commit(0, 100);
a.commit(100, 50);
a.release(100, 150);
assert_eq!(a.tail(1).0, 100);
a.release(0, 100);
assert_eq!(a.tail(1).0, 0);
});
}
/// Non-LIFO drops strand bytes only until the last guard goes away.
#[test]
fn arena_resets_when_last_guard_drops() {
let a = arena();
a.commit(0, 100);
a.commit(100, 50);
a.release(0, 100);
assert_eq!(a.tail(1).0, 150, "inner range stays stranded while a guard is live");
a.release(100, 150);
assert_eq!(a.tail(1).0, 0);
with_arena(|a| {
a.commit(0, 100);
a.commit(100, 50);
a.release(0, 100);
assert_eq!(a.tail(1).0, 150, "inner range stays stranded while a guard is live");
a.release(100, 150);
assert_eq!(a.tail(1).0, 0);
});
}
/// A utf16 fill after an odd utf8 commit must get a 2-aligned range.
#[test]
fn utf16_tail_is_aligned() {
let a = arena();
a.commit(0, 7);
let (start, len) = a.tail(2);
assert_eq!(start, 8);
assert_eq!(len, SCRATCH_LEN - 8);
with_arena(|a| {
a.commit(0, 7);
let (start, len) = a.tail(2);
assert_eq!(start, 8);
assert_eq!(len, SCRATCH_LEN - 8);
a.release(0, 7);
});
}
}
+21 -20
View File
@@ -6,7 +6,7 @@
//! decompression in [`tables`](crate::utok::tables).
//!
//! Input is encoding-generic: [`BpeEncoding::count`]/[`encode`]
//! (BpeEncoding::encode) take `&[U: Unit]`. Pre-tokenization scans the
//! (`BpeEncoding::encode`) take `&[U: Unit]`. Pre-tokenization scans the
//! units natively; each piece is then UTF-8-encoded into a reused buffer
//! for the byte-keyed rank table (`str` input skips that copy entirely,
//! non-UTF-8 flavors narrow ASCII runs 1:1). Steady state performs no
@@ -31,7 +31,7 @@ use crate::utok::{
};
/// Firefox/rustc Fx hash: multiplicative word-at-a-time mixing. Rank
/// lookups hash short byte keys on every merge step; SipHash is the
/// lookups hash short byte keys on every merge step; `SipHash` is the
/// dominant cost there (~30% end-to-end at the default hasher).
#[derive(Default)]
struct FxHasher(u64);
@@ -102,7 +102,7 @@ fn pack(key: &[u8]) -> Option<u128> {
/// - other ≤15 bytes — [`pack`]ed `u128` keys in an Fx map: KV inline in the
/// table, no `Box` pointer chase, no byte-wise compare.
/// - >15 bytes — plain byte-keyed Fx map (~3% of vocab; spans this long are
/// almost always misses).
/// > almost always misses).
pub struct RankTable {
/// Rank of 2-byte token `[a, b]` at `a << 8 | b`; `u32::MAX` where
/// absent (ranks are vocab indices, far below the sentinel).
@@ -190,7 +190,7 @@ impl RankTable {
self
.rank(&piece[start..end])
.expect("utoken: unreachable merge state"),
)
);
});
}
@@ -270,7 +270,7 @@ pub struct BpeEncoding {
/// (GLM-5). The engine already short-circuits whole-piece hits, which
/// is proven equivalent for GLM-5 (see GLM tests); flag kept for
/// documentation and any future divergence.
#[allow(dead_code)]
#[allow(dead_code, reason = "retained to document the GLM-5 tokenizer behavior")]
pub ignore_merges: bool,
}
@@ -301,8 +301,12 @@ impl BpeEncoding {
return self.scan(bytes, f);
}
// Non-UTF-8 flavors: owned UTF-8 needed only when NFC actually has
// work to do, or while the family is still on the regex splitter.
if (self.nfc && !nfc_quick(units)) || self.splitter.is_regex() {
// work to do, or while the test-only regex oracle is active.
#[cfg(test)]
let regex_splitter = self.splitter.is_regex();
#[cfg(not(test))]
let regex_splitter = false;
if (self.nfc && !nfc_quick(units)) || regex_splitter {
let s = decode_lossy(units);
let s = match pretoken::nfc(&s) {
Cow::Owned(o) if self.nfc => o,
@@ -310,7 +314,7 @@ impl BpeEncoding {
};
return self.scan(s.as_bytes(), f);
}
self.scan(units, f)
self.scan(units, f);
}
fn scan<U: Unit>(&self, units: &[U], f: &mut impl FnMut(&RankTable, &[u8])) {
@@ -334,18 +338,15 @@ fn piece_bytes<'a, U: Unit>(piece: &'a [U], buf: &'a mut Vec<u8>) -> &'a [u8] {
// ASCII runs narrow 1:1 without the decode/encode round-trip
// (dominant for code/English u16 input, cf. xutf's ASCII kernels;
// the trivial loop autovectorizes).
match piece[i].ascii() {
Some(b) => {
buf.push(b);
i += 1;
},
None => {
let (c, n) = U::decode(piece, i);
i += n;
let mut tmp = [0u8; 4];
buf.extend_from_slice(c.encode_utf8(&mut tmp).as_bytes());
},
}
if let Some(b) = piece[i].ascii() {
buf.push(b);
i += 1;
} else {
let (c, n) = U::decode(piece, i);
i += n;
let mut tmp = [0u8; 4];
buf.extend_from_slice(c.encode_utf8(&mut tmp).as_bytes());
}
}
buf
}
+2 -2
View File
@@ -156,7 +156,7 @@ pub fn content_token_count<U: Unit>(units: &[U], family: Family) -> u32 {
/// Reconstructed `count_tokens` value for `units` as a single user message:
/// content tiling plus the measured message frame (ctok's `token_count`).
// Exercised by the fixture tests; `lib.rs` only routes content counts.
#[cfg_attr(not(test), allow(dead_code))]
#[cfg_attr(not(test), allow(dead_code, reason = "used only by fixture tests"))]
pub fn message_token_count<U: Unit>(units: &[U], family: Family) -> u32 {
content_token_count(units, family) + family.params().message_overhead
}
@@ -223,7 +223,7 @@ mod tests {
fn content_count_is_message_minus_frame() {
// The public split every consumer relies on: summing fragments must
// never include per-message frame overhead.
assert_eq!(content_token_count("".as_bytes(), Family::V5), 0);
assert_eq!(content_token_count(b"", Family::V5), 0);
for (family, overhead) in
[(Family::V3, 7), (Family::V47, 11), (Family::V5, 6), (Family::V5Sonnet, 6)]
{
+4 -4
View File
@@ -10,7 +10,7 @@
//! | `Cl100kBase` | GPT-3.5 / GPT-4 | byte-level BPE |
//! | `ClaudeV3` / `ClaudeV47` / `ClaudeV5` / `ClaudeV5Sonnet` | Claude generations | ctok count reconstruction (count-only) |
//! | `Qwen3` | Qwen 3.5 / 3.6 / 3.8 (248k vocab) | byte-level BPE + NFC |
//! | `DeepSeekV3` | DeepSeek V3 … V4 | byte-level BPE, 3-stage split chain |
//! | `DeepSeekV3` | `DeepSeek` V3 … V4 | byte-level BPE, 3-stage split chain |
//! | `KimiK2` | Kimi K2 … K3 | byte-level BPE (tiktoken ranks) |
//! | `Glm5` | GLM-5.x (exact), GLM-4.x (near-exact) | byte-level BPE, `ignore_merges` |
//!
@@ -41,9 +41,9 @@ pub use self::{
/// process-wide tables (first use pays one zstd decode of ~0.5–1 MB).
#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)]
pub enum Encoding {
/// GPT-4o / o1 / GPT-5 (OpenAI default).
/// GPT-4o / o1 / GPT-5 (`OpenAI` default).
O200kBase,
/// GPT-3.5 / GPT-4 / older OpenAI.
/// GPT-3.5 / GPT-4 / older `OpenAI`.
Cl100kBase,
/// Claude 3 through Opus 4.6 (and every non-opus Claude < 5).
ClaudeV3,
@@ -55,7 +55,7 @@ pub enum Encoding {
ClaudeV5Sonnet,
/// Qwen 3.5 / 3.6 / 3.8 (248,044-token vocabulary, NFC input).
Qwen3,
/// DeepSeek V3 / V3.1 / V3.2 / R1 / V4 (identical base BPE).
/// `DeepSeek` V3 / V3.1 / V3.2 / R1 / V4 (identical base BPE).
DeepSeekV3,
/// Kimi K2 / K2.5 / K3 (163,584-token base vocabulary).
KimiK2,
+5 -9
View File
@@ -18,7 +18,7 @@ pub enum Splitter {
O200k,
/// tiktoken `cl100k_base` scanner.
Cl100k,
/// DeepSeek V3..V4 three-stage chain scanner.
/// `DeepSeek` V3..V4 three-stage chain scanner.
DeepSeek,
/// Kimi K2/K3 scanner.
Kimi,
@@ -40,14 +40,10 @@ impl Splitter {
}
/// Whether this splitter needs `&str` input (the engine transcodes
/// non-UTF-8 flavors before calling in). Always false at runtime; only
/// the test-only regex oracle transcodes.
pub fn is_regex(&self) -> bool {
#[cfg(test)]
if matches!(self, Self::Regex(_)) {
return true;
}
false
/// non-UTF-8 flavors before calling in).
#[cfg(test)]
pub const fn is_regex(&self) -> bool {
matches!(self, Self::Regex(_))
}
/// Feed every piece of `units` to `f`, in order, covering the input
+5 -6
View File
@@ -1,4 +1,4 @@
//! DeepSeek V3..V4 pre-tokenizer: hand-written codepoint scanner over `&[U]`.
//! `DeepSeek` V3..V4 pre-tokenizer: hand-written codepoint scanner over `&[U]`.
//!
//! Faithful port of the three-stage HF `Split(Isolated)` chain (see
//! `DEEPSEEK_STAGE_*` in `tables.rs`). Every stage applies to each piece
@@ -10,7 +10,7 @@
//! 2. `[一-龥぀-ゟ゠-ヿ]+` — Han U+4E00–9FA5 plus the contiguous
//! hiragana/katakana blocks U+3040–30FF.
//! 3. The main pattern:
//! - `[!"#$%&'()*+,\-./:;<=>?@\[\\\]^_`{|}~][A-Za-z]+` (one ASCII
//! - ``[!"#$%&'()*+,\-./:;<=>?@\[\\\]^_`{|}~][A-Za-z]+`` (one ASCII
//! punctuation codepoint glued to ASCII letters, e.g. `.NET`, `(foo`)
//! - `[^\r\n\p{L}\p{P}\p{S}]?[\p{L}\p{M}]+`
//! - ` ?[\p{P}\p{S}]+[\r\n]*`
@@ -72,7 +72,7 @@ fn isolated<U: Unit>(
/// `[一-龥぀-ゟ゠-ヿ]`.
#[inline]
fn is_cjk(c: char) -> bool {
const fn is_cjk(c: char) -> bool {
matches!(c as u32, 0x3040..=0x30FF | 0x4E00..=0x9FA5)
}
@@ -128,11 +128,10 @@ fn main_end<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
}
if c != '\r' && c != '\n' && !is_ps(c) {
// Not L (checked above), not P/S, not CR/LF: prefix-eligible.
if let Some((c2, n2)) = decode_at(units, pos + n) {
if is_lm(c2) {
if let Some((c2, n2)) = decode_at(units, pos + n)
&& is_lm(c2) {
return Some(lm_run_end(units, pos + n + n2));
}
}
}
// ` ?[\p{P}\p{S}]+[\r\n]*`
+1 -1
View File
@@ -64,7 +64,7 @@ pub mod cls {
}
#[inline]
pub fn is_ws(c: char) -> bool {
pub const fn is_ws(c: char) -> bool {
c.is_whitespace()
}
+1 -1
View File
@@ -1,6 +1,6 @@
//! Qwen3 (3.5/3.6/3.8) split pattern as a codepoint scanner.
//!
//! Reference (qwen3.8.tokenizer.json pre_tokenizer, = families.json qwen3):
//! Reference (qwen3.8.tokenizer.json `pre_tokenizer`, = families.json qwen3):
//! ```text
//! (?i:'s|'t|'re|'ve|'m|'ll|'d)
//! |[^\r\n\p{L}\p{N}]?[\p{L}\p{M}]+
+13 -13
View File
@@ -132,6 +132,19 @@ static GLM5: LazyLock<BpeEncoding> = LazyLock::new(|| BpeEncoding {
ignore_merges: true,
});
/// Resolve the BPE encoding for a non-Claude family.
pub(crate) fn bpe_for(enc: Encoding) -> &'static BpeEncoding {
match enc {
Encoding::O200kBase => &O200K_BASE,
Encoding::Cl100kBase => &CL100K_BASE,
Encoding::Qwen3 => &QWEN3,
Encoding::DeepSeekV3 => &DEEPSEEK3,
Encoding::KimiK2 => &KIMI_K2,
Encoding::Glm5 => &GLM5,
_ => unreachable!("claude families never reach bpe_for"),
}
}
#[cfg(test)]
mod glm_scan_tests {
//! Differential justifying the cl100k-scanner alias: piece boundaries
@@ -212,16 +225,3 @@ mod glm_scan_tests {
}
}
}
/// Resolve the BPE encoding for a non-Claude family.
pub(crate) fn bpe_for(enc: Encoding) -> &'static BpeEncoding {
match enc {
Encoding::O200kBase => &O200K_BASE,
Encoding::Cl100kBase => &CL100K_BASE,
Encoding::Qwen3 => &QWEN3,
Encoding::DeepSeekV3 => &DEEPSEEK3,
Encoding::KimiK2 => &KIMI_K2,
Encoding::Glm5 => &GLM5,
_ => unreachable!("claude families never reach bpe_for"),
}
}
+2 -2
View File
@@ -1,5 +1,5 @@
//! Golden-fixture test: DeepSeekV3 vs reference HF tokenizers encode
//! (add_special_tokens=False) over cache/deepseek-v4.tokenizer.json.
//! Golden-fixture test: `DeepSeekV3` vs reference HF tokenizers encode
//! (`add_special_tokens=False`) over cache/deepseek-v4.tokenizer.json.
//! Base BPE is identical V3..V4 (verified upstream: same vocab + merges
//! hash), so one encoding covers the whole family.
+3 -3
View File
@@ -1,5 +1,5 @@
//! Golden-fixture test: Glm5 vs reference HF tokenizers
//! (add_special_tokens=False).
//! (`add_special_tokens=False`).
//!
//! GLM-4.x note: GLM-5 is an ID-preserving superset of GLM-4.x (~3.5k extra
//! merges), so GLM-4.x counts are near-exact under this table — no separate
@@ -36,10 +36,10 @@ fn glm5_matches_reference() {
}
}
/// Directed ignore_merges semantics: ' 参考' exists in vocab (99855) but
/// Directed `ignore_merges` semantics: ' 参考' exists in vocab (99855) but
/// HF's greedy merge order never forms it from bytes; normal merge-loop BPE
/// yields [26767, 224, 98580] (verified with the Python reference with
/// ignore_merges=false). The whole-piece short-circuit must win.
/// `ignore_merges=false`). The whole-piece short-circuit must win.
#[test]
fn glm5_ignore_merges_whole_piece_wins() {
assert_eq!(Encoding::Glm5.encode(" 参考").unwrap(), vec![99855]);
+1 -1
View File
@@ -1,4 +1,4 @@
//! Golden-fixture test: KimiK2 vs reference tiktoken encode_ordinary.
//! Golden-fixture test: `KimiK2` vs reference tiktoken `encode_ordinary`.
use crate::utok::Encoding;
+2 -2
View File
@@ -1,4 +1,4 @@
//! OpenAI family tests: golden fixtures from Python tiktoken, plus a
//! `OpenAI` family tests: golden fixtures from Python tiktoken, plus a
//! differential test against tiktoken-rs over the corpus and seeded
//! randomized strings.
@@ -8,7 +8,7 @@ use crate::utok::Encoding;
#[derive(Deserialize)]
struct Fixture {
#[allow(dead_code)]
#[allow(dead_code, reason = "fixture provenance is deserialized but not asserted")]
generator: String,
cases: Vec<Case>,
}
+1 -1
View File
@@ -1,5 +1,5 @@
//! Golden-fixture test: Qwen3 vs reference HF tokenizers encode
//! (add_special_tokens=false), including NFC normalization and the
//! (`add_special_tokens=false`), including NFC normalization and the
//! dead-rank (merge-unreachable vocab entry) regressions.
use crate::utok::Encoding;
+3 -3
View File
@@ -119,7 +119,7 @@ impl Unit for u32 {
#[inline]
fn encode(cp: char, out: &mut [Self]) -> usize {
out[0] = cp as u32;
out[0] = cp as Self;
1
}
@@ -203,7 +203,7 @@ pub struct Cursor<'a, U: Unit> {
impl<'a, U: Unit> Cursor<'a, U> {
#[inline]
pub fn new(units: &'a [U]) -> Self {
pub const fn new(units: &'a [U]) -> Self {
Self { units, pos: 0 }
}
@@ -221,7 +221,7 @@ impl<'a, U: Unit> Cursor<'a, U> {
}
#[inline]
pub fn advance(&mut self, n: usize) {
pub const fn advance(&mut self, n: usize) {
self.pos += n;
}
}
@@ -1,5 +1,4 @@
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs/promises";
import * as path from "node:path";
import { Agent, type AgentMessage, type StreamFn } from "@oh-my-pi/pi-agent-core";
import * as compactionModule from "@oh-my-pi/pi-agent-core/compaction";
@@ -14,7 +13,6 @@ import {
loadExtensionFromFactory,
loadExtensions,
} from "@oh-my-pi/pi-coding-agent/extensibility/extensions";
import { resolveLocalUrlToPath } from "@oh-my-pi/pi-coding-agent/internal-urls";
import { SecretObfuscator } from "@oh-my-pi/pi-coding-agent/secrets";
import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";