feat(pi-natives/tools): implemented utok tokenizer for multiple models
- Replaced the `ctok` implementation with the `utok` universal tokenizer supporting multiple model families and UTF text encodings. - Added tokenizer support and embedding data for Qwen3, DeepSeek V3, Kimi K2, and GLM-5 model variants. - Added fixture generation scripts, vocabulary packers, and golden test suites for validating tokenization parity. - Updated dependency requirements and Bazel workspace definitions for new crates and tools.
This commit is contained in:
Generated
+46
-2
@@ -5059,6 +5059,7 @@ dependencies = [
|
|||||||
"clipboard-win",
|
"clipboard-win",
|
||||||
"core-graphics",
|
"core-graphics",
|
||||||
"enigo",
|
"enigo",
|
||||||
|
"fancy-regex 0.16.2",
|
||||||
"flume",
|
"flume",
|
||||||
"fontdue",
|
"fontdue",
|
||||||
"foreign-types",
|
"foreign-types",
|
||||||
@@ -5108,6 +5109,7 @@ dependencies = [
|
|||||||
"uiautomation",
|
"uiautomation",
|
||||||
"unicode-normalization",
|
"unicode-normalization",
|
||||||
"unicode-properties",
|
"unicode-properties",
|
||||||
|
"unicode-script",
|
||||||
"unicode-segmentation",
|
"unicode-segmentation",
|
||||||
"unicode-width 0.2.2",
|
"unicode-width 0.2.2",
|
||||||
"windows-sys 0.61.2",
|
"windows-sys 0.61.2",
|
||||||
@@ -5115,7 +5117,9 @@ dependencies = [
|
|||||||
"x11rb",
|
"x11rb",
|
||||||
"xcap",
|
"xcap",
|
||||||
"xkeysym",
|
"xkeysym",
|
||||||
|
"xutf",
|
||||||
"xxhash-rust",
|
"xxhash-rust",
|
||||||
|
"zstd",
|
||||||
"zune-jpeg",
|
"zune-jpeg",
|
||||||
]
|
]
|
||||||
|
|
||||||
@@ -7519,9 +7523,15 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "unicode-properties"
|
name = "unicode-properties"
|
||||||
version = "0.1.4"
|
version = "0.1.3"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "7df058c713841ad818f1dc5d3fd88063241cc61f49f5fbea4b951e8cf5a8d71d"
|
checksum = "e70f2a8b45122e719eb623c01822704c4e0907e7e426a05927e1a1cfff5b75d0"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "unicode-script"
|
||||||
|
version = "0.5.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "9fb421b350c9aff471779e262955939f565ec18b86c15364e6bdf0d662ca7c1f"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "unicode-segmentation"
|
name = "unicode-segmentation"
|
||||||
@@ -8831,6 +8841,12 @@ version = "0.8.29"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "e450f9b2ed1dff33c94c12589a87338689467b9c4f5d8a5710bd09a847d2c8a7"
|
checksum = "e450f9b2ed1dff33c94c12589a87338689467b9c4f5d8a5710bd09a847d2c8a7"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "xutf"
|
||||||
|
version = "1.2.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "fa2ab198275c47f70ceb92678c000b1bb9468a52297c5ce20c5ab5e99fa2c33d"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "xxhash-rust"
|
name = "xxhash-rust"
|
||||||
version = "0.8.18"
|
version = "0.8.18"
|
||||||
@@ -9099,6 +9115,34 @@ version = "1.0.23"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
|
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd"
|
||||||
|
version = "0.13.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a"
|
||||||
|
dependencies = [
|
||||||
|
"zstd-safe",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd-safe"
|
||||||
|
version = "7.2.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "8f49c4d5f0abb602a93fb8736af2a4f4dd9512e36f7f570d66e65ff867ed3b9d"
|
||||||
|
dependencies = [
|
||||||
|
"zstd-sys",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd-sys"
|
||||||
|
version = "2.0.16+zstd.1.5.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "91e19ebc2adc8f83e43039e79776e3fda8ca919132d68a1fed6a5faca2683748"
|
||||||
|
dependencies = [
|
||||||
|
"cc",
|
||||||
|
"pkg-config",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "zune-core"
|
name = "zune-core"
|
||||||
version = "0.5.3"
|
version = "0.5.3"
|
||||||
|
|||||||
+4
-1
@@ -234,7 +234,7 @@ regex = "1"
|
|||||||
similar = "3.1.0"
|
similar = "3.1.0"
|
||||||
unicode-segmentation = "1.13"
|
unicode-segmentation = "1.13"
|
||||||
unicode-normalization = "0.1"
|
unicode-normalization = "0.1"
|
||||||
unicode-properties = "0.1"
|
unicode-properties = "=0.1.3" # Unicode 16 - must match fancy-regex/HF/CPython oracles (utok)
|
||||||
unicode-width = "0.2"
|
unicode-width = "0.2"
|
||||||
fontdue = { version = "0.9", default-features = false }
|
fontdue = { version = "0.9", default-features = false }
|
||||||
|
|
||||||
@@ -336,6 +336,9 @@ html-to-markdown-rs = { version = "3.9.2", default-features = false }
|
|||||||
# ──────────────────────────────────────────────────────────────────────────────
|
# ──────────────────────────────────────────────────────────────────────────────
|
||||||
# Tokenization
|
# Tokenization
|
||||||
# ──────────────────────────────────────────────────────────────────────────────
|
# ──────────────────────────────────────────────────────────────────────────────
|
||||||
|
xutf = "1.4"
|
||||||
|
zstd = "0.13"
|
||||||
|
fancy-regex = "0.16"
|
||||||
tiktoken-rs = "0.11"
|
tiktoken-rs = "0.11"
|
||||||
|
|
||||||
# ──────────────────────────────────────────────────────────────────────────────
|
# ──────────────────────────────────────────────────────────────────────────────
|
||||||
|
|||||||
@@ -121,6 +121,17 @@ register_toolchains(
|
|||||||
"@zig_sdk//libc_aware/toolchain:linux_arm64_musl",
|
"@zig_sdk//libc_aware/toolchain:linux_arm64_musl",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# --- Host tools (crate_universe splicing / lockfile generation) ----------------
|
||||||
|
# Must be nightly: cargo's rustc target probe inherits user-level RUSTFLAGS
|
||||||
|
# (~/.cargo/config.toml commonly carries -Z flags on this repo's nightly
|
||||||
|
# toolchain), and a stable host rustc rejects them. Match rust.toolchain.
|
||||||
|
rust_host_tools = use_extension("@rules_rust//rust:extensions.bzl", "rust_host_tools")
|
||||||
|
rust_host_tools.host_tools(
|
||||||
|
name = "rust_host_tools_nightly",
|
||||||
|
version = "nightly/2026-04-29",
|
||||||
|
)
|
||||||
|
use_repo(rust_host_tools, "rust_host_tools_nightly")
|
||||||
|
|
||||||
# --- Third-party crates (crate_universe over the cargo workspace) --------------
|
# --- Third-party crates (crate_universe over the cargo workspace) --------------
|
||||||
crate = use_extension("@rules_rust//crate_universe:extensions.bzl", "crate")
|
crate = use_extension("@rules_rust//crate_universe:extensions.bzl", "crate")
|
||||||
crate.render_config(
|
crate.render_config(
|
||||||
@@ -131,6 +142,9 @@ crate.render_config(
|
|||||||
crate.from_cargo(
|
crate.from_cargo(
|
||||||
name = "crates",
|
name = "crates",
|
||||||
cargo_lockfile = "//:Cargo.lock",
|
cargo_lockfile = "//:Cargo.lock",
|
||||||
|
# Nightly host tools (see the rust_host_tools extension above): the
|
||||||
|
# splice's rustc probe must accept user-level -Z RUSTFLAGS.
|
||||||
|
host_tools = "@rust_host_tools_nightly",
|
||||||
# Exactly the shipped addon triples: crate BUILD files get
|
# Exactly the shipped addon triples: crate BUILD files get
|
||||||
# target_compatible_with selects over this set (defaults omit darwin-x64
|
# target_compatible_with selects over this set (defaults omit darwin-x64
|
||||||
# and musl), and features/deps resolve per-triple from Cargo.lock.
|
# and musl), and features/deps resolve per-triple from Cargo.lock.
|
||||||
|
|||||||
Generated
+71
-18
File diff suppressed because one or more lines are too long
@@ -18,8 +18,9 @@ rust_shared_library(
|
|||||||
compile_data = glob([
|
compile_data = glob([
|
||||||
"src/syntaxes/*.sublime-syntax",
|
"src/syntaxes/*.sublime-syntax",
|
||||||
"src/fonts/*",
|
"src/fonts/*",
|
||||||
"src/ctok/data/*.bin",
|
"data/*.bin.zst",
|
||||||
"src/ctok/testdata/*.json",
|
"src/utok/claude/testdata/*.json",
|
||||||
|
"fixtures/*.json",
|
||||||
]),
|
]),
|
||||||
crate_features = [],
|
crate_features = [],
|
||||||
edition = "2024",
|
edition = "2024",
|
||||||
|
|||||||
@@ -60,13 +60,15 @@ serde.workspace = true
|
|||||||
serde_json.workspace = true
|
serde_json.workspace = true
|
||||||
smallvec.workspace = true
|
smallvec.workspace = true
|
||||||
syntect.workspace = true
|
syntect.workspace = true
|
||||||
tiktoken-rs.workspace = true
|
|
||||||
tokio.workspace = true
|
tokio.workspace = true
|
||||||
tokio-util.workspace = true
|
tokio-util.workspace = true
|
||||||
toml.workspace = true
|
toml.workspace = true
|
||||||
unicode-segmentation.workspace = true
|
unicode-segmentation.workspace = true
|
||||||
unicode-normalization.workspace = true
|
unicode-normalization.workspace = true
|
||||||
unicode-properties.workspace = true
|
unicode-properties.workspace = true
|
||||||
|
unicode-script.workspace = true
|
||||||
|
xutf.workspace = true
|
||||||
|
zstd.workspace = true
|
||||||
unicode-width.workspace = true
|
unicode-width.workspace = true
|
||||||
xxhash-rust.workspace = true
|
xxhash-rust.workspace = true
|
||||||
|
|
||||||
@@ -76,6 +78,8 @@ atspi = { version = "=0.30.0", features = ["tokio", "zbus"] }
|
|||||||
pipewire = { version = "=0.9.2", optional = true }
|
pipewire = { version = "=0.9.2", optional = true }
|
||||||
reis = { version = "=0.5.0", features = ["tokio"] }
|
reis = { version = "=0.5.0", features = ["tokio"] }
|
||||||
x11rb = { version = "=0.13.2", features = ["randr", "xinput", "xtest"] }
|
x11rb = { version = "=0.13.2", features = ["randr", "xinput", "xtest"] }
|
||||||
|
fancy-regex.workspace = true # utok scanner differential oracle
|
||||||
|
tiktoken-rs.workspace = true # utok openai differential oracle
|
||||||
xkeysym = "=0.2.1"
|
xkeysym = "=0.2.1"
|
||||||
|
|
||||||
[target.'cfg(any(target_os = "macos", target_os = "windows"))'.dependencies]
|
[target.'cfg(any(target_os = "macos", target_os = "windows"))'.dependencies]
|
||||||
@@ -102,6 +106,8 @@ uiautomation = "=0.25.0"
|
|||||||
winreg.workspace = true
|
winreg.workspace = true
|
||||||
|
|
||||||
[dev-dependencies]
|
[dev-dependencies]
|
||||||
|
fancy-regex.workspace = true # utok scanner differential oracle
|
||||||
|
tiktoken-rs.workspace = true # utok openai differential oracle
|
||||||
xkeysym = "=0.2.1"
|
xkeysym = "=0.2.1"
|
||||||
|
|
||||||
[build-dependencies]
|
[build-dependencies]
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ Copyright (c) 2026 Sander Land
|
|||||||
implementation: https://github.com/sanderland/ctok)
|
implementation: https://github.com/sanderland/ctok)
|
||||||
Copyright (c) 2026 Can Bölük and the Oh My Pi contributors
|
Copyright (c) 2026 Can Bölük and the Oh My Pi contributors
|
||||||
(Rust implementation and the compact binary vocabulary encoding in
|
(Rust implementation and the compact binary vocabulary encoding in
|
||||||
crates/pi-natives/src/ctok)
|
crates/pi-natives/src/utok/claude)
|
||||||
|
|
||||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
of this software and associated documentation files (the "Software"), to deal
|
of this software and associated documentation files (the "Software"), to deal
|
||||||
@@ -0,0 +1,34 @@
|
|||||||
|
# utok vocabulary data
|
||||||
|
|
||||||
|
`ctok_v3.bin.zst` and `ctok_v4_7.bin.zst` are **generated** — do not
|
||||||
|
hand-edit. They are compacted from the measured vocabulary files of
|
||||||
|
[sanderland/ctok](https://github.com/sanderland/ctok) v1.0.0 (revision
|
||||||
|
`df3b59b5e645289a5eadc8e24036b99d39c333c4`), MIT licensed — see
|
||||||
|
`LICENSE.ctok`. The vocabulary data is Sander Land's measurement work
|
||||||
|
("On the biology of Claude's tokenizer",
|
||||||
|
<https://tokencontributions.substack.com/p/on-the-biology-of-claudes-tokenizer>);
|
||||||
|
the Rust implementation in `../src/utok/claude/` is this repository's own.
|
||||||
|
|
||||||
|
Upstream ships every piece with a `count_tokens` witness probe; compaction
|
||||||
|
drops that metadata, parses the public `⟨bow⟩the⟨eow⟩` key notation into the
|
||||||
|
compact C0 marker alphabet (single bytes `0x01`–`0x05`; safe because `nfc`
|
||||||
|
strips C0 controls from input), adds the glued contraction spellings, and
|
||||||
|
front-codes the sorted piece list into the version-2 binary format produced
|
||||||
|
by `../tools/gen-ctok-vocab.ts` (~4.7 MB of upstream JSON →
|
||||||
|
~254 KB front-coded → ~106 KB after zstd -19).
|
||||||
|
|
||||||
|
Regenerate the front-coded binaries, then compress them here:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
cd ../tools
|
||||||
|
bun gen-ctok-vocab.ts # fetch upstream, emit raw bins into cache/
|
||||||
|
bun pack-ctok.ts # zstd -19 into ../data/
|
||||||
|
```
|
||||||
|
|
||||||
|
If the upstream pin moves, also regenerate
|
||||||
|
`../src/utok/claude/testdata/fixtures.json` against the same ctok release
|
||||||
|
(see the fixture doc in `../src/utok/claude/mod.rs`).
|
||||||
|
|
||||||
|
The other `*.bin.zst` files here are the UTOK1 BPE rank tables packed by
|
||||||
|
the per-family scripts in `../tools/` (container format and per-family
|
||||||
|
split specs: `families.json` in this directory).
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,108 @@
|
|||||||
|
{
|
||||||
|
"container": "UTOK1: magic 'UTOK1\\n' (6B), u32le entry count, then per entry varint(len)+raw token bytes; rank = entry index (contiguous, assert at pack time). Whole file zstd -19 -> data/<name>.bin.zst",
|
||||||
|
"o200k_base": {
|
||||||
|
"source": "openaipublic .tiktoken (cache/o200k_base.tiktoken)",
|
||||||
|
"vocab": 199998,
|
||||||
|
"pattern": "VERIFY against tiktoken-rs o200k_base source"
|
||||||
|
},
|
||||||
|
"cl100k_base": {
|
||||||
|
"source": "cache/cl100k_base.tiktoken",
|
||||||
|
"vocab": 100256,
|
||||||
|
"pattern": "VERIFY against tiktoken-rs cl100k_base source"
|
||||||
|
},
|
||||||
|
"qwen3": {
|
||||||
|
"source": "cache/qwen3.8.tokenizer.json (Qwen/Qwen3.8-27B; identical for Qwen3.5/3.6)",
|
||||||
|
"normalizer": "NFC",
|
||||||
|
"pre": {
|
||||||
|
"type": "Sequence",
|
||||||
|
"pretokenizers": [
|
||||||
|
{
|
||||||
|
"type": "Split",
|
||||||
|
"pattern": {
|
||||||
|
"Regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+"
|
||||||
|
},
|
||||||
|
"behavior": "Isolated",
|
||||||
|
"invert": false
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": "ByteLevel",
|
||||||
|
"add_prefix_space": false,
|
||||||
|
"trim_offsets": false,
|
||||||
|
"use_regex": false
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"note": "HF vocab id 0 decoded to '|' — NOT plain GPT-2 byte alphabet; agent must determine actual vocab string encoding before packing"
|
||||||
|
},
|
||||||
|
"deepseek3": {
|
||||||
|
"source": "cache/deepseek-v4.tokenizer.json (base BPE identical V3..V4, verified: same 128000 vocab + 127741 merges hash)",
|
||||||
|
"normalizer": null,
|
||||||
|
"pre": {
|
||||||
|
"type": "Sequence",
|
||||||
|
"pretokenizers": [
|
||||||
|
{
|
||||||
|
"type": "Split",
|
||||||
|
"pattern": {
|
||||||
|
"Regex": "\\p{N}{1,3}"
|
||||||
|
},
|
||||||
|
"behavior": "Isolated",
|
||||||
|
"invert": false
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": "Split",
|
||||||
|
"pattern": {
|
||||||
|
"Regex": "[一-龥-ゟ゠-ヿ]+"
|
||||||
|
},
|
||||||
|
"behavior": "Isolated",
|
||||||
|
"invert": false
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": "Split",
|
||||||
|
"pattern": {
|
||||||
|
"Regex": "[!\"#$%&'()*+,\\-./:;<=>?@\\[\\\\\\]^_`{|}~][A-Za-z]+|[^\r\n\\p{L}\\p{P}\\p{S}]?[\\p{L}\\p{M}]+| ?[\\p{P}\\p{S}]+[\r\n]*|\\s*[\r\n]+|\\s+(?!\\S)|\\s+"
|
||||||
|
},
|
||||||
|
"behavior": "Isolated",
|
||||||
|
"invert": false
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": "ByteLevel",
|
||||||
|
"add_prefix_space": false,
|
||||||
|
"trim_offsets": true,
|
||||||
|
"use_regex": false
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"addedTokensExcluded": 1283
|
||||||
|
},
|
||||||
|
"kimi_k2": {
|
||||||
|
"source": "cache/kimi.tiktoken.model (K3; base identical K2)",
|
||||||
|
"vocab": 163584,
|
||||||
|
"pattern": "[\\p{Han}]+|[^\\r\\n\\p{L}\\p{N}]?[\\p{Lu}\\p{Lt}\\p{Lm}\\p{Lo}\\p{M}&&[^\\p{Han}]]*[\\p{Ll}\\p{Lm}\\p{Lo}\\p{M}&&[^\\p{Han}]]+(?i:'s|'t|'re|'ve|'m|'ll|'d)?|[^\\r\\n\\p{L}\\p{N}]?[\\p{Lu}\\p{Lt}\\p{Lm}\\p{Lo}\\p{M}&&[^\\p{Han}]]+[\\p{Ll}\\p{Lm}\\p{Lo}\\p{M}&&[^\\p{Han}]]*(?i:'s|'t|'re|'ve|'m|'ll|'d)?|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
||||||
|
"patternNote": "uses class intersection && plus lookahead; from tokenization_kimi.py"
|
||||||
|
},
|
||||||
|
"glm5": {
|
||||||
|
"source": "cache/glm-5.tokenizer.json (zai-org/GLM-5; superset of GLM-4.x, ids preserved)",
|
||||||
|
"vocab": 154820,
|
||||||
|
"normalizer": null,
|
||||||
|
"pre": {
|
||||||
|
"type": "Sequence",
|
||||||
|
"pretokenizers": [
|
||||||
|
{
|
||||||
|
"type": "Split",
|
||||||
|
"pattern": {
|
||||||
|
"Regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?\\p{L}+|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+"
|
||||||
|
},
|
||||||
|
"behavior": "Isolated",
|
||||||
|
"invert": false
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": "ByteLevel",
|
||||||
|
"add_prefix_space": false,
|
||||||
|
"trim_offsets": true,
|
||||||
|
"use_regex": false
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"ignore_merges": true
|
||||||
|
}
|
||||||
|
}
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
File diff suppressed because one or more lines are too long
@@ -0,0 +1,20 @@
|
|||||||
|
[
|
||||||
|
"",
|
||||||
|
" ",
|
||||||
|
" \t\t\n\n\r\n ",
|
||||||
|
"hello world",
|
||||||
|
"Hello, World! It's a test. We'll see; don't worry, y'all've been warned.",
|
||||||
|
"The quick brown fox jumps over the lazy dog. 1234567890 12 345 6789",
|
||||||
|
"fn main() {\n let x: Vec<u32> = (0..10).map(|i| i * 2).collect();\n println!(\"{x:?}\");\n}",
|
||||||
|
"{\"key\": [1, 2.5, -3e8], \"nested\": {\"a\": null, \"b\": true}}",
|
||||||
|
"东京は日本の首都であり、世界で最も人口の多い都市圏の一つです。深度求索发布了新一代基座模型。",
|
||||||
|
"한국어 텍스트도 테스트합니다. 안녕하세요!",
|
||||||
|
"Многоязычный текст: русский, ελληνικά, עברית, العربية.",
|
||||||
|
"naïve café résumé — em-dash…ellipsis",
|
||||||
|
"naïve café",
|
||||||
|
"👍🏽 emoji test 👨👩👧👦 family, flags 🇹🇷🇯🇵, math 𝕏≈∑∫",
|
||||||
|
"https://example.com/path?query=value&other=%20escaped#fragment",
|
||||||
|
"supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious",
|
||||||
|
" indented\n\tmixed whitespace runs\n\n\n",
|
||||||
|
"CamelCaseIdentifier snake_case_name SCREAMING_SNAKE kebab-case-name"
|
||||||
|
]
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,840 @@
|
|||||||
|
{
|
||||||
|
"generator": "uv run --with tokenizers tools/gen-glm-fixtures.py (tokenizers reference, add_special_tokens=False)",
|
||||||
|
"cases": [
|
||||||
|
{
|
||||||
|
"text": "",
|
||||||
|
"ids": [],
|
||||||
|
"count": 0
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " ",
|
||||||
|
"ids": [
|
||||||
|
220
|
||||||
|
],
|
||||||
|
"count": 1
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " \t\t\n\n\r\n ",
|
||||||
|
"ids": [
|
||||||
|
256,
|
||||||
|
33984,
|
||||||
|
319,
|
||||||
|
262
|
||||||
|
],
|
||||||
|
"count": 4
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "hello world",
|
||||||
|
"ids": [
|
||||||
|
14978,
|
||||||
|
1879
|
||||||
|
],
|
||||||
|
"count": 2
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "Hello, World! It's a test. We'll see; don't worry, y'all've been warned.",
|
||||||
|
"ids": [
|
||||||
|
9703,
|
||||||
|
11,
|
||||||
|
4337,
|
||||||
|
0,
|
||||||
|
1084,
|
||||||
|
594,
|
||||||
|
264,
|
||||||
|
1273,
|
||||||
|
13,
|
||||||
|
1205,
|
||||||
|
3278,
|
||||||
|
1490,
|
||||||
|
26,
|
||||||
|
1513,
|
||||||
|
944,
|
||||||
|
10950,
|
||||||
|
11,
|
||||||
|
379,
|
||||||
|
64374,
|
||||||
|
3003,
|
||||||
|
1012,
|
||||||
|
18650,
|
||||||
|
13
|
||||||
|
],
|
||||||
|
"count": 23
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "The quick brown fox jumps over the lazy dog. 1234567890 12 345 6789",
|
||||||
|
"ids": [
|
||||||
|
785,
|
||||||
|
3974,
|
||||||
|
13867,
|
||||||
|
38627,
|
||||||
|
34041,
|
||||||
|
916,
|
||||||
|
279,
|
||||||
|
15666,
|
||||||
|
5562,
|
||||||
|
13,
|
||||||
|
220,
|
||||||
|
108714,
|
||||||
|
100461,
|
||||||
|
21,
|
||||||
|
100928,
|
||||||
|
24,
|
||||||
|
15,
|
||||||
|
220,
|
||||||
|
98886,
|
||||||
|
220,
|
||||||
|
18,
|
||||||
|
100461,
|
||||||
|
220,
|
||||||
|
21,
|
||||||
|
100928,
|
||||||
|
24
|
||||||
|
],
|
||||||
|
"count": 26
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "fn main() {\n let x: Vec<u32> = (0..10).map(|i| i * 2).collect();\n println!(\"{x:?}\");\n}",
|
||||||
|
"ids": [
|
||||||
|
8821,
|
||||||
|
1887,
|
||||||
|
368,
|
||||||
|
341,
|
||||||
|
262,
|
||||||
|
1077,
|
||||||
|
856,
|
||||||
|
25,
|
||||||
|
11307,
|
||||||
|
34664,
|
||||||
|
101175,
|
||||||
|
29,
|
||||||
|
284,
|
||||||
|
320,
|
||||||
|
15,
|
||||||
|
496,
|
||||||
|
98668,
|
||||||
|
568,
|
||||||
|
2186,
|
||||||
|
22369,
|
||||||
|
72,
|
||||||
|
91,
|
||||||
|
600,
|
||||||
|
353,
|
||||||
|
220,
|
||||||
|
17,
|
||||||
|
568,
|
||||||
|
17362,
|
||||||
|
543,
|
||||||
|
262,
|
||||||
|
13742,
|
||||||
|
88193,
|
||||||
|
87,
|
||||||
|
75867,
|
||||||
|
20264,
|
||||||
|
92
|
||||||
|
],
|
||||||
|
"count": 36
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "{\"key\": [1, 2.5, -3e8], \"nested\": {\"a\": null, \"b\": true}}",
|
||||||
|
"ids": [
|
||||||
|
4913,
|
||||||
|
792,
|
||||||
|
788,
|
||||||
|
508,
|
||||||
|
16,
|
||||||
|
11,
|
||||||
|
220,
|
||||||
|
17,
|
||||||
|
13,
|
||||||
|
20,
|
||||||
|
11,
|
||||||
|
481,
|
||||||
|
18,
|
||||||
|
68,
|
||||||
|
23,
|
||||||
|
1125,
|
||||||
|
330,
|
||||||
|
58860,
|
||||||
|
788,
|
||||||
|
5212,
|
||||||
|
64,
|
||||||
|
788,
|
||||||
|
845,
|
||||||
|
11,
|
||||||
|
330,
|
||||||
|
65,
|
||||||
|
788,
|
||||||
|
830,
|
||||||
|
3417
|
||||||
|
],
|
||||||
|
"count": 29
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "东京は日本の首都であり、世界で最も人口の多い都市圏の一つです。深度求索发布了新一代基座模型。",
|
||||||
|
"ids": [
|
||||||
|
105935,
|
||||||
|
15310,
|
||||||
|
99799,
|
||||||
|
15755,
|
||||||
|
106552,
|
||||||
|
145713,
|
||||||
|
5373,
|
||||||
|
99011,
|
||||||
|
16147,
|
||||||
|
98430,
|
||||||
|
31742,
|
||||||
|
100742,
|
||||||
|
15755,
|
||||||
|
134584,
|
||||||
|
103266,
|
||||||
|
12268,
|
||||||
|
237,
|
||||||
|
138683,
|
||||||
|
58236,
|
||||||
|
37346,
|
||||||
|
1773,
|
||||||
|
102148,
|
||||||
|
98624,
|
||||||
|
99459,
|
||||||
|
108284,
|
||||||
|
110330,
|
||||||
|
98526,
|
||||||
|
99218,
|
||||||
|
100484,
|
||||||
|
1773
|
||||||
|
],
|
||||||
|
"count": 30
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "한국어 텍스트도 테스트합니다. 안녕하세요!",
|
||||||
|
"ids": [
|
||||||
|
23508,
|
||||||
|
130419,
|
||||||
|
30953,
|
||||||
|
10759,
|
||||||
|
43849,
|
||||||
|
52836,
|
||||||
|
47683,
|
||||||
|
10759,
|
||||||
|
71953,
|
||||||
|
52836,
|
||||||
|
60412,
|
||||||
|
13,
|
||||||
|
94372,
|
||||||
|
73588,
|
||||||
|
243,
|
||||||
|
90385,
|
||||||
|
0
|
||||||
|
],
|
||||||
|
"count": 17
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "Многоязычный текст: русский, ελληνικά, עברית, العربية.",
|
||||||
|
"ids": [
|
||||||
|
37793,
|
||||||
|
38592,
|
||||||
|
62539,
|
||||||
|
4552,
|
||||||
|
133834,
|
||||||
|
70363,
|
||||||
|
25,
|
||||||
|
145232,
|
||||||
|
11,
|
||||||
|
58738,
|
||||||
|
137790,
|
||||||
|
130624,
|
||||||
|
11,
|
||||||
|
17263,
|
||||||
|
95,
|
||||||
|
74926,
|
||||||
|
49904,
|
||||||
|
41995,
|
||||||
|
103,
|
||||||
|
11,
|
||||||
|
137930,
|
||||||
|
13
|
||||||
|
],
|
||||||
|
"count": 22
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "naïve café résumé — em-dash…ellipsis",
|
||||||
|
"ids": [
|
||||||
|
3376,
|
||||||
|
37377,
|
||||||
|
586,
|
||||||
|
51609,
|
||||||
|
9330,
|
||||||
|
1242,
|
||||||
|
963,
|
||||||
|
1959,
|
||||||
|
976,
|
||||||
|
1737,
|
||||||
|
988,
|
||||||
|
1940,
|
||||||
|
71373
|
||||||
|
],
|
||||||
|
"count": 13
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "naïve café",
|
||||||
|
"ids": [
|
||||||
|
77,
|
||||||
|
2143,
|
||||||
|
151464,
|
||||||
|
586,
|
||||||
|
40702,
|
||||||
|
53481
|
||||||
|
],
|
||||||
|
"count": 6
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "👍🏽 emoji test 👨👩👧👦 family, flags 🇹🇷🇯🇵, math 𝕏≈∑∫",
|
||||||
|
"ids": [
|
||||||
|
151901,
|
||||||
|
151821,
|
||||||
|
42123,
|
||||||
|
1273,
|
||||||
|
61370,
|
||||||
|
101,
|
||||||
|
124564,
|
||||||
|
151929,
|
||||||
|
124564,
|
||||||
|
151927,
|
||||||
|
124564,
|
||||||
|
151926,
|
||||||
|
2997,
|
||||||
|
11,
|
||||||
|
8041,
|
||||||
|
11157,
|
||||||
|
229,
|
||||||
|
117,
|
||||||
|
9281,
|
||||||
|
229,
|
||||||
|
115,
|
||||||
|
9281,
|
||||||
|
229,
|
||||||
|
107,
|
||||||
|
9281,
|
||||||
|
229,
|
||||||
|
113,
|
||||||
|
11,
|
||||||
|
6888,
|
||||||
|
80590,
|
||||||
|
243,
|
||||||
|
237,
|
||||||
|
153445,
|
||||||
|
120775,
|
||||||
|
153417
|
||||||
|
],
|
||||||
|
"count": 35
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "https://example.com/path?query=value&other=%20escaped#fragment",
|
||||||
|
"ids": [
|
||||||
|
2428,
|
||||||
|
1110,
|
||||||
|
8686,
|
||||||
|
905,
|
||||||
|
50642,
|
||||||
|
30,
|
||||||
|
1631,
|
||||||
|
46252,
|
||||||
|
5,
|
||||||
|
1575,
|
||||||
|
7846,
|
||||||
|
98360,
|
||||||
|
65346,
|
||||||
|
2,
|
||||||
|
41962
|
||||||
|
],
|
||||||
|
"count": 15
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious",
|
||||||
|
"ids": [
|
||||||
|
12770,
|
||||||
|
2962,
|
||||||
|
278,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570,
|
||||||
|
2256,
|
||||||
|
5416,
|
||||||
|
333,
|
||||||
|
4101,
|
||||||
|
321,
|
||||||
|
4532,
|
||||||
|
4580,
|
||||||
|
530,
|
||||||
|
307,
|
||||||
|
76570
|
||||||
|
],
|
||||||
|
"count": 201
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " indented\n\tmixed whitespace runs\n\n\n",
|
||||||
|
"ids": [
|
||||||
|
262,
|
||||||
|
1257,
|
||||||
|
15852,
|
||||||
|
198,
|
||||||
|
2109,
|
||||||
|
3286,
|
||||||
|
256,
|
||||||
|
36188,
|
||||||
|
257,
|
||||||
|
8472,
|
||||||
|
1406
|
||||||
|
],
|
||||||
|
"count": 11
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "CamelCaseIdentifier snake_case_name SCREAMING_SNAKE kebab-case-name",
|
||||||
|
"ids": [
|
||||||
|
25328,
|
||||||
|
301,
|
||||||
|
4207,
|
||||||
|
8713,
|
||||||
|
25187,
|
||||||
|
19061,
|
||||||
|
1269,
|
||||||
|
7531,
|
||||||
|
15903,
|
||||||
|
1718,
|
||||||
|
1098,
|
||||||
|
7326,
|
||||||
|
3390,
|
||||||
|
1962,
|
||||||
|
47427,
|
||||||
|
38278,
|
||||||
|
11489
|
||||||
|
],
|
||||||
|
"count": 17
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "智谱清言是由北京智谱华章科技有限公司开发的大语言模型。",
|
||||||
|
"ids": [
|
||||||
|
99126,
|
||||||
|
100789,
|
||||||
|
98691,
|
||||||
|
98856,
|
||||||
|
102047,
|
||||||
|
99334,
|
||||||
|
99126,
|
||||||
|
100789,
|
||||||
|
98762,
|
||||||
|
98648,
|
||||||
|
102595,
|
||||||
|
99495,
|
||||||
|
99707,
|
||||||
|
100132,
|
||||||
|
100484,
|
||||||
|
1773
|
||||||
|
],
|
||||||
|
"count": 16
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "你好,世界!这是一个测试。",
|
||||||
|
"ids": [
|
||||||
|
109377,
|
||||||
|
3837,
|
||||||
|
99011,
|
||||||
|
6313,
|
||||||
|
103974,
|
||||||
|
100838,
|
||||||
|
1773
|
||||||
|
],
|
||||||
|
"count": 7
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "人工智能正在改变世界,深度学习模型的参数规模不断增长。",
|
||||||
|
"ids": [
|
||||||
|
104668,
|
||||||
|
100296,
|
||||||
|
100070,
|
||||||
|
99011,
|
||||||
|
3837,
|
||||||
|
102148,
|
||||||
|
98935,
|
||||||
|
111595,
|
||||||
|
100955,
|
||||||
|
100200,
|
||||||
|
99421,
|
||||||
|
99816,
|
||||||
|
1773
|
||||||
|
],
|
||||||
|
"count": 13
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "中英文混排 mixed CJK and English 123 数字。",
|
||||||
|
"ids": [
|
||||||
|
98322,
|
||||||
|
103577,
|
||||||
|
99817,
|
||||||
|
98930,
|
||||||
|
9515,
|
||||||
|
356,
|
||||||
|
33905,
|
||||||
|
323,
|
||||||
|
6364,
|
||||||
|
220,
|
||||||
|
108714,
|
||||||
|
124543,
|
||||||
|
1773
|
||||||
|
],
|
||||||
|
"count": 13
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 全角空格和标点符号:《引号》、【括号】——破折号……省略号",
|
||||||
|
"ids": [
|
||||||
|
22382,
|
||||||
|
98402,
|
||||||
|
99025,
|
||||||
|
98745,
|
||||||
|
98639,
|
||||||
|
98327,
|
||||||
|
98593,
|
||||||
|
98428,
|
||||||
|
105423,
|
||||||
|
102462,
|
||||||
|
98662,
|
||||||
|
98760,
|
||||||
|
106843,
|
||||||
|
10899,
|
||||||
|
99231,
|
||||||
|
98760,
|
||||||
|
10953,
|
||||||
|
8544,
|
||||||
|
99222,
|
||||||
|
100048,
|
||||||
|
98760,
|
||||||
|
14044,
|
||||||
|
98737,
|
||||||
|
99260,
|
||||||
|
98760
|
||||||
|
],
|
||||||
|
"count": 25
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 参考",
|
||||||
|
"ids": [
|
||||||
|
99855
|
||||||
|
],
|
||||||
|
"count": 1
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 参考资料",
|
||||||
|
"ids": [
|
||||||
|
99924
|
||||||
|
],
|
||||||
|
"count": 1
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 而",
|
||||||
|
"ids": [
|
||||||
|
101502
|
||||||
|
],
|
||||||
|
"count": 1
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 者",
|
||||||
|
"ids": [
|
||||||
|
102222
|
||||||
|
],
|
||||||
|
"count": 1
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 王",
|
||||||
|
"ids": [
|
||||||
|
102322
|
||||||
|
],
|
||||||
|
"count": 1
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 参考龘",
|
||||||
|
"ids": [
|
||||||
|
26767,
|
||||||
|
224,
|
||||||
|
98580,
|
||||||
|
82225,
|
||||||
|
246
|
||||||
|
],
|
||||||
|
"count": 5
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 参考资料龘",
|
||||||
|
"ids": [
|
||||||
|
26767,
|
||||||
|
224,
|
||||||
|
98580,
|
||||||
|
99304,
|
||||||
|
82225,
|
||||||
|
246
|
||||||
|
],
|
||||||
|
"count": 6
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 而龘",
|
||||||
|
"ids": [
|
||||||
|
8905,
|
||||||
|
222,
|
||||||
|
234,
|
||||||
|
82225,
|
||||||
|
246
|
||||||
|
],
|
||||||
|
"count": 5
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 者龘",
|
||||||
|
"ids": [
|
||||||
|
8905,
|
||||||
|
222,
|
||||||
|
227,
|
||||||
|
82225,
|
||||||
|
246
|
||||||
|
],
|
||||||
|
"count": 5
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 王龘",
|
||||||
|
"ids": [
|
||||||
|
10231,
|
||||||
|
236,
|
||||||
|
233,
|
||||||
|
82225,
|
||||||
|
246
|
||||||
|
],
|
||||||
|
"count": 5
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "龘 参考",
|
||||||
|
"ids": [
|
||||||
|
82225,
|
||||||
|
246,
|
||||||
|
99855
|
||||||
|
],
|
||||||
|
"count": 3
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "龘 参考资料",
|
||||||
|
"ids": [
|
||||||
|
82225,
|
||||||
|
246,
|
||||||
|
99924
|
||||||
|
],
|
||||||
|
"count": 3
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "龘 而",
|
||||||
|
"ids": [
|
||||||
|
82225,
|
||||||
|
246,
|
||||||
|
101502
|
||||||
|
],
|
||||||
|
"count": 3
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "龘 者",
|
||||||
|
"ids": [
|
||||||
|
82225,
|
||||||
|
246,
|
||||||
|
102222
|
||||||
|
],
|
||||||
|
"count": 3
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "龘 王",
|
||||||
|
"ids": [
|
||||||
|
82225,
|
||||||
|
246,
|
||||||
|
102322
|
||||||
|
],
|
||||||
|
"count": 3
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,890 @@
|
|||||||
|
{
|
||||||
|
"generator": "tiktoken 0.14.0 Encoding(kimi, tokenization_kimi.py pat_str) encode_ordinary",
|
||||||
|
"cases": [
|
||||||
|
{
|
||||||
|
"text": "",
|
||||||
|
"ids": [],
|
||||||
|
"count": 0
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " ",
|
||||||
|
"ids": [
|
||||||
|
220
|
||||||
|
],
|
||||||
|
"count": 1
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " \t\t\n\n\r\n ",
|
||||||
|
"ids": [
|
||||||
|
99000,
|
||||||
|
382,
|
||||||
|
462,
|
||||||
|
274
|
||||||
|
],
|
||||||
|
"count": 4
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "hello world",
|
||||||
|
"ids": [
|
||||||
|
22931,
|
||||||
|
2695
|
||||||
|
],
|
||||||
|
"count": 2
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "Hello, World! It's a test. We'll see; don't worry, y'all've been warned.",
|
||||||
|
"ids": [
|
||||||
|
19180,
|
||||||
|
11,
|
||||||
|
6949,
|
||||||
|
0,
|
||||||
|
8629,
|
||||||
|
261,
|
||||||
|
1812,
|
||||||
|
13,
|
||||||
|
40251,
|
||||||
|
2050,
|
||||||
|
26,
|
||||||
|
4536,
|
||||||
|
13980,
|
||||||
|
11,
|
||||||
|
364,
|
||||||
|
120581,
|
||||||
|
7962,
|
||||||
|
1479,
|
||||||
|
43495,
|
||||||
|
13
|
||||||
|
],
|
||||||
|
"count": 20
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "The quick brown fox jumps over the lazy dog. 1234567890 12 345 6789",
|
||||||
|
"ids": [
|
||||||
|
1008,
|
||||||
|
5072,
|
||||||
|
16331,
|
||||||
|
69275,
|
||||||
|
60062,
|
||||||
|
1312,
|
||||||
|
276,
|
||||||
|
29292,
|
||||||
|
7751,
|
||||||
|
13,
|
||||||
|
220,
|
||||||
|
6694,
|
||||||
|
12972,
|
||||||
|
16242,
|
||||||
|
15,
|
||||||
|
220,
|
||||||
|
1042,
|
||||||
|
220,
|
||||||
|
18439,
|
||||||
|
220,
|
||||||
|
22523,
|
||||||
|
24
|
||||||
|
],
|
||||||
|
"count": 22
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "fn main() {\n let x: Vec<u32> = (0..10).map(|i| i * 2).collect();\n println!(\"{x:?}\");\n}",
|
||||||
|
"ids": [
|
||||||
|
10964,
|
||||||
|
2777,
|
||||||
|
539,
|
||||||
|
440,
|
||||||
|
274,
|
||||||
|
2391,
|
||||||
|
1288,
|
||||||
|
25,
|
||||||
|
20431,
|
||||||
|
52794,
|
||||||
|
1202,
|
||||||
|
29,
|
||||||
|
327,
|
||||||
|
347,
|
||||||
|
15,
|
||||||
|
690,
|
||||||
|
795,
|
||||||
|
1083,
|
||||||
|
3719,
|
||||||
|
32322,
|
||||||
|
72,
|
||||||
|
91,
|
||||||
|
1032,
|
||||||
|
397,
|
||||||
|
220,
|
||||||
|
17,
|
||||||
|
1083,
|
||||||
|
28994,
|
||||||
|
1094,
|
||||||
|
274,
|
||||||
|
47647,
|
||||||
|
28547,
|
||||||
|
90,
|
||||||
|
87,
|
||||||
|
93194,
|
||||||
|
47724,
|
||||||
|
92
|
||||||
|
],
|
||||||
|
"count": 37
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "{\"key\": [1, 2.5, -3e8], \"nested\": {\"a\": null, \"b\": true}}",
|
||||||
|
"ids": [
|
||||||
|
8264,
|
||||||
|
1319,
|
||||||
|
1289,
|
||||||
|
793,
|
||||||
|
16,
|
||||||
|
11,
|
||||||
|
220,
|
||||||
|
17,
|
||||||
|
13,
|
||||||
|
20,
|
||||||
|
11,
|
||||||
|
635,
|
||||||
|
18,
|
||||||
|
68,
|
||||||
|
23,
|
||||||
|
2287,
|
||||||
|
414,
|
||||||
|
82046,
|
||||||
|
1289,
|
||||||
|
9392,
|
||||||
|
64,
|
||||||
|
1289,
|
||||||
|
1704,
|
||||||
|
11,
|
||||||
|
414,
|
||||||
|
65,
|
||||||
|
1289,
|
||||||
|
1653,
|
||||||
|
5375
|
||||||
|
],
|
||||||
|
"count": 29
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "东京は日本の首都であり、世界で最も人口の多い都市圏の一つです。深度求索发布了新一代基座模型。",
|
||||||
|
"ids": [
|
||||||
|
26268,
|
||||||
|
8831,
|
||||||
|
4546,
|
||||||
|
4663,
|
||||||
|
26476,
|
||||||
|
126596,
|
||||||
|
343,
|
||||||
|
2243,
|
||||||
|
8846,
|
||||||
|
742,
|
||||||
|
20624,
|
||||||
|
9530,
|
||||||
|
4663,
|
||||||
|
561,
|
||||||
|
9102,
|
||||||
|
16807,
|
||||||
|
321,
|
||||||
|
237,
|
||||||
|
4663,
|
||||||
|
331,
|
||||||
|
30090,
|
||||||
|
54319,
|
||||||
|
292,
|
||||||
|
13165,
|
||||||
|
124414,
|
||||||
|
27613,
|
||||||
|
32136,
|
||||||
|
1320,
|
||||||
|
2989,
|
||||||
|
15662,
|
||||||
|
292
|
||||||
|
],
|
||||||
|
"count": 31
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "한국어 텍스트도 테스트합니다. 안녕하세요!",
|
||||||
|
"ids": [
|
||||||
|
17079,
|
||||||
|
54078,
|
||||||
|
28763,
|
||||||
|
83194,
|
||||||
|
235,
|
||||||
|
67178,
|
||||||
|
31474,
|
||||||
|
160618,
|
||||||
|
71146,
|
||||||
|
13,
|
||||||
|
102589,
|
||||||
|
34910,
|
||||||
|
243,
|
||||||
|
13181,
|
||||||
|
102898,
|
||||||
|
0
|
||||||
|
],
|
||||||
|
"count": 16
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "Многоязычный текст: русский, ελληνικά, עברית, العربية.",
|
||||||
|
"ids": [
|
||||||
|
24181,
|
||||||
|
23143,
|
||||||
|
2716,
|
||||||
|
51193,
|
||||||
|
3990,
|
||||||
|
26646,
|
||||||
|
148468,
|
||||||
|
25,
|
||||||
|
139598,
|
||||||
|
32890,
|
||||||
|
11,
|
||||||
|
22121,
|
||||||
|
61500,
|
||||||
|
14287,
|
||||||
|
8711,
|
||||||
|
125776,
|
||||||
|
11,
|
||||||
|
33480,
|
||||||
|
51589,
|
||||||
|
34403,
|
||||||
|
11,
|
||||||
|
120413,
|
||||||
|
12441,
|
||||||
|
13
|
||||||
|
],
|
||||||
|
"count": 24
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "naïve café résumé — em-dash…ellipsis",
|
||||||
|
"ids": [
|
||||||
|
4768,
|
||||||
|
44187,
|
||||||
|
367,
|
||||||
|
70591,
|
||||||
|
23606,
|
||||||
|
7671,
|
||||||
|
1941,
|
||||||
|
4275,
|
||||||
|
1516,
|
||||||
|
150743,
|
||||||
|
4635,
|
||||||
|
151078
|
||||||
|
],
|
||||||
|
"count": 12
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "naïve café",
|
||||||
|
"ids": [
|
||||||
|
59911,
|
||||||
|
136,
|
||||||
|
230,
|
||||||
|
367,
|
||||||
|
51158,
|
||||||
|
37484
|
||||||
|
],
|
||||||
|
"count": 6
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "👍🏽 emoji test 👨👩👧👦 family, flags 🇹🇷🇯🇵, math 𝕏≈∑∫",
|
||||||
|
"ids": [
|
||||||
|
64390,
|
||||||
|
235,
|
||||||
|
92949,
|
||||||
|
121,
|
||||||
|
68369,
|
||||||
|
1812,
|
||||||
|
130732,
|
||||||
|
101,
|
||||||
|
67963,
|
||||||
|
64390,
|
||||||
|
102,
|
||||||
|
67963,
|
||||||
|
64390,
|
||||||
|
100,
|
||||||
|
67963,
|
||||||
|
64390,
|
||||||
|
99,
|
||||||
|
3545,
|
||||||
|
11,
|
||||||
|
10141,
|
||||||
|
67496,
|
||||||
|
117,
|
||||||
|
32166,
|
||||||
|
115,
|
||||||
|
32166,
|
||||||
|
107,
|
||||||
|
32166,
|
||||||
|
113,
|
||||||
|
11,
|
||||||
|
15601,
|
||||||
|
161507,
|
||||||
|
237,
|
||||||
|
23149,
|
||||||
|
230,
|
||||||
|
12273,
|
||||||
|
239,
|
||||||
|
12273,
|
||||||
|
104
|
||||||
|
],
|
||||||
|
"count": 38
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "https://example.com/path?query=value&other=%20escaped#fragment",
|
||||||
|
"ids": [
|
||||||
|
1927,
|
||||||
|
1316,
|
||||||
|
17480,
|
||||||
|
1304,
|
||||||
|
73505,
|
||||||
|
30,
|
||||||
|
4964,
|
||||||
|
100520,
|
||||||
|
5,
|
||||||
|
2348,
|
||||||
|
9458,
|
||||||
|
577,
|
||||||
|
91967,
|
||||||
|
2,
|
||||||
|
46495
|
||||||
|
],
|
||||||
|
"count": 15
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious supercalifragilisticexpialidocious",
|
||||||
|
"ids": [
|
||||||
|
18785,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249,
|
||||||
|
3945,
|
||||||
|
12440,
|
||||||
|
361,
|
||||||
|
24772,
|
||||||
|
334,
|
||||||
|
6340,
|
||||||
|
8549,
|
||||||
|
682,
|
||||||
|
338,
|
||||||
|
162249
|
||||||
|
],
|
||||||
|
"count": 200
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " indented\n\tmixed whitespace runs\n\n\n",
|
||||||
|
"ids": [
|
||||||
|
274,
|
||||||
|
155238,
|
||||||
|
198,
|
||||||
|
3575,
|
||||||
|
4970,
|
||||||
|
256,
|
||||||
|
60368,
|
||||||
|
257,
|
||||||
|
11082,
|
||||||
|
3898
|
||||||
|
],
|
||||||
|
"count": 10
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "CamelCaseIdentifier snake_case_name SCREAMING_SNAKE kebab-case-name",
|
||||||
|
"ids": [
|
||||||
|
111821,
|
||||||
|
8081,
|
||||||
|
11633,
|
||||||
|
58108,
|
||||||
|
39188,
|
||||||
|
2928,
|
||||||
|
368,
|
||||||
|
8034,
|
||||||
|
154066,
|
||||||
|
1343,
|
||||||
|
10435,
|
||||||
|
4010,
|
||||||
|
2591,
|
||||||
|
58689,
|
||||||
|
56344,
|
||||||
|
16492
|
||||||
|
],
|
||||||
|
"count": 16
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "中文分词是自然语言处理的基础任务之一。月之暗面发布了千亿参数模型。",
|
||||||
|
"ids": [
|
||||||
|
16717,
|
||||||
|
671,
|
||||||
|
3961,
|
||||||
|
153780,
|
||||||
|
6826,
|
||||||
|
3934,
|
||||||
|
10301,
|
||||||
|
6282,
|
||||||
|
4584,
|
||||||
|
292,
|
||||||
|
892,
|
||||||
|
685,
|
||||||
|
5519,
|
||||||
|
696,
|
||||||
|
27613,
|
||||||
|
64960,
|
||||||
|
10517,
|
||||||
|
15662,
|
||||||
|
292
|
||||||
|
],
|
||||||
|
"count": 19
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "汉字漢字汉字漢字",
|
||||||
|
"ids": [
|
||||||
|
38325,
|
||||||
|
1435,
|
||||||
|
95,
|
||||||
|
1855,
|
||||||
|
38325,
|
||||||
|
1435,
|
||||||
|
95,
|
||||||
|
1855
|
||||||
|
],
|
||||||
|
"count": 8
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "中文English中文",
|
||||||
|
"ids": [
|
||||||
|
16717,
|
||||||
|
44372,
|
||||||
|
16717
|
||||||
|
],
|
||||||
|
"count": 3
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "中文english中文ENGLISH中文",
|
||||||
|
"ids": [
|
||||||
|
16717,
|
||||||
|
89901,
|
||||||
|
16717,
|
||||||
|
1254,
|
||||||
|
159820,
|
||||||
|
16717
|
||||||
|
],
|
||||||
|
"count": 6
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "GPT4发布于2023年3月14日,共有1750亿个参数。",
|
||||||
|
"ids": [
|
||||||
|
111021,
|
||||||
|
19,
|
||||||
|
5073,
|
||||||
|
624,
|
||||||
|
2975,
|
||||||
|
18,
|
||||||
|
532,
|
||||||
|
18,
|
||||||
|
892,
|
||||||
|
1626,
|
||||||
|
778,
|
||||||
|
11,
|
||||||
|
15298,
|
||||||
|
16484,
|
||||||
|
15,
|
||||||
|
2936,
|
||||||
|
439,
|
||||||
|
10517,
|
||||||
|
292
|
||||||
|
],
|
||||||
|
"count": 19
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "深度学习deep learning模型model需要大量GPU资源,如A100或H100。",
|
||||||
|
"ids": [
|
||||||
|
104448,
|
||||||
|
54283,
|
||||||
|
6857,
|
||||||
|
15662,
|
||||||
|
7940,
|
||||||
|
1777,
|
||||||
|
7825,
|
||||||
|
36399,
|
||||||
|
4127,
|
||||||
|
11,
|
||||||
|
697,
|
||||||
|
32,
|
||||||
|
1570,
|
||||||
|
1081,
|
||||||
|
39,
|
||||||
|
1570,
|
||||||
|
292
|
||||||
|
],
|
||||||
|
"count": 17
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "价格是99.99元,折扣为8.5折。",
|
||||||
|
"ids": [
|
||||||
|
96932,
|
||||||
|
3004,
|
||||||
|
13,
|
||||||
|
3004,
|
||||||
|
1218,
|
||||||
|
11,
|
||||||
|
28799,
|
||||||
|
441,
|
||||||
|
23,
|
||||||
|
13,
|
||||||
|
20,
|
||||||
|
4692,
|
||||||
|
292
|
||||||
|
],
|
||||||
|
"count": 13
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "他说:'It's fine'然后离开了。",
|
||||||
|
"ids": [
|
||||||
|
9085,
|
||||||
|
16229,
|
||||||
|
15881,
|
||||||
|
8201,
|
||||||
|
6,
|
||||||
|
3250,
|
||||||
|
24750,
|
||||||
|
292
|
||||||
|
],
|
||||||
|
"count": 8
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "中文Word中文WORD中文word",
|
||||||
|
"ids": [
|
||||||
|
16717,
|
||||||
|
16390,
|
||||||
|
16717,
|
||||||
|
11744,
|
||||||
|
16717,
|
||||||
|
2906
|
||||||
|
],
|
||||||
|
"count": 6
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "日本語テスト中文한국어",
|
||||||
|
"ids": [
|
||||||
|
4546,
|
||||||
|
63011,
|
||||||
|
25875,
|
||||||
|
33360,
|
||||||
|
16717,
|
||||||
|
17079,
|
||||||
|
54078,
|
||||||
|
28763
|
||||||
|
],
|
||||||
|
"count": 8
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "第一行\n第二行\r\n 第三行\t结束",
|
||||||
|
"ids": [
|
||||||
|
2122,
|
||||||
|
616,
|
||||||
|
198,
|
||||||
|
2944,
|
||||||
|
616,
|
||||||
|
462,
|
||||||
|
220,
|
||||||
|
220,
|
||||||
|
4028,
|
||||||
|
616,
|
||||||
|
197,
|
||||||
|
6123
|
||||||
|
],
|
||||||
|
"count": 12
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "中文 English 中文 English 中文",
|
||||||
|
"ids": [
|
||||||
|
16717,
|
||||||
|
9159,
|
||||||
|
220,
|
||||||
|
16717,
|
||||||
|
220,
|
||||||
|
9159,
|
||||||
|
256,
|
||||||
|
220,
|
||||||
|
16717
|
||||||
|
],
|
||||||
|
"count": 9
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "12345678901234567890",
|
||||||
|
"ids": [
|
||||||
|
6694,
|
||||||
|
12972,
|
||||||
|
16242,
|
||||||
|
16349,
|
||||||
|
18439,
|
||||||
|
22523,
|
||||||
|
2788
|
||||||
|
],
|
||||||
|
"count": 7
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "١٢٣٤٥٦٧٨٩٠ ๑๒๓ 一二三",
|
||||||
|
"ids": [
|
||||||
|
111991,
|
||||||
|
161592,
|
||||||
|
149,
|
||||||
|
96,
|
||||||
|
149,
|
||||||
|
97,
|
||||||
|
149,
|
||||||
|
98,
|
||||||
|
149,
|
||||||
|
99,
|
||||||
|
149,
|
||||||
|
100,
|
||||||
|
149,
|
||||||
|
101,
|
||||||
|
148209,
|
||||||
|
122674,
|
||||||
|
220,
|
||||||
|
7518,
|
||||||
|
239,
|
||||||
|
7518,
|
||||||
|
240,
|
||||||
|
7518,
|
||||||
|
241,
|
||||||
|
220,
|
||||||
|
160386
|
||||||
|
],
|
||||||
|
"count": 25
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "éé中文éé é́中文",
|
||||||
|
"ids": [
|
||||||
|
1941,
|
||||||
|
1941,
|
||||||
|
16717,
|
||||||
|
1941,
|
||||||
|
1941,
|
||||||
|
350,
|
||||||
|
37484,
|
||||||
|
37484,
|
||||||
|
16717
|
||||||
|
],
|
||||||
|
"count": 9
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": "foo!!!\n\nbar???\r\n",
|
||||||
|
"ids": [
|
||||||
|
10570,
|
||||||
|
20118,
|
||||||
|
382,
|
||||||
|
4566,
|
||||||
|
46518,
|
||||||
|
462
|
||||||
|
],
|
||||||
|
"count": 6
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " trailing spaces ",
|
||||||
|
"ids": [
|
||||||
|
256,
|
||||||
|
42711,
|
||||||
|
14803,
|
||||||
|
274
|
||||||
|
],
|
||||||
|
"count": 4
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"text": " 中文 a 中文 A1中文",
|
||||||
|
"ids": [
|
||||||
|
220,
|
||||||
|
16717,
|
||||||
|
261,
|
||||||
|
220,
|
||||||
|
16717,
|
||||||
|
401,
|
||||||
|
16,
|
||||||
|
16717
|
||||||
|
],
|
||||||
|
"count": 8
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
File diff suppressed because one or more lines are too long
File diff suppressed because it is too large
Load Diff
@@ -1,27 +0,0 @@
|
|||||||
# ctok vocabulary data
|
|
||||||
|
|
||||||
`ctok_v3.bin` and `ctok_v4_7.bin` are **generated** — do not hand-edit.
|
|
||||||
They are compacted from the measured vocabulary files of
|
|
||||||
[sanderland/ctok](https://github.com/sanderland/ctok) v1.0.0 (revision
|
|
||||||
`df3b59b5e645289a5eadc8e24036b99d39c333c4`), MIT licensed — see
|
|
||||||
`LICENSE.ctok`. The vocabulary data is Sander Land's measurement work
|
|
||||||
("On the biology of Claude's tokenizer",
|
|
||||||
<https://tokencontributions.substack.com/p/on-the-biology-of-claudes-tokenizer>);
|
|
||||||
the Rust implementation in the parent directory is this repository's own.
|
|
||||||
|
|
||||||
Upstream ships every piece with a `count_tokens` witness probe; compaction
|
|
||||||
drops that metadata, parses the public `⟨bow⟩the⟨eow⟩` key notation into the
|
|
||||||
internal marked form — one byte per marker, a third shorter than ctok's
|
|
||||||
noncharacter spelling in both the pieces and the stream they tile — adds the
|
|
||||||
glued contraction spellings, and front-codes the sorted piece list into the
|
|
||||||
binary format documented in `packages/natives/scripts/gen-ctok-vocab.ts`
|
|
||||||
(~4.7 MB of upstream JSON → ~254 KB embedded).
|
|
||||||
|
|
||||||
Regenerate with:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
bun --cwd packages/natives run gen:ctok
|
|
||||||
```
|
|
||||||
|
|
||||||
If the upstream pin moves, also regenerate `../testdata/fixtures.json`
|
|
||||||
against the same ctok release (see the fixture doc in `../mod.rs`).
|
|
||||||
@@ -28,7 +28,6 @@ pub mod audio;
|
|||||||
pub mod block;
|
pub mod block;
|
||||||
pub mod clipboard;
|
pub mod clipboard;
|
||||||
pub mod crash_handler;
|
pub mod crash_handler;
|
||||||
pub mod ctok;
|
|
||||||
pub mod desktop;
|
pub mod desktop;
|
||||||
pub mod devicecheck;
|
pub mod devicecheck;
|
||||||
pub mod diff;
|
pub mod diff;
|
||||||
@@ -46,6 +45,7 @@ pub mod live;
|
|||||||
pub mod pdf;
|
pub mod pdf;
|
||||||
pub mod sixel;
|
pub mod sixel;
|
||||||
pub mod snapcompact;
|
pub mod snapcompact;
|
||||||
|
pub mod utok;
|
||||||
pub use pi_ast::language;
|
pub use pi_ast::language;
|
||||||
|
|
||||||
pub mod power;
|
pub mod power;
|
||||||
|
|||||||
@@ -1,29 +1,29 @@
|
|||||||
//! Token counting via tiktoken-rs and the ctok Claude reconstruction.
|
//! Token counting via the embedded utok universal tokenizer.
|
||||||
//!
|
//!
|
||||||
//! Encodings:
|
//! Encodings:
|
||||||
//!
|
//!
|
||||||
//! - `O200kBase` — GPT-4o / o1 / GPT-5 (the modern `OpenAI` default).
|
//! - `O200kBase` — GPT-4o / o1 / GPT-5 (the modern `OpenAI` default).
|
||||||
//! - `Cl100kBase` — GPT-3.5 / GPT-4 / older models.
|
//! - `Cl100kBase` — GPT-3.5 / GPT-4 / older models.
|
||||||
//! - `ClaudeV3` / `ClaudeV47` / `ClaudeV5` — offline reconstructions of
|
//! - `ClaudeV3` / `ClaudeV47` / `ClaudeV5` / `ClaudeV5Sonnet` — offline
|
||||||
//! Anthropic's `count_tokens` (see [`crate::ctok`]): v3 serves Claude 3
|
//! reconstructions of Anthropic's `count_tokens` (see [`crate::utok`]): v3
|
||||||
//! through Opus 4.6, v4.7 serves Opus 4.7–4.9, v5 serves the 5-series.
|
//! serves Claude 3 through Opus 4.6, v4.7 serves Opus 4.7–4.9, v5 the
|
||||||
|
//! 5-series.
|
||||||
|
//! - `Qwen3` — Qwen 3.5 / 3.6 / 3.8 (248k vocabulary, NFC input).
|
||||||
|
//! - `DeepSeekV3` — `DeepSeek` V3 through V4 (identical base BPE).
|
||||||
|
//! - `KimiK2` — Kimi K2 through K3.
|
||||||
|
//! - `Glm5` — GLM-5.x exact; GLM-4.x near-exact (ID-preserving subset).
|
||||||
//!
|
//!
|
||||||
//! `o200k_base` is the default. For Claude models the ctok encodings count
|
//! `o200k_base` is the default. All vocabularies are zstd-embedded in the
|
||||||
//! exactly (message content, excluding the fixed per-message frame), so they
|
//! binary and decoded once on first use. Counting consumes the JS string's
|
||||||
//! are the right choice wherever the model is known to be Claude.
|
//! UTF-16 code units directly (no UTF-8 transcode on the napi crossing).
|
||||||
//!
|
|
||||||
//! BPE tables and the ctok vocabularies are embedded in the binary; encoders
|
|
||||||
//! are built once on first use and reused thereafter.
|
|
||||||
|
|
||||||
use std::sync::LazyLock;
|
use napi::{
|
||||||
|
JsString,
|
||||||
use napi::bindgen_prelude::Either;
|
bindgen_prelude::{Array, Either},
|
||||||
|
};
|
||||||
use napi_derive::napi;
|
use napi_derive::napi;
|
||||||
use pi_shell::rayon_global_pool_available;
|
|
||||||
use rayon::prelude::*;
|
|
||||||
use tiktoken_rs::{CoreBPE, cl100k_base, o200k_base};
|
|
||||||
|
|
||||||
use crate::ctok;
|
use crate::utok;
|
||||||
|
|
||||||
/// Tokenizer encoding to use.
|
/// Tokenizer encoding to use.
|
||||||
#[napi(string_enum)]
|
#[napi(string_enum)]
|
||||||
@@ -40,36 +40,29 @@ pub enum Encoding {
|
|||||||
ClaudeV5,
|
ClaudeV5,
|
||||||
/// Claude Sonnet/Fable 5+ (live-measured non-opus v5 frame).
|
/// Claude Sonnet/Fable 5+ (live-measured non-opus v5 frame).
|
||||||
ClaudeV5Sonnet,
|
ClaudeV5Sonnet,
|
||||||
|
/// Qwen 3.5 / 3.6 / 3.8 (248k vocabulary).
|
||||||
|
Qwen3,
|
||||||
|
/// `DeepSeek` V3 … V4 (identical base BPE).
|
||||||
|
DeepSeekV3,
|
||||||
|
/// Kimi K2 … K3.
|
||||||
|
KimiK2,
|
||||||
|
/// GLM-5.x exact; GLM-4.x near-exact.
|
||||||
|
Glm5,
|
||||||
}
|
}
|
||||||
|
|
||||||
static O200K: LazyLock<CoreBPE> =
|
impl Encoding {
|
||||||
LazyLock::new(|| o200k_base().expect("failed to initialize o200k_base BPE tables"));
|
fn utok(encoding: Option<Self>) -> utok::Encoding {
|
||||||
|
match encoding.unwrap_or(Self::O200kBase) {
|
||||||
static CL100K: LazyLock<CoreBPE> =
|
Self::O200kBase => utok::Encoding::O200kBase,
|
||||||
LazyLock::new(|| cl100k_base().expect("failed to initialize cl100k_base BPE tables"));
|
Self::Cl100kBase => utok::Encoding::Cl100kBase,
|
||||||
|
Self::ClaudeV3 => utok::Encoding::ClaudeV3,
|
||||||
/// A resolved counting backend: a BPE encoder or a ctok family.
|
Self::ClaudeV47 => utok::Encoding::ClaudeV47,
|
||||||
enum Counter {
|
Self::ClaudeV5 => utok::Encoding::ClaudeV5,
|
||||||
Bpe(&'static CoreBPE),
|
Self::ClaudeV5Sonnet => utok::Encoding::ClaudeV5Sonnet,
|
||||||
Claude(ctok::Family),
|
Self::Qwen3 => utok::Encoding::Qwen3,
|
||||||
}
|
Self::DeepSeekV3 => utok::Encoding::DeepSeekV3,
|
||||||
|
Self::KimiK2 => utok::Encoding::KimiK2,
|
||||||
impl Counter {
|
Self::Glm5 => utok::Encoding::Glm5,
|
||||||
fn resolve(encoding: Option<Encoding>) -> Self {
|
|
||||||
match encoding.unwrap_or(Encoding::O200kBase) {
|
|
||||||
Encoding::O200kBase => Self::Bpe(&O200K),
|
|
||||||
Encoding::Cl100kBase => Self::Bpe(&CL100K),
|
|
||||||
Encoding::ClaudeV3 => Self::Claude(ctok::Family::V3),
|
|
||||||
Encoding::ClaudeV47 => Self::Claude(ctok::Family::V47),
|
|
||||||
Encoding::ClaudeV5 => Self::Claude(ctok::Family::V5),
|
|
||||||
Encoding::ClaudeV5Sonnet => Self::Claude(ctok::Family::V5Sonnet),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn count(&self, text: &str) -> u32 {
|
|
||||||
match self {
|
|
||||||
Self::Bpe(bpe) => bpe.encode_ordinary(text).len() as u32,
|
|
||||||
Self::Claude(family) => ctok::content_token_count(text, *family),
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -77,22 +70,38 @@ impl Counter {
|
|||||||
/// Count tokens in `input`.
|
/// Count tokens in `input`.
|
||||||
///
|
///
|
||||||
/// `input` may be a single string or an array of strings; an array returns
|
/// `input` may be a single string or an array of strings; an array returns
|
||||||
/// the sum across all elements (encoded in parallel via rayon when the global
|
/// the sum across all elements. Always returns a single token total — use
|
||||||
/// pool is available). Always returns a single token total — use this for any
|
/// this for any aggregate budget question without paying a per-element napi
|
||||||
/// aggregate budget question without paying a per-element napi crossing.
|
/// crossing.
|
||||||
///
|
///
|
||||||
/// Measures user/model content, not wire-protocol tokens: BPE encodings use
|
/// Measures user/model content, not wire-protocol tokens: BPE encodings
|
||||||
/// ordinary encoding (no special-token handling) and the Claude encodings
|
/// use ordinary encoding (no special-token handling) and the Claude
|
||||||
/// count message content without the fixed per-message frame. Defaults to
|
/// encodings count message content without the fixed per-message frame.
|
||||||
/// `o200k_base`; pass a `Claude*` encoding for exact Claude counts.
|
/// Defaults to `o200k_base`; pass a `Claude*` encoding for exact Claude
|
||||||
|
/// counts, or the matching family encoding for Qwen/DeepSeek/Kimi/GLM.
|
||||||
#[napi]
|
#[napi]
|
||||||
pub fn count_tokens(input: Either<String, Vec<String>>, encoding: Option<Encoding>) -> u32 {
|
pub fn count_tokens(
|
||||||
let counter = Counter::resolve(encoding);
|
#[napi(ts_arg_type = "string | string[]")] input: Either<JsString, Array>,
|
||||||
|
encoding: Option<Encoding>,
|
||||||
|
) -> napi::Result<u32> {
|
||||||
|
let enc = Encoding::utok(encoding);
|
||||||
match input {
|
match input {
|
||||||
Either::A(text) => counter.count(&text),
|
Either::A(js_str) => {
|
||||||
Either::B(texts) if rayon_global_pool_available() => {
|
let text = js_str.into_utf16()?;
|
||||||
texts.par_iter().map(|s| counter.count(s)).sum()
|
let (_, units) = text.as_slice().split_last().expect("napi UTF-16 buffer has a terminator");
|
||||||
|
Ok(enc.count(units))
|
||||||
|
},
|
||||||
|
Either::B(array) => {
|
||||||
|
let mut total = 0u32;
|
||||||
|
for index in 0..array.len() {
|
||||||
|
let text = array
|
||||||
|
.get::<JsString>(index)?
|
||||||
|
.ok_or_else(|| napi::Error::from_reason("array changed during token counting"))?
|
||||||
|
.into_utf16()?;
|
||||||
|
let (_, units) = text.as_slice().split_last().expect("napi UTF-16 buffer has a terminator");
|
||||||
|
total += enc.count(units);
|
||||||
|
}
|
||||||
|
Ok(total)
|
||||||
},
|
},
|
||||||
Either::B(texts) => texts.iter().map(|s| counter.count(s)).sum(),
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,381 @@
|
|||||||
|
//! Core byte-pair encoding engine over rank tables (tiktoken algorithm).
|
||||||
|
//!
|
||||||
|
//! A [`RankTable`] maps token byte sequences to ranks; merge priority is
|
||||||
|
//! rank order, so no merges list exists. Tables parse from the UTOK1
|
||||||
|
//! container (see `data/families.json` for the format) after zstd
|
||||||
|
//! decompression in [`tables`](crate::utok::tables).
|
||||||
|
//!
|
||||||
|
//! Input is encoding-generic: [`BpeEncoding::count`]/[`encode`]
|
||||||
|
//! (BpeEncoding::encode) take `&[U: Unit]`. Pre-tokenization scans the
|
||||||
|
//! units natively; each piece is then UTF-8-encoded into a reused buffer
|
||||||
|
//! for the byte-keyed rank table (`str` input skips that copy entirely,
|
||||||
|
//! non-UTF-8 flavors narrow ASCII runs 1:1). Steady state performs no
|
||||||
|
//! per-call allocation beyond one scratch buffer for non-UTF-8 flavors.
|
||||||
|
//!
|
||||||
|
//! Per-flavor native rank-table views (`HashMap<Box<[u16]>, u32>` etc.)
|
||||||
|
//! were considered and measured out: with the ASCII narrow path, u16
|
||||||
|
//! input already runs at 81-97% of the str path per codepoint (M4 Max,
|
||||||
|
//! english/CJK), so a second table per flavor (2x memory, plus a
|
||||||
|
//! ragged-token eligibility rule for tokens that split codepoints) buys
|
||||||
|
//! almost nothing. Revisit only with profile evidence.
|
||||||
|
|
||||||
|
use std::{
|
||||||
|
borrow::Cow,
|
||||||
|
collections::HashMap,
|
||||||
|
hash::{BuildHasherDefault, Hasher},
|
||||||
|
};
|
||||||
|
|
||||||
|
use crate::utok::{
|
||||||
|
pretoken::{self, Splitter},
|
||||||
|
utf::Unit,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Firefox/rustc Fx hash: multiplicative word-at-a-time mixing. Rank
|
||||||
|
/// lookups hash short byte keys on every merge step; SipHash is the
|
||||||
|
/// dominant cost there (~30% end-to-end at the default hasher).
|
||||||
|
#[derive(Default)]
|
||||||
|
struct FxHasher(u64);
|
||||||
|
|
||||||
|
impl Hasher for FxHasher {
|
||||||
|
#[inline]
|
||||||
|
fn write(&mut self, bytes: &[u8]) {
|
||||||
|
const SEED: u64 = 0x51_7c_c1_b7_27_22_0a_95;
|
||||||
|
let mut h = self.0;
|
||||||
|
let mut b = bytes;
|
||||||
|
while let Some(chunk) = b.first_chunk::<8>() {
|
||||||
|
h = (h.rotate_left(5) ^ u64::from_le_bytes(*chunk)).wrapping_mul(SEED);
|
||||||
|
b = &b[8..];
|
||||||
|
}
|
||||||
|
if let Some(chunk) = b.first_chunk::<4>() {
|
||||||
|
h = (h.rotate_left(5) ^ u64::from(u32::from_le_bytes(*chunk))).wrapping_mul(SEED);
|
||||||
|
b = &b[4..];
|
||||||
|
}
|
||||||
|
for &byte in b {
|
||||||
|
h = (h.rotate_left(5) ^ u64::from(byte)).wrapping_mul(SEED);
|
||||||
|
}
|
||||||
|
self.0 = h;
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn finish(&self) -> u64 {
|
||||||
|
self.0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type Fx = BuildHasherDefault<FxHasher>;
|
||||||
|
type FxMap = HashMap<Box<[u8]>, u32, Fx>;
|
||||||
|
|
||||||
|
/// Pack a key of ≤15 bytes losslessly into a `u128`: bytes little-endian
|
||||||
|
/// at bits 0..len*8, zero padding, length tag at bits 120..128 (a
|
||||||
|
/// 15-byte key leaves the top byte free, so equal packs imply equal keys
|
||||||
|
/// even across lengths and with NUL bytes). Built from two overlapping
|
||||||
|
/// unaligned reads — a variable-length memcpy here benched slower than
|
||||||
|
/// hashing the raw bytes; the overlap region ORs identical bits.
|
||||||
|
#[inline]
|
||||||
|
fn pack(key: &[u8]) -> Option<u128> {
|
||||||
|
let n = key.len();
|
||||||
|
if n > 15 {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let v: u128 = if let (Some(lo), Some(hi)) = (key.first_chunk::<8>(), key.last_chunk::<8>()) {
|
||||||
|
u128::from(u64::from_le_bytes(*lo)) | u128::from(u64::from_le_bytes(*hi)) << ((n - 8) * 8)
|
||||||
|
} else if let (Some(lo), Some(hi)) = (key.first_chunk::<4>(), key.last_chunk::<4>()) {
|
||||||
|
u128::from(u32::from_le_bytes(*lo)) | u128::from(u32::from_le_bytes(*hi)) << ((n - 4) * 8)
|
||||||
|
} else if let (Some(lo), Some(hi)) = (key.first_chunk::<2>(), key.last_chunk::<2>()) {
|
||||||
|
u128::from(u16::from_le_bytes(*lo)) | u128::from(u16::from_le_bytes(*hi)) << ((n - 2) * 8)
|
||||||
|
} else if let [b] = key {
|
||||||
|
u128::from(*b)
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
Some(v | (n as u128) << 120)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Token bytes → rank map decoded from a UTOK1 blob.
|
||||||
|
///
|
||||||
|
/// Split by key length into three stores, matching the merge loop's
|
||||||
|
/// query mix (measured on o200k, M4 Max: +18% english / +34% code /
|
||||||
|
/// +14% cjk end-to-end vs a single `FxMap<Box<[u8]>, u32>`):
|
||||||
|
///
|
||||||
|
/// - 2 bytes — direct-indexed table: the merge seed loop queries every adjacent
|
||||||
|
/// byte pair, so over half of all lookups land here as one array load.
|
||||||
|
/// - other ≤15 bytes — [`pack`]ed `u128` keys in an Fx map: KV inline in the
|
||||||
|
/// table, no `Box` pointer chase, no byte-wise compare.
|
||||||
|
/// - >15 bytes — plain byte-keyed Fx map (~3% of vocab; spans this long are
|
||||||
|
/// almost always misses).
|
||||||
|
pub struct RankTable {
|
||||||
|
/// Rank of 2-byte token `[a, b]` at `a << 8 | b`; `u32::MAX` where
|
||||||
|
/// absent (ranks are vocab indices, far below the sentinel).
|
||||||
|
pairs: Box<[u32; 65536]>,
|
||||||
|
/// Tokens of 1 or 3..=15 bytes, keyed by [`pack`].
|
||||||
|
short: HashMap<u128, u32, Fx>,
|
||||||
|
/// Tokens longer than 15 bytes.
|
||||||
|
long: FxMap,
|
||||||
|
/// Longest token in bytes; callers may use it to bound scans.
|
||||||
|
pub max_token_len: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl RankTable {
|
||||||
|
/// Parse a zstd-compressed UTOK1 blob. Panics on malformed data — the
|
||||||
|
/// blobs are compile-time embedded, so corruption is a build error.
|
||||||
|
///
|
||||||
|
/// Zero-length entries are *skipped*: packers emit merge-unreachable
|
||||||
|
/// ("dead") vocab slots as empty strings to keep rank contiguity, and
|
||||||
|
/// those ranks must never be produced.
|
||||||
|
pub fn parse(zst: &[u8]) -> Self {
|
||||||
|
let raw = zstd::decode_all(zst).expect("utoken: zstd decode failed");
|
||||||
|
let mut p = &raw[..];
|
||||||
|
assert_eq!(&p[..6], b"UTOK1\n", "utoken: bad magic");
|
||||||
|
p = &p[6..];
|
||||||
|
let n = u32::from_le_bytes(p[..4].try_into().unwrap()) as usize;
|
||||||
|
p = &p[4..];
|
||||||
|
let mut pairs: Box<[u32; 65536]> =
|
||||||
|
vec![u32::MAX; 65536].into_boxed_slice().try_into().unwrap();
|
||||||
|
let mut short = HashMap::with_capacity_and_hasher(n, Fx::default());
|
||||||
|
let mut long = FxMap::default();
|
||||||
|
let mut max_token_len = 0usize;
|
||||||
|
for rank in 0..n as u32 {
|
||||||
|
let mut len = 0usize;
|
||||||
|
let mut shift = 0;
|
||||||
|
loop {
|
||||||
|
let b = p[0];
|
||||||
|
p = &p[1..];
|
||||||
|
len |= ((b & 0x7f) as usize) << shift;
|
||||||
|
if b < 0x80 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
shift += 7;
|
||||||
|
}
|
||||||
|
if len > 0 {
|
||||||
|
let key = &p[..len];
|
||||||
|
if let [a, b] = key {
|
||||||
|
pairs[usize::from(*a) << 8 | usize::from(*b)] = rank;
|
||||||
|
} else if let Some(k) = pack(key) {
|
||||||
|
short.insert(k, rank);
|
||||||
|
} else {
|
||||||
|
long.insert(key.into(), rank);
|
||||||
|
}
|
||||||
|
max_token_len = max_token_len.max(len);
|
||||||
|
p = &p[len..];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(p.is_empty(), "utoken: trailing bytes in UTOK1 blob");
|
||||||
|
Self { pairs, short, long, max_token_len }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Rank of an exact token byte sequence, if present.
|
||||||
|
#[inline]
|
||||||
|
pub fn rank(&self, piece: &[u8]) -> Option<u32> {
|
||||||
|
if let [a, b] = piece {
|
||||||
|
let r = self.pairs[usize::from(*a) << 8 | usize::from(*b)];
|
||||||
|
return (r != u32::MAX).then_some(r);
|
||||||
|
}
|
||||||
|
match pack(piece) {
|
||||||
|
Some(k) => self.short.get(&k).copied(),
|
||||||
|
None => self.long.get(piece).copied(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Append the BPE token ids of one pre-tokenized piece to `out`.
|
||||||
|
pub fn encode_piece(&self, piece: &[u8], out: &mut Vec<u32>) {
|
||||||
|
if piece.is_empty() {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if let Some(rank) = self.rank(piece) {
|
||||||
|
out.push(rank);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
self.merge(piece, |start, end| {
|
||||||
|
out.push(
|
||||||
|
self
|
||||||
|
.rank(&piece[start..end])
|
||||||
|
.expect("utoken: unreachable merge state"),
|
||||||
|
)
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Token count of one pre-tokenized piece without materializing ids.
|
||||||
|
pub fn count_piece(&self, piece: &[u8]) -> u32 {
|
||||||
|
if piece.is_empty() {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
if self.rank(piece).is_some() {
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
let mut n = 0u32;
|
||||||
|
self.merge(piece, |_, _| n += 1);
|
||||||
|
n
|
||||||
|
}
|
||||||
|
|
||||||
|
/// tiktoken's `byte_pair_merge`: start from single bytes, repeatedly
|
||||||
|
/// merge the adjacent pair with the lowest rank, then emit each final
|
||||||
|
/// span via `emit(start, end)`.
|
||||||
|
fn merge(&self, piece: &[u8], mut emit: impl FnMut(usize, usize)) {
|
||||||
|
// parts[k] = (start offset, rank of merging part k with part k+1).
|
||||||
|
// Two sentinels keep `parts[i + 3].0` in-bounds when recomputing
|
||||||
|
// the rank of the pair formed after a merge at the end.
|
||||||
|
let mut parts: Vec<(usize, u32)> = Vec::with_capacity(piece.len() + 1);
|
||||||
|
let mut min_rank: (u32, usize) = (u32::MAX, usize::MAX);
|
||||||
|
for i in 0..piece.len() - 1 {
|
||||||
|
let rank = self.rank(&piece[i..i + 2]).unwrap_or(u32::MAX);
|
||||||
|
if rank < min_rank.0 {
|
||||||
|
min_rank = (rank, i);
|
||||||
|
}
|
||||||
|
parts.push((i, rank));
|
||||||
|
}
|
||||||
|
parts.push((piece.len() - 1, u32::MAX));
|
||||||
|
parts.push((piece.len(), u32::MAX));
|
||||||
|
|
||||||
|
// Rank of merging part `k` with part `k+1` once parts `i` and
|
||||||
|
// `i+1` have conceptually fused (called before the `remove`, so
|
||||||
|
// the fused pair spans parts[k].0 .. parts[k + 3].0).
|
||||||
|
let get_rank = |parts: &[(usize, u32)], k: usize| -> u32 {
|
||||||
|
if k + 3 < parts.len() {
|
||||||
|
self
|
||||||
|
.rank(&piece[parts[k].0..parts[k + 3].0])
|
||||||
|
.unwrap_or(u32::MAX)
|
||||||
|
} else {
|
||||||
|
u32::MAX
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
while min_rank.0 != u32::MAX {
|
||||||
|
let i = min_rank.1;
|
||||||
|
if i > 0 {
|
||||||
|
parts[i - 1].1 = get_rank(&parts, i - 1);
|
||||||
|
}
|
||||||
|
parts[i].1 = get_rank(&parts, i);
|
||||||
|
parts.remove(i + 1);
|
||||||
|
|
||||||
|
min_rank = (u32::MAX, usize::MAX);
|
||||||
|
for (k, &(_, rank)) in parts[..parts.len() - 1].iter().enumerate() {
|
||||||
|
if rank < min_rank.0 {
|
||||||
|
min_rank = (rank, k);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for w in parts.windows(2) {
|
||||||
|
emit(w[0].0, w[1].0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A full BPE tokenizer: piece splitter + rank table + family flags.
|
||||||
|
pub struct BpeEncoding {
|
||||||
|
pub table: RankTable,
|
||||||
|
pub splitter: Splitter,
|
||||||
|
/// Apply Unicode NFC to input before splitting (Qwen3).
|
||||||
|
pub nfc: bool,
|
||||||
|
/// HF `ignore_merges`: whole-piece vocab hit bypasses the merge loop
|
||||||
|
/// (GLM-5). The engine already short-circuits whole-piece hits, which
|
||||||
|
/// is proven equivalent for GLM-5 (see GLM tests); flag kept for
|
||||||
|
/// documentation and any future divergence.
|
||||||
|
#[allow(dead_code)]
|
||||||
|
pub ignore_merges: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl BpeEncoding {
|
||||||
|
pub fn count<U: Unit>(&self, units: &[U]) -> u32 {
|
||||||
|
let mut n = 0u32;
|
||||||
|
self.run(units, &mut |t, p| n += t.count_piece(p));
|
||||||
|
n
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn encode<U: Unit>(&self, units: &[U]) -> Vec<u32> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
self.run(units, &mut |t, p| t.encode_piece(p, &mut out));
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Normalize/transcode as required, split, and feed each piece's
|
||||||
|
/// UTF-8 bytes to `f` alongside the rank table.
|
||||||
|
fn run<U: Unit>(&self, units: &[U], f: &mut impl FnMut(&RankTable, &[u8])) {
|
||||||
|
if let Some(bytes) = U::as_utf8(units) {
|
||||||
|
// UTF-8 flavor: valid by construction (`str`/`String` input).
|
||||||
|
if self.nfc
|
||||||
|
&& let Ok(text) = std::str::from_utf8(bytes)
|
||||||
|
&& let Cow::Owned(norm) = pretoken::nfc(text)
|
||||||
|
{
|
||||||
|
return self.scan(norm.as_bytes(), f);
|
||||||
|
}
|
||||||
|
return self.scan(bytes, f);
|
||||||
|
}
|
||||||
|
// Non-UTF-8 flavors: owned UTF-8 needed only when NFC actually has
|
||||||
|
// work to do, or while the family is still on the regex splitter.
|
||||||
|
if (self.nfc && !nfc_quick(units)) || self.splitter.is_regex() {
|
||||||
|
let s = decode_lossy(units);
|
||||||
|
let s = match pretoken::nfc(&s) {
|
||||||
|
Cow::Owned(o) if self.nfc => o,
|
||||||
|
_ => s,
|
||||||
|
};
|
||||||
|
return self.scan(s.as_bytes(), f);
|
||||||
|
}
|
||||||
|
self.scan(units, f)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn scan<U: Unit>(&self, units: &[U], f: &mut impl FnMut(&RankTable, &[u8])) {
|
||||||
|
let mut buf = Vec::new();
|
||||||
|
self
|
||||||
|
.splitter
|
||||||
|
.for_each_piece(units, |piece| f(&self.table, piece_bytes(piece, &mut buf)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// UTF-8 bytes of one piece: identity for `u8`, otherwise re-encoded into
|
||||||
|
/// `buf` (reused across pieces — one allocation per call, amortized nil).
|
||||||
|
fn piece_bytes<'a, U: Unit>(piece: &'a [U], buf: &'a mut Vec<u8>) -> &'a [u8] {
|
||||||
|
if let Some(bytes) = U::as_utf8(piece) {
|
||||||
|
return bytes;
|
||||||
|
}
|
||||||
|
buf.clear();
|
||||||
|
buf.reserve(piece.len());
|
||||||
|
let mut i = 0;
|
||||||
|
while i < piece.len() {
|
||||||
|
// ASCII runs narrow 1:1 without the decode/encode round-trip
|
||||||
|
// (dominant for code/English u16 input, cf. xutf's ASCII kernels;
|
||||||
|
// the trivial loop autovectorizes).
|
||||||
|
match piece[i].ascii() {
|
||||||
|
Some(b) => {
|
||||||
|
buf.push(b);
|
||||||
|
i += 1;
|
||||||
|
},
|
||||||
|
None => {
|
||||||
|
let (c, n) = U::decode(piece, i);
|
||||||
|
i += n;
|
||||||
|
let mut tmp = [0u8; 4];
|
||||||
|
buf.extend_from_slice(c.encode_utf8(&mut tmp).as_bytes());
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Permissive whole-input decode (malformed units → U+FFFD).
|
||||||
|
fn decode_lossy<U: Unit>(units: &[U]) -> String {
|
||||||
|
let mut s = String::with_capacity(units.len());
|
||||||
|
let mut i = 0;
|
||||||
|
while i < units.len() {
|
||||||
|
let (c, n) = U::decode(units, i);
|
||||||
|
i += n;
|
||||||
|
s.push(c);
|
||||||
|
}
|
||||||
|
s
|
||||||
|
}
|
||||||
|
|
||||||
|
/// NFC quick-check over the decoded codepoint stream, allocation-free
|
||||||
|
/// (conservative: `false` means "may need normalization").
|
||||||
|
fn nfc_quick<U: Unit>(units: &[U]) -> bool {
|
||||||
|
struct Cps<'a, U: Unit>(&'a [U], usize);
|
||||||
|
impl<U: Unit> Iterator for Cps<'_, U> {
|
||||||
|
type Item = u32;
|
||||||
|
|
||||||
|
fn next(&mut self) -> Option<u32> {
|
||||||
|
(self.1 < self.0.len()).then(|| {
|
||||||
|
let (c, n) = U::decode(self.0, self.1);
|
||||||
|
self.1 += n;
|
||||||
|
c as u32
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
xutf::is_nfc_codepoints(Cps(units, 0))
|
||||||
|
}
|
||||||
@@ -420,7 +420,7 @@ pub struct VocabCore {
|
|||||||
|
|
||||||
impl VocabCore {
|
impl VocabCore {
|
||||||
/// Parse one front-coded binary vocabulary blob produced by
|
/// Parse one front-coded binary vocabulary blob produced by
|
||||||
/// `packages/natives/scripts/gen-ctok-vocab.ts` (format documented there;
|
/// `crates/pi-natives/tools/gen-ctok-vocab.ts` (format documented there;
|
||||||
/// pieces arrive in the compact marker alphabet, sorted by those bytes).
|
/// pieces arrive in the compact marker alphabet, sorted by those bytes).
|
||||||
/// Pieces stream straight into the automaton builder; nothing is buffered
|
/// Pieces stream straight into the automaton builder; nothing is buffered
|
||||||
/// beyond the front-coding scratch.
|
/// beyond the front-coding scratch.
|
||||||
@@ -4,8 +4,9 @@
|
|||||||
//! by [ctok](https://github.com/sanderland/ctok): the algorithm port and its
|
//! by [ctok](https://github.com/sanderland/ctok): the algorithm port and its
|
||||||
//! optimizations are this repository's; the measured vocabulary *data* is
|
//! optimizations are this repository's; the measured vocabulary *data* is
|
||||||
//! Sander Land's (MIT — see `data/LICENSE.ctok`), pinned at upstream revision
|
//! Sander Land's (MIT — see `data/LICENSE.ctok`), pinned at upstream revision
|
||||||
//! `df3b59b` (v1.0.0) and embedded in the front-coded binary form produced by
|
//! `df3b59b` (v1.0.0), embedded in the front-coded binary form produced by
|
||||||
//! `packages/natives/scripts/gen-ctok-vocab.ts`. The research behind the
|
//! pi-natives' `gen-ctok-vocab.ts` and zstd-compressed by `tools/pack-ctok.ts`
|
||||||
|
//! into `data/ctok_*.bin.zst`. The research behind the
|
||||||
//! model is described in "On the biology of Claude's tokenizer"
|
//! model is described in "On the biology of Claude's tokenizer"
|
||||||
//! (<https://tokencontributions.substack.com/p/on-the-biology-of-claudes-tokenizer>).
|
//! (<https://tokencontributions.substack.com/p/on-the-biology-of-claudes-tokenizer>).
|
||||||
//!
|
//!
|
||||||
@@ -22,9 +23,12 @@
|
|||||||
//! fallback, matching pieces with one Aho-Corasick transition per byte;
|
//! fallback, matching pieces with one Aho-Corasick transition per byte;
|
||||||
//! 4. add the measured message frame.
|
//! 4. add the measured message frame.
|
||||||
//!
|
//!
|
||||||
//! Nothing in the pipeline materializes decoded characters: the stream, the
|
//! Nothing in the pipeline materializes decoded characters for valid UTF-8
|
||||||
//! vocabulary and the tiling are all byte-level, and Unicode tables are read
|
//! input: the stream, the vocabulary and the tiling are all byte-level, and
|
||||||
//! only for the non-ASCII, non-ideograph characters whose class needs them.
|
//! Unicode tables are read only for the non-ASCII, non-ideograph characters
|
||||||
|
//! whose class needs them. UTF-16/UTF-32 (and malformed UTF-8) input decodes
|
||||||
|
//! permissively into the normalization stream (utf.rs semantics), so valid
|
||||||
|
//! text counts flavor-invariantly.
|
||||||
//!
|
//!
|
||||||
//! Exactness inherited from upstream: 0 mismatches on ~3.4 M recorded
|
//! Exactness inherited from upstream: 0 mismatches on ~3.4 M recorded
|
||||||
//! `count_tokens` responses across the v3 and v4.7 corpora. The port is
|
//! `count_tokens` responses across the v3 and v4.7 corpora. The port is
|
||||||
@@ -37,7 +41,9 @@ mod normalize;
|
|||||||
use std::sync::LazyLock;
|
use std::sync::LazyLock;
|
||||||
|
|
||||||
use engine::VocabCore;
|
use engine::VocabCore;
|
||||||
use normalize::{FrameParams, nfc, raw_head_space, stream_norm};
|
use normalize::{FrameParams, nfc_units, raw_head_space_units, stream_norm, trim_end_ws};
|
||||||
|
|
||||||
|
use crate::utok::utf::Unit;
|
||||||
|
|
||||||
/// One reconstructed tokenizer generation.
|
/// One reconstructed tokenizer generation.
|
||||||
///
|
///
|
||||||
@@ -66,11 +72,17 @@ pub enum Family {
|
|||||||
V5Sonnet,
|
V5Sonnet,
|
||||||
}
|
}
|
||||||
|
|
||||||
static CORE_V3: LazyLock<VocabCore> =
|
static CORE_V3: LazyLock<VocabCore> = LazyLock::new(|| {
|
||||||
LazyLock::new(|| VocabCore::parse(include_bytes!("data/ctok_v3.bin")));
|
let raw = zstd::decode_all(&include_bytes!("../../../data/ctok_v3.bin.zst")[..])
|
||||||
|
.expect("utoken: ctok v3 zstd decode failed");
|
||||||
|
VocabCore::parse(&raw)
|
||||||
|
});
|
||||||
|
|
||||||
static CORE_V47: LazyLock<VocabCore> =
|
static CORE_V47: LazyLock<VocabCore> = LazyLock::new(|| {
|
||||||
LazyLock::new(|| VocabCore::parse(include_bytes!("data/ctok_v4_7.bin")));
|
let raw = zstd::decode_all(&include_bytes!("../../../data/ctok_v4_7.bin.zst")[..])
|
||||||
|
.expect("utoken: ctok v4.7 zstd decode failed");
|
||||||
|
VocabCore::parse(&raw)
|
||||||
|
});
|
||||||
|
|
||||||
impl Family {
|
impl Family {
|
||||||
fn core(self) -> &'static VocabCore {
|
fn core(self) -> &'static VocabCore {
|
||||||
@@ -107,18 +119,21 @@ impl Family {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Token count of `text` as message *content*: the min-cost tiling of the
|
/// Token count of `units` (any UTF flavor) as message *content*: the
|
||||||
/// marked stream, without the fixed per-message frame. This is the right
|
/// min-cost tiling of the marked stream, without the fixed per-message
|
||||||
/// quantity for budget estimates that sum fragments.
|
/// frame. This is the right quantity for budget estimates that sum
|
||||||
pub fn content_token_count(text: &str, family: Family) -> u32 {
|
/// fragments. Valid text counts flavor-invariantly; malformed units decode
|
||||||
|
/// permissively as U+FFFD (utf.rs semantics).
|
||||||
|
pub fn content_token_count<U: Unit>(units: &[U], family: Family) -> u32 {
|
||||||
let core = family.core();
|
let core = family.core();
|
||||||
let p = family.params();
|
let p = family.params();
|
||||||
|
let head_space = raw_head_space_units(units);
|
||||||
if p.ladder {
|
if p.ladder {
|
||||||
let norm = nfc(text, p.fold_quotes);
|
let norm = nfc_units(units, p.fold_quotes);
|
||||||
// The frame appends newline(s) and one token can span into them: read
|
// The frame appends newline(s) and one token can span into them: read
|
||||||
// the content-final newline run before `stream_norm` strips it.
|
// the content-final newline run before `stream_norm` strips it.
|
||||||
let n_tail = norm.bytes().rev().take_while(|&b| b == b'\n').count();
|
let n_tail = norm.bytes().rev().take_while(|&b| b == b'\n').count();
|
||||||
let stream = stream_norm(&norm, &p, raw_head_space(text));
|
let stream = stream_norm(&norm, &p, head_space);
|
||||||
let tail = core.ladder_tail_cost(n_tail, family.appended_newlines());
|
let tail = core.ladder_tail_cost(n_tail, family.appended_newlines());
|
||||||
if stream.is_empty() {
|
if stream.is_empty() {
|
||||||
return tail;
|
return tail;
|
||||||
@@ -127,9 +142,9 @@ pub fn content_token_count(text: &str, family: Family) -> u32 {
|
|||||||
} else {
|
} else {
|
||||||
// The v5 frame absorbs raw ASCII whitespace, so strip before NFC:
|
// The v5 frame absorbs raw ASCII whitespace, so strip before NFC:
|
||||||
// NFC folds NBSP etc. to U+0020, and those are not free at the end.
|
// NFC folds NBSP etc. to U+0020, and those are not free at the end.
|
||||||
let stripped = text.trim_end_matches([' ', '\t', '\n', '\r', '\u{0b}', '\u{0c}']);
|
let stripped = trim_end_ws(units);
|
||||||
let norm = nfc(stripped, p.fold_quotes);
|
let norm = nfc_units(stripped, p.fold_quotes);
|
||||||
let stream = stream_norm(&norm, &p, raw_head_space(text));
|
let stream = stream_norm(&norm, &p, head_space);
|
||||||
if stream.is_empty() {
|
if stream.is_empty() {
|
||||||
0
|
0
|
||||||
} else {
|
} else {
|
||||||
@@ -138,10 +153,12 @@ pub fn content_token_count(text: &str, family: Family) -> u32 {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Reconstructed `count_tokens` value for `text` as a single user message:
|
/// Reconstructed `count_tokens` value for `units` as a single user message:
|
||||||
/// content tiling plus the measured message frame (ctok's `token_count`).
|
/// content tiling plus the measured message frame (ctok's `token_count`).
|
||||||
pub fn message_token_count(text: &str, family: Family) -> u32 {
|
// Exercised by the fixture tests; `lib.rs` only routes content counts.
|
||||||
content_token_count(text, family) + family.params().message_overhead
|
#[cfg_attr(not(test), allow(dead_code))]
|
||||||
|
pub fn message_token_count<U: Unit>(units: &[U], family: Family) -> u32 {
|
||||||
|
content_token_count(units, family) + family.params().message_overhead
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -170,7 +187,7 @@ mod tests {
|
|||||||
let mut checked = 0usize;
|
let mut checked = 0usize;
|
||||||
for f in fixtures() {
|
for f in fixtures() {
|
||||||
for (family, want) in [(Family::V3, f.v3), (Family::V47, f.v4_7), (Family::V5, f.v5)] {
|
for (family, want) in [(Family::V3, f.v3), (Family::V47, f.v4_7), (Family::V5, f.v5)] {
|
||||||
let got = message_token_count(&f.text, family);
|
let got = message_token_count(f.text.as_bytes(), family);
|
||||||
assert_eq!(got, want, "family {family:?} text {:?}", f.text);
|
assert_eq!(got, want, "family {family:?} text {:?}", f.text);
|
||||||
checked += 1;
|
checked += 1;
|
||||||
}
|
}
|
||||||
@@ -194,7 +211,7 @@ mod tests {
|
|||||||
assert!(rows.len() >= 50, "live corpus unexpectedly small: {}", rows.len());
|
assert!(rows.len() >= 50, "live corpus unexpectedly small: {}", rows.len());
|
||||||
for row in rows {
|
for row in rows {
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
message_token_count(&row.text, Family::V5Sonnet),
|
message_token_count(row.text.as_bytes(), Family::V5Sonnet),
|
||||||
row.count,
|
row.count,
|
||||||
"text {:?}",
|
"text {:?}",
|
||||||
row.text
|
row.text
|
||||||
@@ -206,14 +223,14 @@ mod tests {
|
|||||||
fn content_count_is_message_minus_frame() {
|
fn content_count_is_message_minus_frame() {
|
||||||
// The public split every consumer relies on: summing fragments must
|
// The public split every consumer relies on: summing fragments must
|
||||||
// never include per-message frame overhead.
|
// never include per-message frame overhead.
|
||||||
assert_eq!(content_token_count("", Family::V5), 0);
|
assert_eq!(content_token_count("".as_bytes(), Family::V5), 0);
|
||||||
for (family, overhead) in
|
for (family, overhead) in
|
||||||
[(Family::V3, 7), (Family::V47, 11), (Family::V5, 6), (Family::V5Sonnet, 6)]
|
[(Family::V3, 7), (Family::V47, 11), (Family::V5, 6), (Family::V5Sonnet, 6)]
|
||||||
{
|
{
|
||||||
let text = "hello, world";
|
let text = "hello, world";
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
message_token_count(text, family),
|
message_token_count(text.as_bytes(), family),
|
||||||
content_token_count(text, family) + overhead,
|
content_token_count(text.as_bytes(), family) + overhead,
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -242,11 +259,38 @@ mod tests {
|
|||||||
let families = [Family::V3, Family::V47, Family::V5, Family::V5Sonnet];
|
let families = [Family::V3, Family::V47, Family::V5, Family::V5Sonnet];
|
||||||
for (family, &expected) in families.into_iter().zip(want) {
|
for (family, &expected) in families.into_iter().zip(want) {
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
message_token_count(text, family),
|
message_token_count(text.as_bytes(), family),
|
||||||
expected,
|
expected,
|
||||||
"family {family:?} text {text:?}"
|
"family {family:?} text {text:?}"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn utf16_and_utf32_flavor_parity() {
|
||||||
|
// Valid text must count identically in every input flavor: the
|
||||||
|
// UTF-16/UTF-32 paths decode into the same normalization stream the
|
||||||
|
// &str/u8 path sees.
|
||||||
|
let families = [Family::V3, Family::V47, Family::V5, Family::V5Sonnet];
|
||||||
|
for f in fixtures() {
|
||||||
|
let u16s: Vec<u16> = f.text.encode_utf16().collect();
|
||||||
|
let u32s: Vec<u32> = f.text.chars().map(u32::from).collect();
|
||||||
|
for family in families {
|
||||||
|
let want = content_token_count(f.text.as_bytes(), family);
|
||||||
|
assert_eq!(
|
||||||
|
content_token_count(&u16s, family),
|
||||||
|
want,
|
||||||
|
"utf16 family {family:?} text {:?}",
|
||||||
|
f.text
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
content_token_count(&u32s, family),
|
||||||
|
want,
|
||||||
|
"utf32 family {family:?} text {:?}",
|
||||||
|
f.text
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
+89
-10
@@ -19,14 +19,17 @@
|
|||||||
|
|
||||||
use std::borrow::Cow;
|
use std::borrow::Cow;
|
||||||
|
|
||||||
use unicode_normalization::{UnicodeNormalization, char::canonical_combining_class};
|
use xutf::{
|
||||||
use unicode_properties::{GeneralCategory, GeneralCategoryGroup, UnicodeGeneralCategory};
|
GeneralCategory, GeneralCategoryGroup, IntoUnicodeNormalized, ToUnicodeNormalized, Ucd,
|
||||||
|
canonical_combining_class, is_nfc, is_nfc_codepoints,
|
||||||
|
};
|
||||||
|
|
||||||
use super::constants::{
|
use super::constants::{
|
||||||
BOW, CAPS, EOW, NON_SEPARATOR, SHIFT, fold_quote, in_separator_ranges, is_contraction_suffix,
|
BOW, CAPS, EOW, NON_SEPARATOR, SHIFT, fold_quote, in_separator_ranges, is_contraction_suffix,
|
||||||
is_funny_space, is_punct_sym, is_stripped_control, is_stripped_private, is_symbol_letter,
|
is_funny_space, is_punct_sym, is_stripped_control, is_stripped_private, is_symbol_letter,
|
||||||
is_variation_selector,
|
is_variation_selector,
|
||||||
};
|
};
|
||||||
|
use crate::utok::utf::Unit;
|
||||||
|
|
||||||
/// The stream class of one codepoint.
|
/// The stream class of one codepoint.
|
||||||
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
|
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
|
||||||
@@ -150,8 +153,20 @@ pub fn nfc(text: &str, fold_quotes: bool) -> Cow<'_, str> {
|
|||||||
}
|
}
|
||||||
return Cow::Owned(out);
|
return Cow::Owned(out);
|
||||||
}
|
}
|
||||||
let mut out = String::with_capacity(text.len());
|
if is_nfc(text) {
|
||||||
for c in text.chars().nfc() {
|
// Quick-check: composing is a no-op, fold straight off the input.
|
||||||
|
return Cow::Owned(fold_chars(text.chars(), fold_quotes, text.len()));
|
||||||
|
}
|
||||||
|
let composed = text.to_nfc();
|
||||||
|
let folded = fold_chars(composed.chars(), fold_quotes, composed.len());
|
||||||
|
Cow::Owned(folded)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The post-NFC folding loop over an already-composed codepoint stream:
|
||||||
|
/// every fold [`nfc`] documents. `cap` is a capacity hint in bytes.
|
||||||
|
fn fold_chars(chars: impl Iterator<Item = char>, fold_quotes: bool, cap: usize) -> String {
|
||||||
|
let mut out = String::with_capacity(cap);
|
||||||
|
for c in chars {
|
||||||
if is_stripped_control(c) {
|
if is_stripped_control(c) {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -169,7 +184,50 @@ pub fn nfc(text: &str, fold_quotes: bool) -> Cow<'_, str> {
|
|||||||
let c = if fold_quotes { fold_quote(c) } else { c };
|
let c = if fold_quotes { fold_quote(c) } else { c };
|
||||||
out.push(if is_funny_space(c) { ' ' } else { c });
|
out.push(if is_funny_space(c) { ' ' } else { c });
|
||||||
}
|
}
|
||||||
Cow::Owned(out)
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `units` reinterpreted as `&str` when the flavor is UTF-8 and the bytes are
|
||||||
|
/// valid — the common case, which keeps [`nfc`]'s borrowed fast path.
|
||||||
|
fn as_str<U: Unit>(units: &[U]) -> Option<&str> {
|
||||||
|
std::str::from_utf8(U::as_utf8(units)?).ok()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Codepoints decoded permissively off raw units (utf.rs semantics: malformed
|
||||||
|
/// sequences and lone surrogates yield U+FFFD, consuming minimally).
|
||||||
|
struct UnitChars<'a, U: Unit> {
|
||||||
|
units: &'a [U],
|
||||||
|
pos: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<U: Unit> Iterator for UnitChars<'_, U> {
|
||||||
|
type Item = char;
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn next(&mut self) -> Option<char> {
|
||||||
|
if self.pos >= self.units.len() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let (c, n) = U::decode(self.units, self.pos);
|
||||||
|
self.pos += n;
|
||||||
|
Some(c)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`nfc`] over any input flavor. Valid UTF-8 keeps the borrowed fast path;
|
||||||
|
/// UTF-16/UTF-32 (and malformed UTF-8) decode permissively straight into the
|
||||||
|
/// folding loop when the stream is already NFC (the common case, one pass);
|
||||||
|
/// only NFC-dirty input pays a materialize-and-compose round.
|
||||||
|
pub fn nfc_units<U: Unit>(units: &[U], fold_quotes: bool) -> Cow<'_, str> {
|
||||||
|
if let Some(text) = as_str(units) {
|
||||||
|
return nfc(text, fold_quotes);
|
||||||
|
}
|
||||||
|
if is_nfc_codepoints(UnitChars { units, pos: 0 }.map(u32::from)) {
|
||||||
|
return Cow::Owned(fold_chars(UnitChars { units, pos: 0 }, fold_quotes, units.len()));
|
||||||
|
}
|
||||||
|
let composed: String = UnitChars { units, pos: 0 }.collect::<String>().into_nfc();
|
||||||
|
let folded = fold_chars(composed.chars(), fold_quotes, composed.len());
|
||||||
|
Cow::Owned(folded)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether a codepoint uses the isolated character path for letters.
|
/// Whether a codepoint uses the isolated character path for letters.
|
||||||
@@ -196,7 +254,7 @@ pub fn is_separator(c: char) -> bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn is_separator_general(c: char) -> bool {
|
fn is_separator_general(c: char) -> bool {
|
||||||
(canonical_combining_class(c) == 9 && c != NON_SEPARATOR) || in_separator_ranges(c as u32)
|
(canonical_combining_class(c as u32) == 9 && c != NON_SEPARATOR) || in_separator_ranges(c as u32)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Stream class of every ASCII codepoint: the fast path of [`classify`], since
|
/// Stream class of every ASCII codepoint: the fast path of [`classify`], since
|
||||||
@@ -304,7 +362,7 @@ fn is_stray_mark_general(c: char) -> bool {
|
|||||||
if is_syriac_vowel(c) {
|
if is_syriac_vowel(c) {
|
||||||
return false; // a baseless Syriac vowel is a word-forming letter instead
|
return false; // a baseless Syriac vowel is a word-forming letter instead
|
||||||
}
|
}
|
||||||
canonical_combining_class(c) != 0 && !is_separator(c)
|
canonical_combining_class(c as u32) != 0 && !is_separator(c)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether a digit receives a border marker: ASCII digits take none, every
|
/// Whether a digit receives a border marker: ASCII digits take none, every
|
||||||
@@ -661,10 +719,31 @@ fn contraction_seam(s: &str, runs: &[Run], i: usize) -> bool {
|
|||||||
i < 2 || !takes_right_border(s, &runs[i - 2])
|
i < 2 || !takes_right_border(s, &runs[i - 2])
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Whether raw (pre-normalization) text supplies the leading space the frame
|
/// Whether raw (pre-normalization) units supply the leading space the frame
|
||||||
/// absorbs. A space a fold produced or exposed is not absorbed.
|
/// absorbs. A space a fold produced or exposed is not absorbed.
|
||||||
pub fn raw_head_space(text: &str) -> bool {
|
pub fn raw_head_space_units<U: Unit>(units: &[U]) -> bool {
|
||||||
text.starts_with(' ')
|
!units.is_empty() && U::decode(units, 0).0 == ' '
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `units` with the trailing ASCII-whitespace run removed (the v5 frame
|
||||||
|
/// absorbs it raw, before normalization). An ASCII-valued unit is a
|
||||||
|
/// standalone character in every flavor — never a UTF-8 continuation byte or
|
||||||
|
/// half a surrogate pair — so suffix trimming equals trimming decoded text.
|
||||||
|
pub fn trim_end_ws<U: Unit>(units: &[U]) -> &[U] {
|
||||||
|
let Some(&last) = units.last() else {
|
||||||
|
return units;
|
||||||
|
};
|
||||||
|
let mut buf = [last; 4];
|
||||||
|
let ws: [U; 6] = [' ', '\t', '\n', '\r', '\u{0b}', '\u{0c}'].map(|c| {
|
||||||
|
let n = U::encode(c, &mut buf);
|
||||||
|
debug_assert_eq!(n, 1, "ASCII must encode as one unit");
|
||||||
|
buf[0]
|
||||||
|
});
|
||||||
|
let mut end = units.len();
|
||||||
|
while end > 0 && ws.contains(&units[end - 1]) {
|
||||||
|
end -= 1;
|
||||||
|
}
|
||||||
|
&units[..end]
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Byte offset where the character before `at` starts (`at` is a character
|
/// Byte offset where the character before `at` starts (`at` is a character
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
//! Universal offline tokenizer.
|
||||||
|
//!
|
||||||
|
//! Exact token counting (and, where meaningful, encoding) for six model
|
||||||
|
//! families, with all vocabulary data zstd-compressed and embedded in the
|
||||||
|
//! binary. No network, no files, no external tokenizer runtimes.
|
||||||
|
//!
|
||||||
|
//! | [`Encoding`] | family | mechanism |
|
||||||
|
//! |---|---|---|
|
||||||
|
//! | `O200kBase` | GPT-4o / o1 / GPT-5+ | byte-level BPE (tiktoken ranks) |
|
||||||
|
//! | `Cl100kBase` | GPT-3.5 / GPT-4 | byte-level BPE |
|
||||||
|
//! | `ClaudeV3` / `ClaudeV47` / `ClaudeV5` / `ClaudeV5Sonnet` | Claude generations | ctok count reconstruction (count-only) |
|
||||||
|
//! | `Qwen3` | Qwen 3.5 / 3.6 / 3.8 (248k vocab) | byte-level BPE + NFC |
|
||||||
|
//! | `DeepSeekV3` | DeepSeek V3 … V4 | byte-level BPE, 3-stage split chain |
|
||||||
|
//! | `KimiK2` | Kimi K2 … K3 | byte-level BPE (tiktoken ranks) |
|
||||||
|
//! | `Glm5` | GLM-5.x (exact), GLM-4.x (near-exact) | byte-level BPE, `ignore_merges` |
|
||||||
|
//!
|
||||||
|
//! Semantics are `encode_ordinary`: plain content, no special tokens, no
|
||||||
|
//! chat-template frame. This matches budget-estimation use where fragments
|
||||||
|
//! are summed.
|
||||||
|
//!
|
||||||
|
//! Input is encoding-generic ([`Utf`]): `&str`/`String` (UTF-8), `&[u16]`
|
||||||
|
//! (UTF-16, e.g. a JS string over napi) and `&[u32]` (UTF-32) tokenize
|
||||||
|
//! natively in their own code units — no UTF-8 transcode, no scratch
|
||||||
|
//! buffer. The pre-tokenizer scans a codepoint [`Cursor`], and rank tables
|
||||||
|
//! lazily expand per-flavor lookup views (`bpe.rs`). Valid text yields
|
||||||
|
//! flavor-invariant ids/counts; malformed units decode permissively as
|
||||||
|
//! U+FFFD (matching a lossy JS crossing).
|
||||||
|
|
||||||
|
mod bpe;
|
||||||
|
mod claude;
|
||||||
|
mod pretoken;
|
||||||
|
mod scan;
|
||||||
|
mod tables;
|
||||||
|
mod utf;
|
||||||
|
pub use self::{
|
||||||
|
bpe::RankTable,
|
||||||
|
utf::{Cursor, Unit, Utf},
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A tokenizer family. Copy-cheap; all state is in lazily-initialized
|
||||||
|
/// process-wide tables (first use pays one zstd decode of ~0.5–1 MB).
|
||||||
|
#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)]
|
||||||
|
pub enum Encoding {
|
||||||
|
/// GPT-4o / o1 / GPT-5 (OpenAI default).
|
||||||
|
O200kBase,
|
||||||
|
/// GPT-3.5 / GPT-4 / older OpenAI.
|
||||||
|
Cl100kBase,
|
||||||
|
/// Claude 3 through Opus 4.6 (and every non-opus Claude < 5).
|
||||||
|
ClaudeV3,
|
||||||
|
/// Claude Opus 4.7–4.9.
|
||||||
|
ClaudeV47,
|
||||||
|
/// Claude Opus 5+.
|
||||||
|
ClaudeV5,
|
||||||
|
/// Claude Sonnet/Fable 5+ (non-opus 5-series frame variant).
|
||||||
|
ClaudeV5Sonnet,
|
||||||
|
/// Qwen 3.5 / 3.6 / 3.8 (248,044-token vocabulary, NFC input).
|
||||||
|
Qwen3,
|
||||||
|
/// DeepSeek V3 / V3.1 / V3.2 / R1 / V4 (identical base BPE).
|
||||||
|
DeepSeekV3,
|
||||||
|
/// Kimi K2 / K2.5 / K3 (163,584-token base vocabulary).
|
||||||
|
KimiK2,
|
||||||
|
/// GLM-5 (154,820-token vocabulary; ID-preserving superset of GLM-4.x).
|
||||||
|
Glm5,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Encoding {
|
||||||
|
/// Exact content token count of `text` (no specials, no message frame),
|
||||||
|
/// in any UTF flavor.
|
||||||
|
pub fn count<T: Utf + ?Sized>(self, text: &T) -> u32 {
|
||||||
|
match self {
|
||||||
|
Self::ClaudeV3 => claude::content_token_count(text.units(), claude::Family::V3),
|
||||||
|
Self::ClaudeV47 => claude::content_token_count(text.units(), claude::Family::V47),
|
||||||
|
Self::ClaudeV5 => claude::content_token_count(text.units(), claude::Family::V5),
|
||||||
|
Self::ClaudeV5Sonnet => {
|
||||||
|
claude::content_token_count(text.units(), claude::Family::V5Sonnet)
|
||||||
|
},
|
||||||
|
_ => tables::bpe_for(self).count(text.units()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Token ids for `text`, in any UTF flavor.
|
||||||
|
///
|
||||||
|
/// `None` for the Claude families: ctok reconstructs counts, not
|
||||||
|
/// boundaries, so no id sequence exists.
|
||||||
|
pub fn encode<T: Utf + ?Sized>(self, text: &T) -> Option<Vec<u32>> {
|
||||||
|
match self {
|
||||||
|
Self::ClaudeV3 | Self::ClaudeV47 | Self::ClaudeV5 | Self::ClaudeV5Sonnet => None,
|
||||||
|
_ => Some(tables::bpe_for(self).encode(text.units())),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
#[path = "tests"]
|
||||||
|
mod tests {
|
||||||
|
#[path = "claude.rs"]
|
||||||
|
mod claude;
|
||||||
|
#[path = "deepseek.rs"]
|
||||||
|
mod deepseek;
|
||||||
|
#[path = "glm.rs"]
|
||||||
|
mod glm;
|
||||||
|
#[path = "kimi.rs"]
|
||||||
|
mod kimi;
|
||||||
|
#[path = "openai.rs"]
|
||||||
|
mod openai;
|
||||||
|
#[path = "qwen.rs"]
|
||||||
|
mod qwen;
|
||||||
|
}
|
||||||
@@ -0,0 +1,145 @@
|
|||||||
|
//! Pre-tokenization: per-family piece splitting.
|
||||||
|
//!
|
||||||
|
//! Families use HF `Split` semantics with `behavior: Isolated` — every
|
||||||
|
//! match becomes its own piece and unmatched gaps survive as pieces too.
|
||||||
|
//! Every family runs a hand-written codepoint scanner ([`crate::utok::scan`])
|
||||||
|
//! natively over any UTF flavor. The regex chain ([`Splitter::Regex`]) is
|
||||||
|
//! test-only: it is the differential oracle the scanners are validated
|
||||||
|
//! against, and fancy-regex is a dev-dependency.
|
||||||
|
|
||||||
|
use crate::utok::{scan, utf::Unit};
|
||||||
|
|
||||||
|
/// A family's piece splitter.
|
||||||
|
pub enum Splitter {
|
||||||
|
/// Compiled regex chain (test-only differential oracle; UTF-8 input).
|
||||||
|
#[cfg(test)]
|
||||||
|
Regex(Vec<fancy_regex::Regex>),
|
||||||
|
/// tiktoken `o200k_base` scanner.
|
||||||
|
O200k,
|
||||||
|
/// tiktoken `cl100k_base` scanner.
|
||||||
|
Cl100k,
|
||||||
|
/// DeepSeek V3..V4 three-stage chain scanner.
|
||||||
|
DeepSeek,
|
||||||
|
/// Kimi K2/K3 scanner.
|
||||||
|
Kimi,
|
||||||
|
/// Qwen3 scanner.
|
||||||
|
Qwen,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Splitter {
|
||||||
|
/// Compile a regex chain oracle. Panics on invalid patterns
|
||||||
|
/// (compile-time constants).
|
||||||
|
#[cfg(test)]
|
||||||
|
pub fn new(patterns: &[&str]) -> Self {
|
||||||
|
Self::Regex(
|
||||||
|
patterns
|
||||||
|
.iter()
|
||||||
|
.map(|p| fancy_regex::Regex::new(p).expect("utoken: invalid split pattern"))
|
||||||
|
.collect(),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether this splitter needs `&str` input (the engine transcodes
|
||||||
|
/// non-UTF-8 flavors before calling in). Always false at runtime; only
|
||||||
|
/// the test-only regex oracle transcodes.
|
||||||
|
pub fn is_regex(&self) -> bool {
|
||||||
|
#[cfg(test)]
|
||||||
|
if matches!(self, Self::Regex(_)) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
false
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Feed every piece of `units` to `f`, in order, covering the input
|
||||||
|
/// exactly. `Regex` requires `U = u8` holding valid UTF-8.
|
||||||
|
pub fn for_each_piece<U: Unit>(&self, units: &[U], mut f: impl FnMut(&[U])) {
|
||||||
|
match self {
|
||||||
|
Self::O200k => scan_loop(units, &mut f, scan::o200k::next_piece),
|
||||||
|
Self::Cl100k => scan_loop(units, &mut f, scan::cl100k::next_piece),
|
||||||
|
Self::DeepSeek => scan::deepseek::for_each_piece(units, &mut f),
|
||||||
|
Self::Kimi => scan_loop(units, &mut f, scan::kimi::next_piece),
|
||||||
|
Self::Qwen => scan_loop(units, &mut f, scan::qwen::next_piece),
|
||||||
|
#[cfg(test)]
|
||||||
|
Self::Regex(_) => {
|
||||||
|
let bytes = U::as_utf8(units).expect("utoken: regex splitter requires UTF-8 input");
|
||||||
|
let text =
|
||||||
|
std::str::from_utf8(bytes).expect("utoken: regex splitter requires valid UTF-8");
|
||||||
|
let base = text.as_ptr() as usize;
|
||||||
|
for piece in self.split(text) {
|
||||||
|
let a = piece.as_ptr() as usize - base;
|
||||||
|
f(&units[a..a + piece.len()]);
|
||||||
|
}
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Split `text` into pieces (any variant; `&str` view). Test-only:
|
||||||
|
/// scanner differentials compare piece lists against the regex oracle.
|
||||||
|
#[cfg(test)]
|
||||||
|
pub fn split<'t>(&self, text: &'t str) -> Vec<&'t str> {
|
||||||
|
match self {
|
||||||
|
#[cfg(test)]
|
||||||
|
Self::Regex(stages) => {
|
||||||
|
let mut pieces = vec![text];
|
||||||
|
for re in stages {
|
||||||
|
let mut next = Vec::with_capacity(pieces.len());
|
||||||
|
for piece in pieces {
|
||||||
|
let mut last = 0usize;
|
||||||
|
for m in re.find_iter(piece) {
|
||||||
|
let m = m.expect("utoken: regex engine error");
|
||||||
|
if m.start() > last {
|
||||||
|
next.push(&piece[last..m.start()]);
|
||||||
|
}
|
||||||
|
if !m.as_str().is_empty() {
|
||||||
|
next.push(m.as_str());
|
||||||
|
}
|
||||||
|
last = m.end();
|
||||||
|
}
|
||||||
|
if last < piece.len() {
|
||||||
|
next.push(&piece[last..]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pieces = next;
|
||||||
|
}
|
||||||
|
pieces
|
||||||
|
},
|
||||||
|
_ => {
|
||||||
|
let mut pieces = Vec::new();
|
||||||
|
// Scanner boundaries are codepoint boundaries, so byte
|
||||||
|
// ranges are valid `str` slices.
|
||||||
|
self.for_each_piece(text.as_bytes(), |p| {
|
||||||
|
let a = p.as_ptr() as usize - text.as_ptr() as usize;
|
||||||
|
pieces.push(&text[a..a + p.len()]);
|
||||||
|
});
|
||||||
|
pieces
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drive a single-stage scanner over `units`.
|
||||||
|
pub(crate) fn scan_loop<U: Unit>(
|
||||||
|
units: &[U],
|
||||||
|
f: &mut impl FnMut(&[U]),
|
||||||
|
next: impl Fn(&[U], usize) -> usize,
|
||||||
|
) {
|
||||||
|
let mut pos = 0;
|
||||||
|
while pos < units.len() {
|
||||||
|
let end = next(units, pos);
|
||||||
|
debug_assert!(pos < end && end <= units.len(), "utoken: scanner must advance within bounds");
|
||||||
|
f(&units[pos..end]);
|
||||||
|
pos = end;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// NFC-normalize (Qwen3 input contract). Borrows when no work is needed:
|
||||||
|
/// ASCII short-circuit (std `is_ascii` is word-vectorized; cf. xutf's SIMD
|
||||||
|
/// ASCII kernels) then the NFC quick-check, so only text that actually
|
||||||
|
/// needs recomposition allocates.
|
||||||
|
pub fn nfc(text: &str) -> std::borrow::Cow<'_, str> {
|
||||||
|
use xutf::ToUnicodeNormalized;
|
||||||
|
if text.is_ascii() || xutf::is_nfc(text) {
|
||||||
|
return std::borrow::Cow::Borrowed(text);
|
||||||
|
}
|
||||||
|
std::borrow::Cow::Owned(text.to_nfc())
|
||||||
|
}
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
//! tiktoken `cl100k_base` split pattern as a codepoint scanner.
|
||||||
|
//!
|
||||||
|
//! Reference (tiktoken-rs 0.7.0 / openai/tiktoken `openai_public`):
|
||||||
|
//! ```text
|
||||||
|
//! (?i:'s|'t|'re|'ve|'m|'ll|'d)
|
||||||
|
//! |[^\r\n\p{L}\p{N}]?\p{L}+
|
||||||
|
//! |\p{N}{1,3}
|
||||||
|
//! | ?[^\s\p{L}\p{N}]+[\r\n]*
|
||||||
|
//! |\s*[\r\n]+
|
||||||
|
//! |\s+(?!\S)
|
||||||
|
//! |\s+
|
||||||
|
//! ```
|
||||||
|
//! Unlike o200k, the contraction is a standalone *leading* alternate
|
||||||
|
//! (words split as `that` + `'s`) and letter runs are plain `\p{L}+`
|
||||||
|
//! (marks excluded, no case split).
|
||||||
|
|
||||||
|
use super::{cls, contraction_end, decode_at, digits_end, punct_end, ws_end};
|
||||||
|
use crate::utok::utf::Unit;
|
||||||
|
|
||||||
|
/// End of the piece starting at `pos` (`pos < units.len()`).
|
||||||
|
pub fn next_piece<U: Unit>(units: &[U], pos: usize) -> usize {
|
||||||
|
let e = contraction_end(units, pos);
|
||||||
|
if e > pos {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
if let Some(e) = word(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
if let Some(e) = digits_end(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
if let Some(e) = punct_end(units, pos, false) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
if let Some(e) = ws_end(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
// Unreachable for real input; isolate one codepoint so the scan
|
||||||
|
// always advances.
|
||||||
|
pos + decode_at(units, pos).map_or(1, |(_, n)| n)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `[^\r\n\p{L}\p{N}]?\p{L}+`. No real backtracking: if the prefix
|
||||||
|
/// codepoint is present it is not a letter, so a failed `\p{L}+` after it
|
||||||
|
/// cannot succeed from the prefix position either.
|
||||||
|
fn word<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let (c, n) = decode_at(units, pos)?;
|
||||||
|
let start = if c != '\r' && c != '\n' && !cls::is_letter(c) && !cls::is_number(c) {
|
||||||
|
pos + n
|
||||||
|
} else {
|
||||||
|
pos
|
||||||
|
};
|
||||||
|
let mut i = start;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !cls::is_letter(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
(i > start).then_some(i)
|
||||||
|
}
|
||||||
@@ -0,0 +1,304 @@
|
|||||||
|
//! DeepSeek V3..V4 pre-tokenizer: hand-written codepoint scanner over `&[U]`.
|
||||||
|
//!
|
||||||
|
//! Faithful port of the three-stage HF `Split(Isolated)` chain (see
|
||||||
|
//! `DEEPSEEK_STAGE_*` in `tables.rs`). Every stage applies to each piece
|
||||||
|
//! produced by the previous one; matches and unmatched gaps both survive
|
||||||
|
//! as pieces:
|
||||||
|
//!
|
||||||
|
//! 1. `\p{N}{1,3}` — after this stage no piece contains a digit outside its own
|
||||||
|
//! 1–3-digit piece.
|
||||||
|
//! 2. `[一-龥-ゟ゠-ヿ]+` — Han U+4E00–9FA5 plus the contiguous
|
||||||
|
//! hiragana/katakana blocks U+3040–30FF.
|
||||||
|
//! 3. The main pattern:
|
||||||
|
//! - `[!"#$%&'()*+,\-./:;<=>?@\[\\\]^_`{|}~][A-Za-z]+` (one ASCII
|
||||||
|
//! punctuation codepoint glued to ASCII letters, e.g. `.NET`, `(foo`)
|
||||||
|
//! - `[^\r\n\p{L}\p{P}\p{S}]?[\p{L}\p{M}]+`
|
||||||
|
//! - ` ?[\p{P}\p{S}]+[\r\n]*`
|
||||||
|
//! - the whitespace trio `\s*[\r\n]+|\s+(?!\S)|\s+` ([`ws_end`])
|
||||||
|
//!
|
||||||
|
//! Stage 3 has no digit alternate, so digit pieces from stage 1 pass
|
||||||
|
//! through as single gaps; stage-2 CJK pieces may re-split (゠ U+30A0 is
|
||||||
|
//! `Pd`, ・ U+30FB is `Po`, ー U+30FC is `Lm`…). Alternates are tried in
|
||||||
|
//! leftmost-first order with greedy backtracking, replicating the
|
||||||
|
//! reference regex; differential-tested against the fancy-regex chain in
|
||||||
|
//! `tests` below.
|
||||||
|
|
||||||
|
use xutf::GeneralCategoryGroup as GCG;
|
||||||
|
|
||||||
|
use super::{cls, decode_at, digits_end, ws_end};
|
||||||
|
use crate::utok::utf::Unit;
|
||||||
|
|
||||||
|
/// Split `units` into pieces of the full three-stage chain, in order,
|
||||||
|
/// covering the input exactly.
|
||||||
|
pub fn for_each_piece<U: Unit>(units: &[U], f: &mut impl FnMut(&[U])) {
|
||||||
|
isolated(units, digits_end, &mut |p1: &[U]| {
|
||||||
|
isolated(p1, cjk_end, &mut |p2: &[U]| {
|
||||||
|
isolated(p2, main_end, f);
|
||||||
|
});
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One `Split(Isolated)` stage: `matcher` returns the end of a match
|
||||||
|
/// starting exactly at the given offset (or `None`); matches become
|
||||||
|
/// pieces, unmatched gaps between them survive as pieces too.
|
||||||
|
fn isolated<U: Unit>(
|
||||||
|
piece: &[U],
|
||||||
|
matcher: impl Fn(&[U], usize) -> Option<usize>,
|
||||||
|
f: &mut impl FnMut(&[U]),
|
||||||
|
) {
|
||||||
|
let mut gap = 0;
|
||||||
|
let mut pos = 0;
|
||||||
|
while pos < piece.len() {
|
||||||
|
match matcher(piece, pos) {
|
||||||
|
Some(end) => {
|
||||||
|
debug_assert!(
|
||||||
|
pos < end && end <= piece.len(),
|
||||||
|
"utoken: stage matcher must advance within bounds"
|
||||||
|
);
|
||||||
|
if pos > gap {
|
||||||
|
f(&piece[gap..pos]);
|
||||||
|
}
|
||||||
|
f(&piece[pos..end]);
|
||||||
|
pos = end;
|
||||||
|
gap = end;
|
||||||
|
},
|
||||||
|
None => pos += U::decode(piece, pos).1,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if gap < piece.len() {
|
||||||
|
f(&piece[gap..]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `[一-龥-ゟ゠-ヿ]`.
|
||||||
|
#[inline]
|
||||||
|
fn is_cjk(c: char) -> bool {
|
||||||
|
matches!(c as u32, 0x3040..=0x30FF | 0x4E00..=0x9FA5)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `[\p{L}\p{M}]`.
|
||||||
|
#[inline]
|
||||||
|
fn is_lm(c: char) -> bool {
|
||||||
|
cls::is_letter(c) || cls::is_mark(c)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `[\p{P}\p{S}]`. Via [`cls::group`], so the 17.0 symbol/punctuation
|
||||||
|
/// additions stay outside the class like the reference engine sees them.
|
||||||
|
#[inline]
|
||||||
|
fn is_ps(c: char) -> bool {
|
||||||
|
matches!(cls::group(c), GCG::Punctuation | GCG::Symbol)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stage 2: `[一-龥-ゟ゠-ヿ]+` at `pos`.
|
||||||
|
fn cjk_end<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let mut i = pos;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !is_cjk(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
(i > pos).then_some(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stage 3 alternates in leftmost-first order at `pos`; `None` on a gap
|
||||||
|
/// codepoint (e.g. digits, which stage 3 has no alternate for).
|
||||||
|
fn main_end<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let (c, n) = decode_at(units, pos)?;
|
||||||
|
|
||||||
|
// `[ascii punct][A-Za-z]+`
|
||||||
|
if c.is_ascii_punctuation() {
|
||||||
|
let mut i = pos + n;
|
||||||
|
while let Some((c2, n2)) = decode_at(units, i) {
|
||||||
|
if !c2.is_ascii_alphabetic() {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n2;
|
||||||
|
}
|
||||||
|
if i > pos + n {
|
||||||
|
return Some(i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// `[^\r\n\p{L}\p{P}\p{S}]?[\p{L}\p{M}]+` — when the first codepoint is
|
||||||
|
// in the run class the greedy optional prefix changes nothing (marks
|
||||||
|
// are in both; the run absorbs them either way).
|
||||||
|
if is_lm(c) {
|
||||||
|
return Some(lm_run_end(units, pos + n));
|
||||||
|
}
|
||||||
|
if c != '\r' && c != '\n' && !is_ps(c) {
|
||||||
|
// Not L (checked above), not P/S, not CR/LF: prefix-eligible.
|
||||||
|
if let Some((c2, n2)) = decode_at(units, pos + n) {
|
||||||
|
if is_lm(c2) {
|
||||||
|
return Some(lm_run_end(units, pos + n + n2));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ` ?[\p{P}\p{S}]+[\r\n]*`
|
||||||
|
if let Some(e) = ps_end(units, pos) {
|
||||||
|
return Some(e);
|
||||||
|
}
|
||||||
|
|
||||||
|
// `\s*[\r\n]+|\s+(?!\S)|\s+`
|
||||||
|
ws_end(units, pos)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Rest of a `[\p{L}\p{M}]+` run whose first codepoint ends at `pos`.
|
||||||
|
fn lm_run_end<U: Unit>(units: &[U], pos: usize) -> usize {
|
||||||
|
let mut i = pos;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !is_lm(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
i
|
||||||
|
}
|
||||||
|
|
||||||
|
/// ` ?[\p{P}\p{S}]+[\r\n]*` at `pos`.
|
||||||
|
fn ps_end<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let mut i = pos;
|
||||||
|
if let Some((' ', n)) = decode_at(units, pos) {
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
let start = i;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !is_ps(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
if i == start {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if c != '\r' && c != '\n' {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
Some(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// Chain pieces as `String`s for cross-flavor comparison.
|
||||||
|
fn scan_pieces<U: Unit>(units: &[U]) -> Vec<String> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
for_each_piece(units, &mut |p: &[U]| {
|
||||||
|
assert!(!p.is_empty(), "empty piece");
|
||||||
|
let mut s = String::new();
|
||||||
|
let mut i = 0;
|
||||||
|
while i < p.len() {
|
||||||
|
let (c, l) = U::decode(p, i);
|
||||||
|
s.push(c);
|
||||||
|
i += l;
|
||||||
|
}
|
||||||
|
out.push(s);
|
||||||
|
});
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
fn regex_pieces(text: &str) -> Vec<String> {
|
||||||
|
crate::utok::pretoken::Splitter::new(&[
|
||||||
|
crate::utok::tables::DEEPSEEK_STAGE_DIGITS,
|
||||||
|
crate::utok::tables::DEEPSEEK_STAGE_CJK,
|
||||||
|
crate::utok::tables::DEEPSEEK_STAGE_MAIN,
|
||||||
|
])
|
||||||
|
.split(text)
|
||||||
|
.into_iter()
|
||||||
|
.map(str::to_owned)
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn check(text: &str) {
|
||||||
|
let want = regex_pieces(text);
|
||||||
|
assert_eq!(scan_pieces(text.as_bytes()), want, "u8 split drift on {text:?}");
|
||||||
|
let u16s: Vec<u16> = text.encode_utf16().collect();
|
||||||
|
assert_eq!(scan_pieces(u16s.as_slice()), want, "u16 split drift on {text:?}");
|
||||||
|
let u32s: Vec<u32> = text.chars().map(|c| c as u32).collect();
|
||||||
|
assert_eq!(scan_pieces(u32s.as_slice()), want, "u32 split drift on {text:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn deepseek_scanner_matches_regex_on_fixtures() {
|
||||||
|
let raw = include_str!("../../../fixtures/deepseek3.json");
|
||||||
|
let v: serde_json::Value = serde_json::from_str(raw).unwrap();
|
||||||
|
for case in v["cases"].as_array().unwrap() {
|
||||||
|
check(case["text"].as_str().unwrap());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn deepseek_scanner_matches_regex_on_edges() {
|
||||||
|
for text in [
|
||||||
|
"",
|
||||||
|
".NET (foo) #include <stdio.h> C++ -O2",
|
||||||
|
".net .NET. ..NET .1 a.b.c", // punct+letters vs punct runs
|
||||||
|
"1234 12345 123456789 ٣٤٥٦ 7890", // digit stage, non-ASCII digits
|
||||||
|
"第123章abc一二三def゠ー・ヿゟ", // CJK block edges, Pd/Lm/Po inside katakana
|
||||||
|
"一2三45六789零",
|
||||||
|
"中文English日本語한국어", // Han vs hangul (hangul is stage-3 letters)
|
||||||
|
"々〇〆 hancount", // U+3005/3007/3006 are OUTSIDE 4E00-9FA5
|
||||||
|
"a\u{301}bc \u{301}abc \u{301}\u{301} x\u{300}", // marks: run + prefix cases
|
||||||
|
" word two ", // space-prefixed words, trailing ws
|
||||||
|
" !! ?x !x ¡Hola! ¿qué?", // space+punct runs vs punct+letter
|
||||||
|
"$100 €50 √2 ≈π", // symbols (S) in punct-run alternate
|
||||||
|
"\r\n \n\r \t\r\n\t", // ws trio backtracking
|
||||||
|
"foo\r\nbar!\r\n\r\n",
|
||||||
|
"(((x))) [a](b){c}",
|
||||||
|
"👍🏽x 👨👩👧👦 𝕏≈∑ 𠀀𠀁a𠀂", // astral: So run, ZWJ (Cf prefix), Ext-B Han outside range
|
||||||
|
"\u{200b}word \u{a0}x \u{3000}中", // Cf/NBSP/ideographic-space prefixes
|
||||||
|
"can't won't it's", // apostrophe: no contraction alternate in deepseek
|
||||||
|
// Unicode-version skew probes: assigned in 17, Cn in 16 (the
|
||||||
|
// pinned unicode-properties =0.1.3 must agree with fancy-regex
|
||||||
|
// here; caught fleet-wide by QwenFamily).
|
||||||
|
"+\u{10953}上 a\u{10953}b \u{10953}\u{10953}",
|
||||||
|
] {
|
||||||
|
check(text);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn deepseek_scanner_matches_regex_on_random_text() {
|
||||||
|
// Deterministic xorshift over a pool biased toward the chain's
|
||||||
|
// trouble spots: digit/CJK/letter boundaries, katakana punctuation,
|
||||||
|
// ASCII punct+letter gluing, marks, whitespace shapes, astral, and
|
||||||
|
// an unassigned-in-16 codepoint (version-skew probe).
|
||||||
|
let pool: Vec<char> = concat!(
|
||||||
|
"abcdefgh XYZ Net include ",
|
||||||
|
"0123456789٠١٢٣๓๔78",
|
||||||
|
"中文漢字一龥ゟ゠ヿー・々〇",
|
||||||
|
"가나다 カナかな éàüñ ΑΒΓαβγ АБВ עבר ",
|
||||||
|
"!?#$%&*()-_=+[]{};:,.<>/\\|\"`~@^'",
|
||||||
|
" \t\r\n\u{a0}\u{2028}\u{3000}\u{200b}",
|
||||||
|
"\u{301}\u{308}\u{5d0}\u{916}\u{1f600}👍🏽𝕏𠀀𐍈€√≈\u{10953}",
|
||||||
|
)
|
||||||
|
.chars()
|
||||||
|
.collect();
|
||||||
|
for seed in [
|
||||||
|
0xdee9_5eec_0000_0000u64,
|
||||||
|
0x9e37_79b9_7f4a_7c15,
|
||||||
|
0x0dd0_c0de_5eed_0001,
|
||||||
|
0xfeed_face_cafe_beef,
|
||||||
|
] {
|
||||||
|
let mut state = seed;
|
||||||
|
let mut rand = move || {
|
||||||
|
state ^= state << 13;
|
||||||
|
state ^= state >> 7;
|
||||||
|
state ^= state << 17;
|
||||||
|
state
|
||||||
|
};
|
||||||
|
for _ in 0..800 {
|
||||||
|
let len = (rand() % 64) as usize;
|
||||||
|
let text: String = (0..len)
|
||||||
|
.map(|_| pool[(rand() as usize) % pool.len()])
|
||||||
|
.collect();
|
||||||
|
check(&text);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,253 @@
|
|||||||
|
//! Kimi K2/K3 pre-tokenizer: hand-written codepoint scanner over `&[U]`.
|
||||||
|
//!
|
||||||
|
//! Faithful port of the 8-alternate `pat_str` from `tokenization_kimi.py`
|
||||||
|
//! (see `KIMI_K2_PATTERN` in `tables.rs`), replicating fancy-regex's
|
||||||
|
//! leftmost-first alternation and greedy-with-backtracking quantifiers:
|
||||||
|
//!
|
||||||
|
//! 1. `[\p{Han}]+`
|
||||||
|
//! 2. `P? A* B+ C?` — P = `[^\r\n\p{L}\p{N}]`, A = upper set minus Han, B =
|
||||||
|
//! lower set minus Han, C = `(?i:'s|'t|'re|'ve|'m|'ll|'d)`
|
||||||
|
//! 3. `P? A+ B* C?`
|
||||||
|
//! 4. `\p{N}{1,3}`
|
||||||
|
//! 5. ` ?[^\s\p{L}\p{N}]+[\r\n]*`
|
||||||
|
//! 6. `\s*[\r\n]+`
|
||||||
|
//! 7. `\s+(?!\S)`
|
||||||
|
//! 8. `\s+`
|
||||||
|
//!
|
||||||
|
//! Alternates 4–8 are the shared helpers ([`digits_end`], [`punct_end`]
|
||||||
|
//! without o200k's `/` tail, [`ws_end`]); only the Han run and the
|
||||||
|
//! Han-excluded letter cores are Kimi-specific. Alternates are tried in
|
||||||
|
//! order at each position; every scalar matches one of them, so no gaps
|
||||||
|
//! arise. Differential-tested against the fancy-regex splitter (the
|
||||||
|
//! reference tiktoken engine) in `tests` below.
|
||||||
|
|
||||||
|
use super::{cls, contraction_end, decode_at, digits_end, punct_end, ws_end};
|
||||||
|
use crate::utok::utf::Unit;
|
||||||
|
|
||||||
|
/// Class A: `[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]`.
|
||||||
|
#[inline]
|
||||||
|
fn in_a(c: char) -> bool {
|
||||||
|
cls::in_upper_set(c) && !cls::is_han(c)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Class B: `[\p{Ll}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]`.
|
||||||
|
#[inline]
|
||||||
|
fn in_b(c: char) -> bool {
|
||||||
|
cls::in_lower_set(c) && !cls::is_han(c)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `A* B+ C?` from `pos`; end unit offset on success.
|
||||||
|
fn core_lower<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
// Greedy A*, remembering the end of the last A-scalar that is also in
|
||||||
|
// B: backtracking scans A* backward, and the first ∈B scalar found is
|
||||||
|
// the last ∈B scalar forward. B's greedy run from there ends where the
|
||||||
|
// A run did (nothing between it and the run end is in B).
|
||||||
|
let mut i = pos;
|
||||||
|
let mut last_b_end = None;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !in_a(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
if in_b(c) {
|
||||||
|
last_b_end = Some(i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// B+ directly after the full A run.
|
||||||
|
let mut e = i;
|
||||||
|
while let Some((c, n)) = decode_at(units, e) {
|
||||||
|
if !in_b(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
e += n;
|
||||||
|
}
|
||||||
|
if e > i {
|
||||||
|
return Some(contraction_end(units, e));
|
||||||
|
}
|
||||||
|
last_b_end.map(|e| contraction_end(units, e))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `A+ B* C?` from `pos`; end unit offset on success. `B*` never forces
|
||||||
|
/// backtracking, so this is two greedy runs.
|
||||||
|
fn core_upper<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let mut i = pos;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !in_a(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
if i == pos {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let mut e = i;
|
||||||
|
while let Some((c, n)) = decode_at(units, e) {
|
||||||
|
if !in_b(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
e += n;
|
||||||
|
}
|
||||||
|
Some(contraction_end(units, e))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// END unit offset of the piece starting at `pos` (`pos < end <= len`).
|
||||||
|
/// The alternates cover every Unicode scalar, so a piece always exists.
|
||||||
|
pub fn next_piece<U: Unit>(units: &[U], pos: usize) -> usize {
|
||||||
|
let (c0, l0) = U::decode(units, pos);
|
||||||
|
|
||||||
|
// 1: [\p{Han}]+
|
||||||
|
if cls::is_han(c0) {
|
||||||
|
let mut i = pos + l0;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !cls::is_han(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
|
||||||
|
// 2/3: the letter cores, each trying the greedy-optional prefix
|
||||||
|
// `[^\r\n\p{L}\p{N}]?` (space, punctuation, symbols, marks, …) before
|
||||||
|
// the bare form — regex backtracking order.
|
||||||
|
let p = c0 != '\r' && c0 != '\n' && !cls::is_letter(c0) && !cls::is_number(c0);
|
||||||
|
if p && let Some(e) = core_lower(units, pos + l0) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
if let Some(e) = core_lower(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
if p && let Some(e) = core_upper(units, pos + l0) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
if let Some(e) = core_upper(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
|
||||||
|
// 4: \p{N}{1,3}
|
||||||
|
if let Some(e) = digits_end(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
// 5: ` ?[^\s\p{L}\p{N}]+[\r\n]*` (no o200k `/` tail)
|
||||||
|
if let Some(e) = punct_end(units, pos, false) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
// 6/7/8: whitespace trio
|
||||||
|
if let Some(e) = ws_end(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Unreachable for assigned scalars; consume one as a defensive gap.
|
||||||
|
pos + l0
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// Split with the scanner over `U` units; pieces come back as decoded
|
||||||
|
/// strings for cross-flavor comparison.
|
||||||
|
fn scan_pieces<U: Unit>(units: &[U]) -> Vec<String> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
let mut pos = 0;
|
||||||
|
while pos < units.len() {
|
||||||
|
let end = next_piece(units, pos);
|
||||||
|
assert!(end > pos && end <= units.len(), "bad piece end {end} at {pos}");
|
||||||
|
let mut s = String::new();
|
||||||
|
let mut i = pos;
|
||||||
|
while i < end {
|
||||||
|
let (c, l) = U::decode(units, i);
|
||||||
|
s.push(c);
|
||||||
|
i += l;
|
||||||
|
}
|
||||||
|
out.push(s);
|
||||||
|
pos = end;
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
fn regex_pieces(text: &str) -> Vec<String> {
|
||||||
|
crate::utok::pretoken::Splitter::new(&[crate::utok::tables::KIMI_K2_PATTERN])
|
||||||
|
.split(text)
|
||||||
|
.into_iter()
|
||||||
|
.map(str::to_owned)
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn check(text: &str) {
|
||||||
|
let want = regex_pieces(text);
|
||||||
|
assert_eq!(scan_pieces(text.as_bytes()), want, "u8 split drift on {text:?}");
|
||||||
|
let u16s: Vec<u16> = text.encode_utf16().collect();
|
||||||
|
assert_eq!(scan_pieces(u16s.as_slice()), want, "u16 split drift on {text:?}");
|
||||||
|
let u32s: Vec<u32> = text.chars().map(|c| c as u32).collect();
|
||||||
|
assert_eq!(scan_pieces(u32s.as_slice()), want, "u32 split drift on {text:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn kimi_scanner_matches_regex_on_fixtures() {
|
||||||
|
let raw = include_str!("../../../fixtures/kimi_k2.json");
|
||||||
|
let v: serde_json::Value = serde_json::from_str(raw).unwrap();
|
||||||
|
for case in v["cases"].as_array().unwrap() {
|
||||||
|
check(case["text"].as_str().unwrap());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn kimi_scanner_matches_regex_on_edges() {
|
||||||
|
for text in [
|
||||||
|
"",
|
||||||
|
"'",
|
||||||
|
"''",
|
||||||
|
"'s",
|
||||||
|
"x'S y'RE z'Ll w'ſ q'ſt", // case-folded contractions incl. U+017F
|
||||||
|
"can't CAN'T Can'T can'tt", // contraction then trailing letters
|
||||||
|
"HELLO Hello hELLO Džungla DŽUNGLA", // titlecase Lt
|
||||||
|
"a\u{301}\u{301}ABC \u{301}abc \u{301}\u{301}", // marks as P/A/B
|
||||||
|
"〇々中文 一二三456 789", // Han Nl/Lm, fullwidth digits
|
||||||
|
"中文English中文 中文english ENGLISH中文",
|
||||||
|
"!HELLO !hello ¡Hola! ¿qué?",
|
||||||
|
" !! !?\r\n\n x",
|
||||||
|
" \t\r\n \n\t ",
|
||||||
|
"\r \n \r\n\r\n\t",
|
||||||
|
"a b c ",
|
||||||
|
" 12 345 6789 12345",
|
||||||
|
"👍🏽x 👨👩👧👦 𝕏≈∑ 𠀀𠀁a𠀂", // astral: emoji, math letters, Ext-B Han
|
||||||
|
"ᵃᵇᶜ modifier ʰʷ letters", // Lm in both A and B
|
||||||
|
"ǰŠŽ æÆœŒ ß ẞ",
|
||||||
|
] {
|
||||||
|
check(text);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn kimi_scanner_matches_regex_on_random_text() {
|
||||||
|
// Deterministic xorshift over a multilingual pool biased toward the
|
||||||
|
// pattern's trouble spots: Han boundaries, case turns, marks,
|
||||||
|
// contractions, digits, whitespace shapes, astral scalars.
|
||||||
|
let pool: Vec<char> = concat!(
|
||||||
|
"abcdefgh XYZ DžDŽdž 中文漢字词语〇々 ",
|
||||||
|
"0123456789٠١٢٣๓๔ 78",
|
||||||
|
"'stremvld ſ ",
|
||||||
|
"!?#$%&*()-_=+[]{};:,.<>/\\|\"`~@^",
|
||||||
|
" \t\r\n\u{a0}\u{2028}\u{3000}",
|
||||||
|
"\u{301}\u{308}\u{4dc}\u{5d0}\u{916}\u{c15}\u{1f600}👍🏽𝕏𠀀𐍈",
|
||||||
|
"éàüñ ΑΒΓαβγ АБВабв 가나다 カナかな",
|
||||||
|
)
|
||||||
|
.chars()
|
||||||
|
.collect();
|
||||||
|
let mut state = 0x9e37_79b9_7f4a_7c15u64;
|
||||||
|
let mut rand = move || {
|
||||||
|
state ^= state << 13;
|
||||||
|
state ^= state >> 7;
|
||||||
|
state ^= state << 17;
|
||||||
|
state
|
||||||
|
};
|
||||||
|
for _ in 0..2000 {
|
||||||
|
let len = (rand() % 64) as usize;
|
||||||
|
let text: String = (0..len)
|
||||||
|
.map(|_| pool[(rand() as usize) % pool.len()])
|
||||||
|
.collect();
|
||||||
|
check(&text);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,282 @@
|
|||||||
|
//! Hand-written per-family codepoint scanners (runtime pre-tokenization).
|
||||||
|
//!
|
||||||
|
//! Contract (see `pretoken`): each family exposes
|
||||||
|
//! `pub fn next_piece<U: Unit>(units: &[U], pos: usize) -> usize` returning
|
||||||
|
//! the END unit offset of the piece starting at `pos` (`pos < end <= len`).
|
||||||
|
//! Scanners run natively over any UTF flavor via [`Unit::decode`] — no
|
||||||
|
//! regex engine, no transcode. Correctness bar: piece boundaries identical
|
||||||
|
//! to the family's reference regex under fancy-regex/`regex`-module
|
||||||
|
//! semantics (leftmost-first alternation, greedy with backtracking);
|
||||||
|
//! fancy-regex remains a dev/test-only differential reference.
|
||||||
|
//!
|
||||||
|
//! Shared pieces: [`cls`] holds the character classes the patterns use,
|
||||||
|
//! and the helpers below implement whole alternates that recur across
|
||||||
|
//! families (contraction suffix, digit runs, punctuation runs, the
|
||||||
|
//! whitespace trio).
|
||||||
|
|
||||||
|
pub mod cl100k;
|
||||||
|
pub mod deepseek;
|
||||||
|
pub mod kimi;
|
||||||
|
pub mod o200k;
|
||||||
|
pub mod qwen;
|
||||||
|
|
||||||
|
use crate::utok::utf::Unit;
|
||||||
|
|
||||||
|
/// Shared character classes. `\s` in the reference regexes is exactly the
|
||||||
|
/// Unicode `White_Space` property, i.e. [`char::is_whitespace`];
|
||||||
|
/// `\p{L}`/`\p{N}`/`\p{M}` are the general-category groups.
|
||||||
|
///
|
||||||
|
/// **UCD generation.** `xutf` defaults to UCD 16.0.0 tables to match the
|
||||||
|
/// reference regex engines (fancy-regex, tiktoken-rs), so lookups call
|
||||||
|
/// `xutf` directly.
|
||||||
|
pub mod cls {
|
||||||
|
use xutf::{GeneralCategory as GC, GeneralCategoryGroup as GCG, Script, Ucd};
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
pub fn category(c: char) -> GC {
|
||||||
|
c.general_category()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
pub fn group(c: char) -> GCG {
|
||||||
|
c.general_category_group()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
pub fn is_letter(c: char) -> bool {
|
||||||
|
if c.is_ascii() {
|
||||||
|
return c.is_ascii_alphabetic();
|
||||||
|
}
|
||||||
|
group(c) == GCG::Letter
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
pub fn is_number(c: char) -> bool {
|
||||||
|
if c.is_ascii() {
|
||||||
|
return c.is_ascii_digit();
|
||||||
|
}
|
||||||
|
group(c) == GCG::Number
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
pub fn is_mark(c: char) -> bool {
|
||||||
|
!c.is_ascii() && group(c) == GCG::Mark
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
pub fn is_ws(c: char) -> bool {
|
||||||
|
c.is_whitespace()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `\p{Han}` = Script=Han (includes e.g. 〇 U+3007 Nl and 々 U+3005 Lm).
|
||||||
|
/// The 17.0 additions include 4321 Han scalars (CJK ext. J, the ext.
|
||||||
|
/// B/F tail fills, and U+16FF2..=U+16FF6) that are `Script=Unknown` for
|
||||||
|
/// the reference engine.
|
||||||
|
#[inline]
|
||||||
|
pub fn is_han(c: char) -> bool {
|
||||||
|
!c.is_ascii() && c.script() == Script::Han
|
||||||
|
}
|
||||||
|
/// `[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]` — tiktoken's "uppercase or
|
||||||
|
/// caseless letter" set. Overlaps [`in_lower_set`] on `Lm`/`Lo`/`M`.
|
||||||
|
#[inline]
|
||||||
|
pub fn in_upper_set(c: char) -> bool {
|
||||||
|
if c.is_ascii() {
|
||||||
|
return c.is_ascii_uppercase();
|
||||||
|
}
|
||||||
|
let cat = category(c);
|
||||||
|
matches!(
|
||||||
|
cat,
|
||||||
|
GC::UppercaseLetter | GC::TitlecaseLetter | GC::ModifierLetter | GC::OtherLetter
|
||||||
|
) || cat.group() == GCG::Mark
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `[\p{Ll}\p{Lm}\p{Lo}\p{M}]` — "lowercase or caseless letter" set.
|
||||||
|
#[inline]
|
||||||
|
pub fn in_lower_set(c: char) -> bool {
|
||||||
|
if c.is_ascii() {
|
||||||
|
return c.is_ascii_lowercase();
|
||||||
|
}
|
||||||
|
let cat = category(c);
|
||||||
|
matches!(cat, GC::LowercaseLetter | GC::ModifierLetter | GC::OtherLetter)
|
||||||
|
|| cat.group() == GCG::Mark
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode the codepoint at `i`, or `None` at end of input.
|
||||||
|
#[inline]
|
||||||
|
pub(crate) fn decode_at<U: Unit>(units: &[U], i: usize) -> Option<(char, usize)> {
|
||||||
|
(i < units.len()).then(|| U::decode(units, i))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `(?i:'s|'t|'re|'ve|'m|'ll|'d)` at `pos`. Returns the end, or `pos` when
|
||||||
|
/// absent (the alternate is used both standalone and as an optional
|
||||||
|
/// suffix). Case-insensitivity is simple case folding, so `'s` also
|
||||||
|
/// matches U+017F ſ — matching the reference regex engines.
|
||||||
|
pub(crate) fn contraction_end<U: Unit>(units: &[U], pos: usize) -> usize {
|
||||||
|
let Some(('\'', n0)) = decode_at(units, pos) else {
|
||||||
|
return pos;
|
||||||
|
};
|
||||||
|
let Some((c1, n1)) = decode_at(units, pos + n0) else {
|
||||||
|
return pos;
|
||||||
|
};
|
||||||
|
let one = pos + n0 + n1;
|
||||||
|
match c1 {
|
||||||
|
's' | 'S' | '\u{17f}' | 't' | 'T' | 'm' | 'M' | 'd' | 'D' => one,
|
||||||
|
'r' | 'R' | 'v' | 'V' => match decode_at(units, one) {
|
||||||
|
Some(('e' | 'E', n2)) => one + n2,
|
||||||
|
_ => pos,
|
||||||
|
},
|
||||||
|
'l' | 'L' => match decode_at(units, one) {
|
||||||
|
Some(('l' | 'L', n2)) => one + n2,
|
||||||
|
_ => pos,
|
||||||
|
},
|
||||||
|
_ => pos,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `\p{N}{1,3}` at `pos`.
|
||||||
|
pub(crate) fn digits_end<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let mut i = pos;
|
||||||
|
for _ in 0..3 {
|
||||||
|
match decode_at(units, i) {
|
||||||
|
Some((c, n)) if cls::is_number(c) => i += n,
|
||||||
|
_ => break,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
(i > pos).then_some(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// ` ?[^\s\p{L}\p{N}]+[\r\n]*` at `pos`; `trailing_slash` adds o200k's `/`
|
||||||
|
/// to the tail class (`[\r\n/]*`).
|
||||||
|
pub(crate) fn punct_end<U: Unit>(units: &[U], pos: usize, trailing_slash: bool) -> Option<usize> {
|
||||||
|
let mut i = pos;
|
||||||
|
if let Some((' ', n)) = decode_at(units, pos) {
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
let start = i;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if cls::is_ws(c) || cls::is_letter(c) || cls::is_number(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
if i == start {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if c == '\r' || c == '\n' || (trailing_slash && c == '/') {
|
||||||
|
i += n;
|
||||||
|
} else {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Some(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The tiktoken whitespace trio `\s*[\r\n]+|\s+(?!\S)|\s+`, in that
|
||||||
|
/// alternation order, at `pos`:
|
||||||
|
/// - a run containing a newline matches through its *last* `\r`/`\n` (trailing
|
||||||
|
/// non-newline whitespace excluded — backtracked `\s*`),
|
||||||
|
/// - otherwise a run at end of input matches whole,
|
||||||
|
/// - otherwise the run gives back one codepoint for the `(?!\S)` lookahead
|
||||||
|
/// (unless it is a single codepoint, which `\s+` takes whole).
|
||||||
|
pub(crate) fn ws_end<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let mut i = pos;
|
||||||
|
let mut last_nl_end = None;
|
||||||
|
let mut last_len = 0;
|
||||||
|
let mut cps = 0usize;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !cls::is_ws(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
if c == '\r' || c == '\n' {
|
||||||
|
last_nl_end = Some(i);
|
||||||
|
}
|
||||||
|
last_len = n;
|
||||||
|
cps += 1;
|
||||||
|
}
|
||||||
|
if cps == 0 {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
if let Some(e) = last_nl_end {
|
||||||
|
return Some(e);
|
||||||
|
}
|
||||||
|
if i == units.len() {
|
||||||
|
return Some(i);
|
||||||
|
}
|
||||||
|
Some(if cps > 1 { i - last_len } else { i })
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use regex::Regex;
|
||||||
|
use xutf::GeneralCategoryGroup as GCG;
|
||||||
|
|
||||||
|
use super::cls;
|
||||||
|
|
||||||
|
/// `xutf` must resolve against UCD 16.0.0 (the default) so the scanners
|
||||||
|
/// match the regex-syntax tables.
|
||||||
|
#[test]
|
||||||
|
fn class_pin_tracks_xutf_ucd_generation() {
|
||||||
|
assert_eq!(
|
||||||
|
xutf::UCD_VERSION,
|
||||||
|
(16, 0, 0),
|
||||||
|
"xutf UCD generation moved: must resolve against UCD 16.0.0"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every scalar the class matches, per the reference regex engine (the
|
||||||
|
/// `regex`/fancy-regex `regex-syntax` tables the differential oracles
|
||||||
|
/// resolve `\p{…}` through).
|
||||||
|
fn reference_members(all: &str, pattern: &str) -> Vec<bool> {
|
||||||
|
let re = Regex::new(pattern).unwrap();
|
||||||
|
let mut set = vec![false; 0x11_0000];
|
||||||
|
for m in re.find_iter(all) {
|
||||||
|
for c in m.as_str().chars() {
|
||||||
|
set[c as usize] = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
set
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A reference `\p{…}` pattern paired with the [`cls`] predicate that
|
||||||
|
/// must reproduce it.
|
||||||
|
type Class = (&'static str, fn(char) -> bool);
|
||||||
|
|
||||||
|
/// Exhaustive scalar sweep: each [`cls`] predicate must agree with its
|
||||||
|
/// reference-regex class on all 0x110000 codepoints. This is what the
|
||||||
|
/// per-family differential tests sample, so it fails first — and it is
|
||||||
|
/// the guard against `xutf`'s UCD generation drifting ahead of the
|
||||||
|
/// engine's (unpinned, the 4803 scalars UCD 17.0.0 added over 16.0.0
|
||||||
|
/// re-split text and change token ids).
|
||||||
|
#[test]
|
||||||
|
fn classes_match_reference_regex_on_every_scalar() {
|
||||||
|
let all: String = (0..0x11_0000u32).filter_map(char::from_u32).collect();
|
||||||
|
let classes: [Class; 8] = [
|
||||||
|
(r"\p{L}", cls::is_letter),
|
||||||
|
(r"\p{N}", cls::is_number),
|
||||||
|
(r"\p{M}", cls::is_mark),
|
||||||
|
(r"\s", cls::is_ws),
|
||||||
|
(r"\p{Han}", cls::is_han),
|
||||||
|
(r"[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]", cls::in_upper_set),
|
||||||
|
(r"[\p{Ll}\p{Lm}\p{Lo}\p{M}]", cls::in_lower_set),
|
||||||
|
(r"[\p{P}\p{S}]", |c| matches!(cls::group(c), GCG::Punctuation | GCG::Symbol)),
|
||||||
|
];
|
||||||
|
for (pattern, ours) in classes {
|
||||||
|
let theirs = reference_members(&all, pattern);
|
||||||
|
let mut bad = Vec::new();
|
||||||
|
for c in (0..0x11_0000u32).filter_map(char::from_u32) {
|
||||||
|
if ours(c) != theirs[c as usize] {
|
||||||
|
bad.push(c as u32);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
bad.is_empty(),
|
||||||
|
"{pattern}: {} scalars diverge, first: {:04X?}",
|
||||||
|
bad.len(),
|
||||||
|
&bad[..bad.len().min(8)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,113 @@
|
|||||||
|
//! tiktoken `o200k_base` split pattern as a codepoint scanner.
|
||||||
|
//!
|
||||||
|
//! Reference (tiktoken-rs 0.7.0 / openai/tiktoken `openai_public`):
|
||||||
|
//! ```text
|
||||||
|
//! [^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]*[\p{Ll}\p{Lm}\p{Lo}\p{M}]+(?i:'s|'t|'re|'ve|'m|'ll|'d)?
|
||||||
|
//! |[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]+[\p{Ll}\p{Lm}\p{Lo}\p{M}]*(?i:'s|'t|'re|'ve|'m|'ll|'d)?
|
||||||
|
//! |\p{N}{1,3}
|
||||||
|
//! | ?[^\s\p{L}\p{N}]+[\r\n/]*
|
||||||
|
//! |\s*[\r\n]+
|
||||||
|
//! |\s+(?!\S)
|
||||||
|
//! |\s+
|
||||||
|
//! ```
|
||||||
|
|
||||||
|
use super::{cls, contraction_end, decode_at, digits_end, punct_end, ws_end};
|
||||||
|
use crate::utok::utf::Unit;
|
||||||
|
|
||||||
|
/// End of the piece starting at `pos` (`pos < units.len()`).
|
||||||
|
pub fn next_piece<U: Unit>(units: &[U], pos: usize) -> usize {
|
||||||
|
if let Some(e) = word(units, pos) {
|
||||||
|
return contraction_end(units, e);
|
||||||
|
}
|
||||||
|
if let Some(e) = digits_end(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
if let Some(e) = punct_end(units, pos, true) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
if let Some(e) = ws_end(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
// Unreachable for real input (the alternates cover every codepoint);
|
||||||
|
// isolate one codepoint so the scan always advances.
|
||||||
|
pos + decode_at(units, pos).map_or(1, |(_, n)| n)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Word alternates 1 and 2, in leftmost-first order, without the
|
||||||
|
/// contraction suffix (shared by both):
|
||||||
|
/// alt1 `[^\r\n\p{L}\p{N}]? upper* lower+`, alt2 `…? upper+ lower*`.
|
||||||
|
/// Each alternate tries its optional one-codepoint prefix first (greedy
|
||||||
|
/// `?`), falling back to prefix-absent before yielding to the next
|
||||||
|
/// alternate.
|
||||||
|
fn word<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let (c, n) = decode_at(units, pos)?;
|
||||||
|
let prefix =
|
||||||
|
(c != '\r' && c != '\n' && !cls::is_letter(c) && !cls::is_number(c)).then_some(pos + n);
|
||||||
|
if let Some(p) = prefix
|
||||||
|
&& let Some(e) = body1(units, p)
|
||||||
|
{
|
||||||
|
return Some(e);
|
||||||
|
}
|
||||||
|
if let Some(e) = body1(units, pos) {
|
||||||
|
return Some(e);
|
||||||
|
}
|
||||||
|
if let Some(p) = prefix
|
||||||
|
&& let Some(e) = body2(units, p)
|
||||||
|
{
|
||||||
|
return Some(e);
|
||||||
|
}
|
||||||
|
body2(units, pos)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `upper* lower+` with greedy backtracking. The classes overlap on
|
||||||
|
/// `Lm`/`Lo`/`M`: when no `lower` codepoint follows the maximal `upper`
|
||||||
|
/// run, `upper*` gives back to the *last* run codepoint that is also in
|
||||||
|
/// the lower set, and `lower+` takes exactly that codepoint (everything
|
||||||
|
/// to its right is `Lu`/`Lt`, outside the lower set).
|
||||||
|
fn body1<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let mut i = pos;
|
||||||
|
let mut last_shared_end = None;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !cls::in_upper_set(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if cls::in_lower_set(c) {
|
||||||
|
last_shared_end = Some(i + n);
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
if let Some((c, n)) = decode_at(units, i)
|
||||||
|
&& cls::in_lower_set(c)
|
||||||
|
{
|
||||||
|
let mut j = i + n;
|
||||||
|
while let Some((c, n)) = decode_at(units, j) {
|
||||||
|
if !cls::in_lower_set(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
j += n;
|
||||||
|
}
|
||||||
|
return Some(j);
|
||||||
|
}
|
||||||
|
last_shared_end
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `upper+ lower*` — never backtracks (both suffix parts may be empty).
|
||||||
|
fn body2<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let mut i = pos;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !cls::in_upper_set(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
if i == pos {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !cls::in_lower_set(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
Some(i)
|
||||||
|
}
|
||||||
@@ -0,0 +1,203 @@
|
|||||||
|
//! Qwen3 (3.5/3.6/3.8) split pattern as a codepoint scanner.
|
||||||
|
//!
|
||||||
|
//! Reference (qwen3.8.tokenizer.json pre_tokenizer, = families.json qwen3):
|
||||||
|
//! ```text
|
||||||
|
//! (?i:'s|'t|'re|'ve|'m|'ll|'d)
|
||||||
|
//! |[^\r\n\p{L}\p{N}]?[\p{L}\p{M}]+
|
||||||
|
//! |\p{N}
|
||||||
|
//! | ?[^\s\p{L}\p{M}\p{N}]+[\r\n]*
|
||||||
|
//! |\s*[\r\n]+
|
||||||
|
//! |\s+(?!\S)
|
||||||
|
//! |\s+
|
||||||
|
//! ```
|
||||||
|
//! Deviations from cl100k: letter runs include marks (`[\p{L}\p{M}]+`)
|
||||||
|
//! and the optional prefix class admits marks (real backtracking: a lone
|
||||||
|
//! mark is both prefix- and run-eligible), `\p{N}` is a SINGLE codepoint
|
||||||
|
//! (not `{1,3}`), and punctuation runs exclude marks.
|
||||||
|
//!
|
||||||
|
//! NFC is not this layer's concern: the engine normalizes before scanning
|
||||||
|
//! (`BpeEncoding` nfc funnel).
|
||||||
|
|
||||||
|
use super::{cls, contraction_end, decode_at, ws_end};
|
||||||
|
use crate::utok::utf::Unit;
|
||||||
|
|
||||||
|
/// End of the piece starting at `pos` (`pos < units.len()`).
|
||||||
|
pub fn next_piece<U: Unit>(units: &[U], pos: usize) -> usize {
|
||||||
|
let e = contraction_end(units, pos);
|
||||||
|
if e > pos {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
if let Some(e) = word(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
let Some((c, n)) = decode_at(units, pos) else {
|
||||||
|
return pos + 1;
|
||||||
|
};
|
||||||
|
if cls::is_number(c) {
|
||||||
|
// `\p{N}`: exactly one codepoint.
|
||||||
|
return pos + n;
|
||||||
|
}
|
||||||
|
if let Some(e) = punct(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
if let Some(e) = ws_end(units, pos) {
|
||||||
|
return e;
|
||||||
|
}
|
||||||
|
// Unreachable for real input (the alternates cover every codepoint);
|
||||||
|
// isolate one codepoint so the scan always advances.
|
||||||
|
pos + n
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `[\p{L}\p{M}]+` run end starting at `pos` (may equal `pos`).
|
||||||
|
#[inline]
|
||||||
|
fn lm_run_end<U: Unit>(units: &[U], pos: usize) -> usize {
|
||||||
|
let mut i = pos;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if !cls::is_letter(c) && !cls::is_mark(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
i
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `[^\r\n\p{L}\p{N}]?[\p{L}\p{M}]+`. The prefix class admits marks, which
|
||||||
|
/// are also run codepoints, so the greedy optional prefix backtracks: when
|
||||||
|
/// nothing follows a mark prefix, the mark itself is the run.
|
||||||
|
fn word<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let (c, n) = decode_at(units, pos)?;
|
||||||
|
if c != '\r' && c != '\n' && !cls::is_letter(c) && !cls::is_number(c) {
|
||||||
|
// Greedy: try with the prefix consumed.
|
||||||
|
let e = lm_run_end(units, pos + n);
|
||||||
|
if e > pos + n {
|
||||||
|
return Some(e);
|
||||||
|
}
|
||||||
|
// Backtrack to no prefix: only a mark is still run-eligible
|
||||||
|
// (letters are excluded from the prefix class).
|
||||||
|
return cls::is_mark(c).then_some(pos + n);
|
||||||
|
}
|
||||||
|
// No prefix possible; run only if `c` itself is a letter.
|
||||||
|
let e = lm_run_end(units, pos);
|
||||||
|
(e > pos).then_some(e)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// ` ?[^\s\p{L}\p{M}\p{N}]+[\r\n]*`. No backtracking needed: the body
|
||||||
|
/// class excludes whitespace, so a failed body after the optional space
|
||||||
|
/// cannot succeed from the space either.
|
||||||
|
fn punct<U: Unit>(units: &[U], pos: usize) -> Option<usize> {
|
||||||
|
let (c, n) = decode_at(units, pos)?;
|
||||||
|
let start = if c == ' ' { pos + n } else { pos };
|
||||||
|
let mut i = start;
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if cls::is_ws(c) || cls::is_letter(c) || cls::is_mark(c) || cls::is_number(c) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
if i == start {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
while let Some((c, n)) = decode_at(units, i) {
|
||||||
|
if c != '\r' && c != '\n' {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
i += n;
|
||||||
|
}
|
||||||
|
Some(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
//! Differential: scanner piece boundaries vs the reference fancy-regex
|
||||||
|
//! splitter, over corpus + torture cases + seeded random strings.
|
||||||
|
|
||||||
|
use crate::utok::{pretoken::Splitter, tables::QWEN3_PATTERN};
|
||||||
|
|
||||||
|
/// splitmix64 (same scheme as tests/openai.rs).
|
||||||
|
struct Rng(u64);
|
||||||
|
|
||||||
|
impl Rng {
|
||||||
|
fn next(&mut self) -> u64 {
|
||||||
|
self.0 = self.0.wrapping_add(0x9e3779b97f4a7c15);
|
||||||
|
let mut z = self.0;
|
||||||
|
z = (z ^ (z >> 30)).wrapping_mul(0xbf58476d1ce4e5b9);
|
||||||
|
z = (z ^ (z >> 27)).wrapping_mul(0x94d049bb133111eb);
|
||||||
|
z ^ (z >> 31)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Qwen-torture: mark prefix/run overlap, single-codepoint numbers,
|
||||||
|
/// mark-free punctuation runs, contraction folds, whitespace trio.
|
||||||
|
const TRICKY: &[&str] = &[
|
||||||
|
"\u{301}", // lone mark: run via prefix backtrack
|
||||||
|
"\u{301}abc", // mark prefix + letter run (one piece)
|
||||||
|
"a\u{301}\u{301}b", // marks ride the run
|
||||||
|
"\u{301}\u{301}", // mark prefix + mark run
|
||||||
|
"!\u{301}x .\u{301}", // prefix vs punct on mark boundary
|
||||||
|
" \u{301}x", // space prefix + mark-led run
|
||||||
|
"१٣456 ½Ⅻ①", // \p{N} single: Devanagari/Arabic/fullwidth/vulgar/roman
|
||||||
|
"'ſ 'S it'\u{17f}", // long-s contraction fold
|
||||||
|
"'Re'VE'lL'd 'r 'v", // two-letter folds and near-misses
|
||||||
|
"。汉字,测试!", // CJK punct prefix on letter runs
|
||||||
|
"x \r\n \n y", // \s*[\r\n]+ through last newline
|
||||||
|
"a\r\nb\rc\nd",
|
||||||
|
" \n\t\n end",
|
||||||
|
"tail ",
|
||||||
|
" x 5 ",
|
||||||
|
"€100 $5.99",
|
||||||
|
"a/b//\n/",
|
||||||
|
"\r",
|
||||||
|
"\n \n",
|
||||||
|
" ",
|
||||||
|
];
|
||||||
|
|
||||||
|
fn random_strings(seed: u64, n: usize) -> Vec<String> {
|
||||||
|
let mut rng = Rng(seed);
|
||||||
|
let mut out = Vec::with_capacity(n * 2);
|
||||||
|
for _ in 0..n {
|
||||||
|
let len = (rng.next() % 300 + 1) as usize;
|
||||||
|
let bytes: Vec<u8> = (0..len).map(|_| rng.next() as u8).collect();
|
||||||
|
out.push(String::from_utf8_lossy(&bytes).into_owned());
|
||||||
|
}
|
||||||
|
for _ in 0..n {
|
||||||
|
let len = (rng.next() % 120 + 1) as usize;
|
||||||
|
let s: String = (0..len)
|
||||||
|
.map(|_| {
|
||||||
|
let c = match rng.next() % 8 {
|
||||||
|
0 => rng.next() % 0x80, // ASCII
|
||||||
|
1 => 0x20 + rng.next() % 4, // spaces/punct
|
||||||
|
2 => rng.next() % 0x250, // Latin+ext
|
||||||
|
3 => 0x4e00 + rng.next() % 0x100, // CJK
|
||||||
|
4 => 0x1f300 + rng.next() % 0x100, // emoji
|
||||||
|
5 => 0x300 + rng.next() % 0x70, // combining marks
|
||||||
|
6 => [9, 10, 13, 32][(rng.next() % 4) as usize], // whitespace
|
||||||
|
_ => rng.next() % 0x11_0000,
|
||||||
|
};
|
||||||
|
char::from_u32(c as u32).unwrap_or('\u{fffd}')
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
out.push(s);
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn qwen_scanner_matches_regex() {
|
||||||
|
let reference = Splitter::new(&[QWEN3_PATTERN]);
|
||||||
|
let scanner = Splitter::Qwen;
|
||||||
|
let corpus: Vec<String> =
|
||||||
|
serde_json::from_str(include_str!("../../../fixtures/corpus.json")).unwrap();
|
||||||
|
let texts: Vec<String> = corpus
|
||||||
|
.into_iter()
|
||||||
|
.chain(TRICKY.iter().map(|s| s.to_string()))
|
||||||
|
.chain(random_strings(0x9e60_03e6, 400))
|
||||||
|
.collect();
|
||||||
|
for text in &texts {
|
||||||
|
assert_eq!(
|
||||||
|
scanner.split(text),
|
||||||
|
reference.split(text),
|
||||||
|
"piece boundaries diverge on {text:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,227 @@
|
|||||||
|
//! Lazily-decoded embedded tables and per-family wiring.
|
||||||
|
//!
|
||||||
|
//! Each family agent: add your `LazyLock`, your pattern constant(s), and
|
||||||
|
//! your arm in [`bpe_for`]. Data blobs live in `data/<name>.bin.zst`
|
||||||
|
//! (UTOK1 + zstd; see `data/families.json`). Patterns come from
|
||||||
|
//! `data/families.json` — do not invent them.
|
||||||
|
|
||||||
|
use std::sync::LazyLock;
|
||||||
|
|
||||||
|
use crate::utok::{
|
||||||
|
Encoding,
|
||||||
|
bpe::{BpeEncoding, RankTable},
|
||||||
|
pretoken::Splitter,
|
||||||
|
};
|
||||||
|
|
||||||
|
// ── OpenAI ──────────────────────────────────────────────────────────────
|
||||||
|
// Codepoint scanners in src/scan/{o200k,cl100k}.rs, hand-ported from the
|
||||||
|
// tiktoken-rs 0.7.0 patterns (src/tiktoken_ext/openai_public.rs, identical
|
||||||
|
// to openai/tiktoken tiktoken_ext/openai_public.py) and differential-tested
|
||||||
|
// against tiktoken-rs in tests/openai.rs.
|
||||||
|
|
||||||
|
static O200K_BASE: LazyLock<BpeEncoding> = LazyLock::new(|| BpeEncoding {
|
||||||
|
table: RankTable::parse(include_bytes!("../../data/o200k_base.bin.zst")),
|
||||||
|
splitter: Splitter::O200k,
|
||||||
|
nfc: false,
|
||||||
|
ignore_merges: false,
|
||||||
|
});
|
||||||
|
|
||||||
|
static CL100K_BASE: LazyLock<BpeEncoding> = LazyLock::new(|| BpeEncoding {
|
||||||
|
table: RankTable::parse(include_bytes!("../../data/cl100k_base.bin.zst")),
|
||||||
|
splitter: Splitter::Cl100k,
|
||||||
|
nfc: false,
|
||||||
|
ignore_merges: false,
|
||||||
|
});
|
||||||
|
|
||||||
|
// ── Qwen3 ───────────────────────────────────────────────────────────────
|
||||||
|
// Pattern: qwen3.8.tokenizer.json pre_tokenizer Split regex, verified equal
|
||||||
|
// to data/families.json qwen3.pre. Note `\p{N}` matches a SINGLE digit
|
||||||
|
// (unlike cl100k's {1,3}) and letters admit trailing marks `[\p{L}\p{M}]+`.
|
||||||
|
// Normalizer is NFC (`nfc: true`). The pack (tools/pack-qwen.ts) empties the
|
||||||
|
// 201 merge-unreachable vocab slots so the engine's whole-piece
|
||||||
|
// short-circuit cannot emit ids the HF reference (ignore_merges=false)
|
||||||
|
// never produces.
|
||||||
|
|
||||||
|
/// Reference regex, kept as the scanner's dev/test differential oracle
|
||||||
|
/// (see `scan::qwen` tests).
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(crate) const QWEN3_PATTERN: &str = r"(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\r\n\p{L}\p{N}]?[\p{L}\p{M}]+|\p{N}| ?[^\s\p{L}\p{M}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+";
|
||||||
|
|
||||||
|
static QWEN3: LazyLock<BpeEncoding> = LazyLock::new(|| BpeEncoding {
|
||||||
|
table: RankTable::parse(include_bytes!("../../data/qwen3.bin.zst")),
|
||||||
|
splitter: Splitter::Qwen,
|
||||||
|
nfc: true,
|
||||||
|
ignore_merges: false,
|
||||||
|
});
|
||||||
|
|
||||||
|
// ── DeepSeek ────────────────────────────────────────────────────────────
|
||||||
|
// Three-stage HF Split(Isolated) chain from data/families.json deepseek3
|
||||||
|
// (cache/deepseek-v4.tokenizer.json pre_tokenizer; base BPE identical
|
||||||
|
// V3..V4): digits ≤3, then CJK runs (Han + hiragana + katakana blocks),
|
||||||
|
// then the main pattern with a punctuation-prefix-letters alternate.
|
||||||
|
|
||||||
|
// Stage patterns kept as the dev/test differential reference for the
|
||||||
|
// scanner (`scan::deepseek`).
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(crate) const DEEPSEEK_STAGE_DIGITS: &str = r"\p{N}{1,3}";
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(crate) const DEEPSEEK_STAGE_CJK: &str = r"[一-龥-ゟ゠-ヿ]+";
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(crate) const DEEPSEEK_STAGE_MAIN: &str = concat!(
|
||||||
|
r##"[!"#$%&'()*+,\-./:;<=>?@\[\\\]^_`{|}~][A-Za-z]+"##,
|
||||||
|
r"|[^\r\n\p{L}\p{P}\p{S}]?[\p{L}\p{M}]+",
|
||||||
|
r"| ?[\p{P}\p{S}]+[\r\n]*",
|
||||||
|
r"|\s*[\r\n]+",
|
||||||
|
r"|\s+(?!\S)",
|
||||||
|
r"|\s+",
|
||||||
|
);
|
||||||
|
|
||||||
|
static DEEPSEEK3: LazyLock<BpeEncoding> = LazyLock::new(|| BpeEncoding {
|
||||||
|
table: RankTable::parse(include_bytes!("../../data/deepseek3.bin.zst")),
|
||||||
|
splitter: Splitter::DeepSeek,
|
||||||
|
nfc: false,
|
||||||
|
ignore_merges: false,
|
||||||
|
});
|
||||||
|
|
||||||
|
// ── Kimi ────────────────────────────────────────────────────────────────
|
||||||
|
// Runtime split: hand-written scanner (`scan::kimi`), a port of the
|
||||||
|
// tokenization_kimi.py pat_str (8-alternate join, class intersection
|
||||||
|
// `&&[^\p{Han}]`, `\s+(?!\S)` lookahead). The regex below is kept as the
|
||||||
|
// dev/test differential reference; it is verified equal to
|
||||||
|
// data/families.json kimi_k2.pattern.
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(crate) const KIMI_K2_PATTERN: &str = concat!(
|
||||||
|
r"[\p{Han}]+",
|
||||||
|
r"|[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]*[\p{Ll}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]+(?i:'s|'t|'re|'ve|'m|'ll|'d)?",
|
||||||
|
r"|[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]+[\p{Ll}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]*(?i:'s|'t|'re|'ve|'m|'ll|'d)?",
|
||||||
|
r"|\p{N}{1,3}",
|
||||||
|
r"| ?[^\s\p{L}\p{N}]+[\r\n]*",
|
||||||
|
r"|\s*[\r\n]+",
|
||||||
|
r"|\s+(?!\S)",
|
||||||
|
r"|\s+",
|
||||||
|
);
|
||||||
|
|
||||||
|
static KIMI_K2: LazyLock<BpeEncoding> = LazyLock::new(|| BpeEncoding {
|
||||||
|
table: RankTable::parse(include_bytes!("../../data/kimi_k2.bin.zst")),
|
||||||
|
splitter: Splitter::Kimi,
|
||||||
|
nfc: false,
|
||||||
|
ignore_merges: false,
|
||||||
|
});
|
||||||
|
|
||||||
|
// ── GLM ─────────────────────────────────────────────────────────────────
|
||||||
|
// Pattern: glm-5.tokenizer.json pre_tokenizer Split regex, verified equal
|
||||||
|
// to data/families.json glm5.pre — and character-identical to tiktoken's
|
||||||
|
// cl100k_base pattern, so the runtime splitter aliases the cl100k scanner
|
||||||
|
// (`scan::cl100k`) instead of duplicating it; `glm_scan_tests` below proves
|
||||||
|
// byte-identical piece boundaries against the GLM reference regex.
|
||||||
|
// `ignore_merges: true` — whole-piece vocab hits bypass merging; the
|
||||||
|
// engine's encode_piece short-circuit implements exactly this (proven by
|
||||||
|
// directed fixtures: greedy-merge-unreachable tokens like ' 参考' encode
|
||||||
|
// as one id).
|
||||||
|
|
||||||
|
/// Reference regex, kept as the dev/test differential oracle for the
|
||||||
|
/// cl100k-scanner alias.
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(crate) const GLM5_PATTERN: &str = r"(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\r\n\p{L}\p{N}]?\p{L}+|\p{N}{1,3}| ?[^\s\p{L}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+";
|
||||||
|
|
||||||
|
static GLM5: LazyLock<BpeEncoding> = LazyLock::new(|| BpeEncoding {
|
||||||
|
table: RankTable::parse(include_bytes!("../../data/glm5.bin.zst")),
|
||||||
|
splitter: Splitter::Cl100k,
|
||||||
|
nfc: false,
|
||||||
|
ignore_merges: true,
|
||||||
|
});
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod glm_scan_tests {
|
||||||
|
//! Differential justifying the cl100k-scanner alias: piece boundaries
|
||||||
|
//! from `Splitter::Cl100k` vs the GLM-5 reference regex, over the
|
||||||
|
//! corpus, every glm5 fixture text, and seeded random CJK/Latin mixes.
|
||||||
|
|
||||||
|
use super::GLM5_PATTERN;
|
||||||
|
use crate::utok::pretoken::Splitter;
|
||||||
|
|
||||||
|
/// splitmix64 (same scheme as tests/openai.rs).
|
||||||
|
struct Rng(u64);
|
||||||
|
|
||||||
|
impl Rng {
|
||||||
|
fn next(&mut self) -> u64 {
|
||||||
|
self.0 = self.0.wrapping_add(0x9e3779b97f4a7c15);
|
||||||
|
let mut z = self.0;
|
||||||
|
z = (z ^ (z >> 30)).wrapping_mul(0xbf58476d1ce4e5b9);
|
||||||
|
z = (z ^ (z >> 27)).wrapping_mul(0x94d049bb133111eb);
|
||||||
|
z ^ (z >> 31)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn random_strings(seed: u64, n: usize) -> Vec<String> {
|
||||||
|
let mut rng = Rng(seed);
|
||||||
|
let mut out = Vec::with_capacity(n * 2);
|
||||||
|
for _ in 0..n {
|
||||||
|
let len = (rng.next() % 300 + 1) as usize;
|
||||||
|
let bytes: Vec<u8> = (0..len).map(|_| rng.next() as u8).collect();
|
||||||
|
out.push(String::from_utf8_lossy(&bytes).into_owned());
|
||||||
|
}
|
||||||
|
for _ in 0..n {
|
||||||
|
let len = (rng.next() % 120 + 1) as usize;
|
||||||
|
let s: String = (0..len)
|
||||||
|
.map(|_| {
|
||||||
|
let c = match rng.next() % 8 {
|
||||||
|
0 => rng.next() % 0x80, // ASCII
|
||||||
|
1 => 0x20 + rng.next() % 4, // spaces/punct
|
||||||
|
2 => rng.next() % 0x250, // Latin+ext
|
||||||
|
3 => 0x4e00 + rng.next() % 0x100, // CJK
|
||||||
|
4 => 0x1f300 + rng.next() % 0x100, // emoji
|
||||||
|
5 => 0x300 + rng.next() % 0x70, // combining marks
|
||||||
|
6 => [9, 10, 13, 32][(rng.next() % 4) as usize], // whitespace
|
||||||
|
_ => rng.next() % 0x11_0000,
|
||||||
|
};
|
||||||
|
char::from_u32(c as u32).unwrap_or('\u{fffd}')
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
out.push(s);
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn glm_cl100k_alias_matches_reference_regex() {
|
||||||
|
let reference = Splitter::new(&[GLM5_PATTERN]);
|
||||||
|
let scanner = Splitter::Cl100k;
|
||||||
|
let corpus: Vec<String> =
|
||||||
|
serde_json::from_str(include_str!("../../fixtures/corpus.json")).unwrap();
|
||||||
|
let fixture: serde_json::Value =
|
||||||
|
serde_json::from_str(include_str!("../../fixtures/glm5.json")).unwrap();
|
||||||
|
let fixture_texts: Vec<String> = fixture["cases"]
|
||||||
|
.as_array()
|
||||||
|
.unwrap()
|
||||||
|
.iter()
|
||||||
|
.map(|c| c["text"].as_str().unwrap().to_string())
|
||||||
|
.collect();
|
||||||
|
let texts: Vec<String> = corpus
|
||||||
|
.into_iter()
|
||||||
|
.chain(fixture_texts)
|
||||||
|
.chain(random_strings(0x61c8_8646, 400))
|
||||||
|
.collect();
|
||||||
|
for text in &texts {
|
||||||
|
assert_eq!(
|
||||||
|
scanner.split(text),
|
||||||
|
reference.split(text),
|
||||||
|
"piece boundaries diverge on {text:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve the BPE encoding for a non-Claude family.
|
||||||
|
pub(crate) fn bpe_for(enc: Encoding) -> &'static BpeEncoding {
|
||||||
|
match enc {
|
||||||
|
Encoding::O200kBase => &O200K_BASE,
|
||||||
|
Encoding::Cl100kBase => &CL100K_BASE,
|
||||||
|
Encoding::Qwen3 => &QWEN3,
|
||||||
|
Encoding::DeepSeekV3 => &DEEPSEEK3,
|
||||||
|
Encoding::KimiK2 => &KIMI_K2,
|
||||||
|
Encoding::Glm5 => &GLM5,
|
||||||
|
_ => unreachable!("claude families never reach bpe_for"),
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,129 @@
|
|||||||
|
//! Claude ctok golden tests over the public [`crate::utok::Encoding`] surface.
|
||||||
|
//!
|
||||||
|
//! The fixture corpora record reference *message* counts (Python ctok 1.0.0
|
||||||
|
//! `token_count`, plus raw live `count_tokens` rows for sonnet-5). Public
|
||||||
|
//! `count()` returns *content* tokens, so each expectation subtracts the
|
||||||
|
//! fixed per-family frame overhead (v3: 7, v4.7: 11, 5-series: 6); the
|
||||||
|
//! content/message split itself is asserted by the module's internal tests.
|
||||||
|
|
||||||
|
use serde::Deserialize;
|
||||||
|
|
||||||
|
use crate::utok::Encoding;
|
||||||
|
|
||||||
|
#[derive(Deserialize)]
|
||||||
|
struct Fixture {
|
||||||
|
text: String,
|
||||||
|
v3: u32,
|
||||||
|
v4_7: u32,
|
||||||
|
v5: u32,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Deserialize)]
|
||||||
|
struct LiveRow {
|
||||||
|
text: String,
|
||||||
|
count: u32,
|
||||||
|
}
|
||||||
|
|
||||||
|
const FRAME_V3: u32 = 7;
|
||||||
|
const FRAME_V47: u32 = 11;
|
||||||
|
const FRAME_V5: u32 = 6;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn matches_ctok_reference_counts() {
|
||||||
|
let fixtures: Vec<Fixture> =
|
||||||
|
serde_json::from_str(include_str!("../claude/testdata/fixtures.json"))
|
||||||
|
.expect("fixtures parse");
|
||||||
|
assert!(fixtures.len() >= 250, "fixture corpus unexpectedly small: {}", fixtures.len());
|
||||||
|
for f in &fixtures {
|
||||||
|
for (enc, want) in [
|
||||||
|
(Encoding::ClaudeV3, f.v3 - FRAME_V3),
|
||||||
|
(Encoding::ClaudeV47, f.v4_7 - FRAME_V47),
|
||||||
|
(Encoding::ClaudeV5, f.v5 - FRAME_V5),
|
||||||
|
] {
|
||||||
|
assert_eq!(enc.count(&f.text), want, "encoding {enc:?} text {:?}", f.text);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn matches_live_sonnet5_counts() {
|
||||||
|
let rows: Vec<LiveRow> =
|
||||||
|
serde_json::from_str(include_str!("../claude/testdata/sonnet5_live.json"))
|
||||||
|
.expect("rows parse");
|
||||||
|
assert!(rows.len() >= 50, "live corpus unexpectedly small: {}", rows.len());
|
||||||
|
for row in &rows {
|
||||||
|
assert_eq!(
|
||||||
|
Encoding::ClaudeV5Sonnet.count(&row.text),
|
||||||
|
row.count - FRAME_V5,
|
||||||
|
"text {:?}",
|
||||||
|
row.text
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn encode_is_none_for_all_claude_families() {
|
||||||
|
// ctok reconstructs counts, not boundaries: no id sequence exists.
|
||||||
|
for enc in
|
||||||
|
[Encoding::ClaudeV3, Encoding::ClaudeV47, Encoding::ClaudeV5, Encoding::ClaudeV5Sonnet]
|
||||||
|
{
|
||||||
|
assert_eq!(enc.encode("x"), None, "{enc:?}");
|
||||||
|
assert_eq!(enc.encode(""), None, "{enc:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn count_routes_per_family() {
|
||||||
|
// Each variant must reach its own family table/frame, not a shared one.
|
||||||
|
// v3 folds curly quotes and marks 4+ caps runs; v4.7+ does neither.
|
||||||
|
let caps = "HELLO “WORLD”";
|
||||||
|
let v3 = Encoding::ClaudeV3.count(caps);
|
||||||
|
let v47 = Encoding::ClaudeV47.count(caps);
|
||||||
|
assert_ne!(v3, v47, "v3 and v4.7 must diverge on caps/quotes");
|
||||||
|
|
||||||
|
// V47 and V5 share a vocabulary but differ on the frame ⟨bow⟩: a
|
||||||
|
// leading space is priced differently.
|
||||||
|
assert_ne!(
|
||||||
|
Encoding::ClaudeV47.count(" hello"),
|
||||||
|
Encoding::ClaudeV5.count(" hello"),
|
||||||
|
"v4.7 and opus-5 must diverge on the frame bow"
|
||||||
|
);
|
||||||
|
|
||||||
|
// Trailing newlines: opus-5 absorbs the run for free, sonnet-5 pays the
|
||||||
|
// ladder (tile(run) - 1), v4.7 pays full price — strict ordering at a
|
||||||
|
// long run where the tiling needs several pieces.
|
||||||
|
let tail = format!("hello{}", "\n".repeat(64));
|
||||||
|
let v5 = Encoding::ClaudeV5.count(&tail);
|
||||||
|
let s5 = Encoding::ClaudeV5Sonnet.count(&tail);
|
||||||
|
let v47 = Encoding::ClaudeV47.count(&tail);
|
||||||
|
assert_eq!(v5, Encoding::ClaudeV5.count("hello"), "opus-5 tail is free");
|
||||||
|
assert!(s5 > v5, "sonnet-5 tail is not free");
|
||||||
|
assert!(s5 < v47, "sonnet-5 tail gets the ladder discount");
|
||||||
|
|
||||||
|
// Empty content is zero on the 5-series; v3/v4.7 still pay the frame
|
||||||
|
// ⟨bow⟩ token (matches ctok / the fixture corpus).
|
||||||
|
assert_eq!(Encoding::ClaudeV5.count(""), 0);
|
||||||
|
assert_eq!(Encoding::ClaudeV5Sonnet.count(""), 0);
|
||||||
|
assert_eq!(Encoding::ClaudeV3.count(""), 1);
|
||||||
|
assert_eq!(Encoding::ClaudeV47.count(""), 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn utf16_and_utf32_flavor_parity() {
|
||||||
|
// Valid text counts flavor-invariantly through the public generic API.
|
||||||
|
let fixtures: Vec<Fixture> =
|
||||||
|
serde_json::from_str(include_str!("../claude/testdata/fixtures.json"))
|
||||||
|
.expect("fixtures parse");
|
||||||
|
let encodings =
|
||||||
|
[Encoding::ClaudeV3, Encoding::ClaudeV47, Encoding::ClaudeV5, Encoding::ClaudeV5Sonnet];
|
||||||
|
for f in &fixtures {
|
||||||
|
let u16s: Vec<u16> = f.text.encode_utf16().collect();
|
||||||
|
let u32s: Vec<u32> = f.text.chars().map(u32::from).collect();
|
||||||
|
for enc in encodings {
|
||||||
|
let want = enc.count(f.text.as_str());
|
||||||
|
assert_eq!(enc.count(&u16s), want, "utf16 {enc:?} text {:?}", f.text);
|
||||||
|
assert_eq!(enc.count(&u32s), want, "utf32 {enc:?} text {:?}", f.text);
|
||||||
|
assert_eq!(enc.encode(&u16s), None, "{enc:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
//! Golden-fixture test: DeepSeekV3 vs reference HF tokenizers encode
|
||||||
|
//! (add_special_tokens=False) over cache/deepseek-v4.tokenizer.json.
|
||||||
|
//! Base BPE is identical V3..V4 (verified upstream: same vocab + merges
|
||||||
|
//! hash), so one encoding covers the whole family.
|
||||||
|
|
||||||
|
use crate::utok::Encoding;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn deepseek_matches_reference() {
|
||||||
|
let raw = include_str!("../../../fixtures/deepseek3.json");
|
||||||
|
let fixture: serde_json::Value = serde_json::from_str(raw).expect("fixture parses");
|
||||||
|
let cases = fixture["cases"].as_array().expect("cases array");
|
||||||
|
assert!(!cases.is_empty());
|
||||||
|
for case in cases {
|
||||||
|
let text = case["text"].as_str().expect("text");
|
||||||
|
let want: Vec<u32> = case["ids"]
|
||||||
|
.as_array()
|
||||||
|
.expect("ids")
|
||||||
|
.iter()
|
||||||
|
.map(|v| v.as_u64().expect("id") as u32)
|
||||||
|
.collect();
|
||||||
|
let count = case["count"].as_u64().expect("count") as u32;
|
||||||
|
assert_eq!(count as usize, want.len(), "fixture self-consistency: {text:?}");
|
||||||
|
|
||||||
|
let got = Encoding::DeepSeekV3
|
||||||
|
.encode(text)
|
||||||
|
.expect("deepseek is a BPE family");
|
||||||
|
assert_eq!(got, want, "encode mismatch on {text:?}");
|
||||||
|
// Merge-unreachable sentinels (ids 0..2) are blanked in the packed
|
||||||
|
// table; encode_ordinary must never emit them — including for the
|
||||||
|
// fixture cases spelling them out verbatim.
|
||||||
|
assert!(got.iter().all(|&id| id > 2), "sentinel id emitted on {text:?}");
|
||||||
|
assert_eq!(Encoding::DeepSeekV3.count(text), count, "count mismatch on {text:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// V3 parity spot check: ids generated with deepseek-ai/DeepSeek-V3's
|
||||||
|
/// tokenizer for one mixed CJK/digit/latin sample (see fixture
|
||||||
|
/// `v3_parity`); must equal our V4-table output byte for byte.
|
||||||
|
#[test]
|
||||||
|
fn deepseek_v3_parity_sample() {
|
||||||
|
let raw = include_str!("../../../fixtures/deepseek3.json");
|
||||||
|
let fixture: serde_json::Value = serde_json::from_str(raw).expect("fixture parses");
|
||||||
|
let parity = &fixture["v3_parity"];
|
||||||
|
let text = parity["text"].as_str().expect("v3_parity.text");
|
||||||
|
let want: Vec<u32> = parity["ids"]
|
||||||
|
.as_array()
|
||||||
|
.expect("v3_parity.ids")
|
||||||
|
.iter()
|
||||||
|
.map(|v| v.as_u64().expect("id") as u32)
|
||||||
|
.collect();
|
||||||
|
assert_eq!(Encoding::DeepSeekV3.encode(text).unwrap(), want);
|
||||||
|
assert_eq!(Encoding::DeepSeekV3.count(text) as usize, want.len());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Flavor invariance: UTF-16 and UTF-32 inputs must yield the same ids
|
||||||
|
/// and counts as `&str` for every fixture case (all valid text).
|
||||||
|
#[test]
|
||||||
|
fn deepseek_utf16_utf32_parity() {
|
||||||
|
let raw = include_str!("../../../fixtures/deepseek3.json");
|
||||||
|
let fixture: serde_json::Value = serde_json::from_str(raw).expect("fixture parses");
|
||||||
|
for case in fixture["cases"].as_array().expect("cases array") {
|
||||||
|
let text = case["text"].as_str().expect("text");
|
||||||
|
let want = Encoding::DeepSeekV3.encode(text).unwrap();
|
||||||
|
let count = Encoding::DeepSeekV3.count(text);
|
||||||
|
|
||||||
|
let u16s: Vec<u16> = text.encode_utf16().collect();
|
||||||
|
assert_eq!(Encoding::DeepSeekV3.encode(&u16s).unwrap(), want, "utf16 ids on {text:?}");
|
||||||
|
assert_eq!(Encoding::DeepSeekV3.count(&u16s), count, "utf16 count on {text:?}");
|
||||||
|
|
||||||
|
let u32s: Vec<u32> = text.chars().map(|c| c as u32).collect();
|
||||||
|
assert_eq!(Encoding::DeepSeekV3.encode(&u32s).unwrap(), want, "utf32 ids on {text:?}");
|
||||||
|
assert_eq!(Encoding::DeepSeekV3.count(&u32s), count, "utf32 count on {text:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,88 @@
|
|||||||
|
//! Golden-fixture test: Glm5 vs reference HF tokenizers
|
||||||
|
//! (add_special_tokens=False).
|
||||||
|
//!
|
||||||
|
//! GLM-4.x note: GLM-5 is an ID-preserving superset of GLM-4.x (~3.5k extra
|
||||||
|
//! merges), so GLM-4.x counts are near-exact under this table — no separate
|
||||||
|
//! 4.x fixtures needed.
|
||||||
|
//!
|
||||||
|
//! The fixture includes directed `ignore_merges` probes: vocab tokens (e.g.
|
||||||
|
//! ' 参考' = 99855) that HF's greedy merge order never produces bottom-up —
|
||||||
|
//! plain BPE emits multiple ids (' 参考' → [26767, 224, 98580]) while
|
||||||
|
//! `ignore_merges` emits the single whole-piece id. Verified against the
|
||||||
|
//! Python reference with the flag toggled (tools/gen-glm-fixtures.py).
|
||||||
|
|
||||||
|
use crate::utok::Encoding;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn glm5_matches_reference() {
|
||||||
|
let raw = include_str!("../../../fixtures/glm5.json");
|
||||||
|
let fixture: serde_json::Value = serde_json::from_str(raw).expect("fixture parses");
|
||||||
|
let cases = fixture["cases"].as_array().expect("cases array");
|
||||||
|
assert!(!cases.is_empty());
|
||||||
|
for case in cases {
|
||||||
|
let text = case["text"].as_str().expect("text");
|
||||||
|
let want: Vec<u32> = case["ids"]
|
||||||
|
.as_array()
|
||||||
|
.expect("ids")
|
||||||
|
.iter()
|
||||||
|
.map(|v| v.as_u64().expect("id") as u32)
|
||||||
|
.collect();
|
||||||
|
let count = case["count"].as_u64().expect("count") as u32;
|
||||||
|
assert_eq!(count as usize, want.len(), "fixture self-consistency: {text:?}");
|
||||||
|
|
||||||
|
let got = Encoding::Glm5.encode(text).expect("glm is a BPE family");
|
||||||
|
assert_eq!(got, want, "encode mismatch on {text:?}");
|
||||||
|
assert_eq!(Encoding::Glm5.count(text), count, "count mismatch on {text:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Directed ignore_merges semantics: ' 参考' exists in vocab (99855) but
|
||||||
|
/// HF's greedy merge order never forms it from bytes; normal merge-loop BPE
|
||||||
|
/// yields [26767, 224, 98580] (verified with the Python reference with
|
||||||
|
/// ignore_merges=false). The whole-piece short-circuit must win.
|
||||||
|
#[test]
|
||||||
|
fn glm5_ignore_merges_whole_piece_wins() {
|
||||||
|
assert_eq!(Encoding::Glm5.encode(" 参考").unwrap(), vec![99855]);
|
||||||
|
assert_eq!(Encoding::Glm5.count(" 参考"), 1);
|
||||||
|
assert_eq!(Encoding::Glm5.encode(" 参考资料").unwrap(), vec![99924]);
|
||||||
|
// Same token mid-text: pretokenizer isolates ' 参考' as its own piece.
|
||||||
|
assert_eq!(Encoding::Glm5.encode("龘 参考").unwrap(), vec![82225, 246, 99855]);
|
||||||
|
// Extended so the piece is NOT a whole-vocab hit: the merge loop runs
|
||||||
|
// and must match the reference (rank-order merging, ' 参考龘' is OOV).
|
||||||
|
assert_eq!(
|
||||||
|
Encoding::Glm5.count(" 参考龘"),
|
||||||
|
Encoding::Glm5.encode(" 参考龘").unwrap().len() as u32
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// UTF-16 / UTF-32 parity: every fixture case must produce identical ids
|
||||||
|
/// and counts when tokenized natively in u16/u32 code units (no transcode).
|
||||||
|
/// The corpus includes emoji (surrogate pairs in UTF-16) — asserted
|
||||||
|
/// explicitly below so the coverage is self-documenting.
|
||||||
|
#[test]
|
||||||
|
fn glm5_utf16_utf32_parity() {
|
||||||
|
let raw = include_str!("../../../fixtures/glm5.json");
|
||||||
|
let fixture: serde_json::Value = serde_json::from_str(raw).expect("fixture parses");
|
||||||
|
let cases = fixture["cases"].as_array().expect("cases array");
|
||||||
|
let mut saw_surrogate_pair = false;
|
||||||
|
for case in cases {
|
||||||
|
let text = case["text"].as_str().expect("text");
|
||||||
|
let want: Vec<u32> = case["ids"]
|
||||||
|
.as_array()
|
||||||
|
.expect("ids")
|
||||||
|
.iter()
|
||||||
|
.map(|v| v.as_u64().expect("id") as u32)
|
||||||
|
.collect();
|
||||||
|
let count = want.len() as u32;
|
||||||
|
|
||||||
|
let utf16: Vec<u16> = text.encode_utf16().collect();
|
||||||
|
let utf32: Vec<u32> = text.chars().map(|c| c as u32).collect();
|
||||||
|
saw_surrogate_pair |= utf16.len() > utf32.len();
|
||||||
|
|
||||||
|
assert_eq!(Encoding::Glm5.encode(&utf16).unwrap(), want, "utf16 encode mismatch on {text:?}");
|
||||||
|
assert_eq!(Encoding::Glm5.count(&utf16), count, "utf16 count mismatch on {text:?}");
|
||||||
|
assert_eq!(Encoding::Glm5.encode(&utf32).unwrap(), want, "utf32 encode mismatch on {text:?}");
|
||||||
|
assert_eq!(Encoding::Glm5.count(&utf32), count, "utf32 count mismatch on {text:?}");
|
||||||
|
}
|
||||||
|
assert!(saw_surrogate_pair, "corpus must exercise a UTF-16 surrogate pair");
|
||||||
|
}
|
||||||
@@ -0,0 +1,60 @@
|
|||||||
|
//! Golden-fixture test: KimiK2 vs reference tiktoken encode_ordinary.
|
||||||
|
|
||||||
|
use crate::utok::Encoding;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn kimi_k2_matches_reference() {
|
||||||
|
let raw = include_str!("../../../fixtures/kimi_k2.json");
|
||||||
|
let fixture: serde_json::Value = serde_json::from_str(raw).expect("fixture parses");
|
||||||
|
let cases = fixture["cases"].as_array().expect("cases array");
|
||||||
|
assert!(!cases.is_empty());
|
||||||
|
for case in cases {
|
||||||
|
let text = case["text"].as_str().expect("text");
|
||||||
|
let want: Vec<u32> = case["ids"]
|
||||||
|
.as_array()
|
||||||
|
.expect("ids")
|
||||||
|
.iter()
|
||||||
|
.map(|v| v.as_u64().expect("id") as u32)
|
||||||
|
.collect();
|
||||||
|
let count = case["count"].as_u64().expect("count") as u32;
|
||||||
|
assert_eq!(count as usize, want.len(), "fixture self-consistency: {text:?}");
|
||||||
|
|
||||||
|
let got = Encoding::KimiK2.encode(text).expect("kimi is a BPE family");
|
||||||
|
assert_eq!(got, want, "encode mismatch on {text:?}");
|
||||||
|
assert_eq!(Encoding::KimiK2.count(text), count, "count mismatch on {text:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every fixture case must produce identical ids/counts when the input is
|
||||||
|
/// re-encoded as UTF-16 or UTF-32 units (flavor invariance for valid text).
|
||||||
|
#[test]
|
||||||
|
fn kimi_k2_utf16_utf32_parity() {
|
||||||
|
let raw = include_str!("../../../fixtures/kimi_k2.json");
|
||||||
|
let fixture: serde_json::Value = serde_json::from_str(raw).expect("fixture parses");
|
||||||
|
let cases = fixture["cases"].as_array().expect("cases array");
|
||||||
|
// Surrogate-pair-heavy extra: Kimi's Han classes meet 2-unit UTF-16
|
||||||
|
// codepoints (emoji, 𝕏 U+1D54D, and astral Han U+20000 𠀀).
|
||||||
|
let extra = "👨👩👧👦𝕏≈中文𠀀𠀁English🇹🇵123'll \n";
|
||||||
|
let texts = cases
|
||||||
|
.iter()
|
||||||
|
.map(|c| c["text"].as_str().expect("text"))
|
||||||
|
.chain(std::iter::once(extra));
|
||||||
|
for text in texts {
|
||||||
|
let want = Encoding::KimiK2.encode(text).expect("kimi is a BPE family");
|
||||||
|
let u16s: Vec<u16> = text.encode_utf16().collect();
|
||||||
|
let u32s: Vec<u32> = text.chars().map(|c| c as u32).collect();
|
||||||
|
assert_eq!(
|
||||||
|
Encoding::KimiK2.encode(u16s.as_slice()),
|
||||||
|
Some(want.clone()),
|
||||||
|
"utf16 encode mismatch on {text:?}"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
Encoding::KimiK2.encode(u32s.as_slice()),
|
||||||
|
Some(want.clone()),
|
||||||
|
"utf32 encode mismatch on {text:?}"
|
||||||
|
);
|
||||||
|
let n = want.len() as u32;
|
||||||
|
assert_eq!(Encoding::KimiK2.count(u16s.as_slice()), n, "utf16 count mismatch on {text:?}");
|
||||||
|
assert_eq!(Encoding::KimiK2.count(u32s.as_slice()), n, "utf32 count mismatch on {text:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,180 @@
|
|||||||
|
//! OpenAI family tests: golden fixtures from Python tiktoken, plus a
|
||||||
|
//! differential test against tiktoken-rs over the corpus and seeded
|
||||||
|
//! randomized strings.
|
||||||
|
|
||||||
|
use serde::Deserialize;
|
||||||
|
|
||||||
|
use crate::utok::Encoding;
|
||||||
|
|
||||||
|
#[derive(Deserialize)]
|
||||||
|
struct Fixture {
|
||||||
|
#[allow(dead_code)]
|
||||||
|
generator: String,
|
||||||
|
cases: Vec<Case>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Deserialize)]
|
||||||
|
struct Case {
|
||||||
|
text: String,
|
||||||
|
ids: Vec<u32>,
|
||||||
|
count: u32,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn check_fixture(enc: Encoding, json: &str) {
|
||||||
|
let fx: Fixture = serde_json::from_str(json).unwrap();
|
||||||
|
assert!(!fx.cases.is_empty());
|
||||||
|
for (i, case) in fx.cases.iter().enumerate() {
|
||||||
|
let ids = enc.encode(&case.text).expect("BPE family must encode");
|
||||||
|
assert_eq!(
|
||||||
|
ids,
|
||||||
|
case.ids,
|
||||||
|
"{enc:?} case {i} ids mismatch: {:?}…",
|
||||||
|
&case.text[..case.text.len().min(60)]
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
enc.count(&case.text),
|
||||||
|
case.count,
|
||||||
|
"{enc:?} case {i} count mismatch: {:?}…",
|
||||||
|
&case.text[..case.text.len().min(60)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn o200k_fixtures() {
|
||||||
|
check_fixture(Encoding::O200kBase, include_str!("../../../fixtures/o200k_base.json"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn cl100k_fixtures() {
|
||||||
|
check_fixture(Encoding::Cl100kBase, include_str!("../../../fixtures/cl100k_base.json"));
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── differential vs tiktoken-rs ─────────────────────────────────────────
|
||||||
|
|
||||||
|
/// splitmix64: tiny deterministic PRNG, no dev-dep needed.
|
||||||
|
struct Rng(u64);
|
||||||
|
|
||||||
|
impl Rng {
|
||||||
|
fn next(&mut self) -> u64 {
|
||||||
|
self.0 = self.0.wrapping_add(0x9e3779b97f4a7c15);
|
||||||
|
let mut z = self.0;
|
||||||
|
z = (z ^ (z >> 30)).wrapping_mul(0xbf58476d1ce4e5b9);
|
||||||
|
z = (z ^ (z >> 27)).wrapping_mul(0x94d049bb133111eb);
|
||||||
|
z ^ (z >> 31)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Deterministic pseudo-random test strings: raw bytes laundered through
|
||||||
|
/// `from_utf8_lossy` (both sides see the same valid-UTF-8 string), plus
|
||||||
|
/// char-sampled strings biased toward tokenizer-relevant ranges.
|
||||||
|
fn random_strings(seed: u64, n: usize) -> Vec<String> {
|
||||||
|
let mut rng = Rng(seed);
|
||||||
|
let mut out = Vec::with_capacity(n * 2);
|
||||||
|
for _ in 0..n {
|
||||||
|
let len = (rng.next() % 300 + 1) as usize;
|
||||||
|
let bytes: Vec<u8> = (0..len).map(|_| rng.next() as u8).collect();
|
||||||
|
out.push(String::from_utf8_lossy(&bytes).into_owned());
|
||||||
|
}
|
||||||
|
for _ in 0..n {
|
||||||
|
let len = (rng.next() % 120 + 1) as usize;
|
||||||
|
let s: String = (0..len)
|
||||||
|
.map(|_| {
|
||||||
|
let c = match rng.next() % 8 {
|
||||||
|
0 => rng.next() % 0x80, // ASCII
|
||||||
|
1 => 0x20 + rng.next() % 4, // spaces/punct
|
||||||
|
2 => rng.next() % 0x250, // Latin+ext
|
||||||
|
3 => 0x4e00 + rng.next() % 0x100, // CJK
|
||||||
|
4 => 0x1f300 + rng.next() % 0x100, // emoji
|
||||||
|
5 => 0x300 + rng.next() % 0x70, // combining marks
|
||||||
|
6 => [9, 10, 13, 32][(rng.next() % 4) as usize], // whitespace
|
||||||
|
_ => rng.next() % 0x11_0000,
|
||||||
|
};
|
||||||
|
char::from_u32(c as u32).unwrap_or('\u{fffd}')
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
out.push(s);
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Scanner-torture strings: class-overlap backtracking, contraction case
|
||||||
|
/// folds, whitespace-trio boundaries, o200k slash tails.
|
||||||
|
const TRICKY: &[&str] = &[
|
||||||
|
"ʰAB", // upper-run backtracks to shared Lm codepoint
|
||||||
|
"XYʰZ", // alt1 wins with "XYʰ" although alt2 would take "XYʰZ"
|
||||||
|
"ʰ", // single shared-class codepoint matches alt1 via backtrack
|
||||||
|
"\u{301}a\u{301}", // mark-initial: prefix-vs-body ambiguity
|
||||||
|
"'ſ 'S 'ſt", // U+017F long s in (?i:'s)
|
||||||
|
"'Re'VE'lL'd", // two-letter contraction folds
|
||||||
|
"'r 'v 'l", // near-miss contractions
|
||||||
|
"\u{c}\n\u{fffd}G", // \s*[\r\n]+ vs \s+(?!\S) (the historical cl100k bug)
|
||||||
|
"x \t\u{b}\r\n \n\t y",
|
||||||
|
"Džungla Dž DŽ", // titlecase letters
|
||||||
|
"a/b//\n/", // o200k punct slash tail
|
||||||
|
" /",
|
||||||
|
"१२३४ ٣٢١", // non-ASCII digits, {1,3} grouping
|
||||||
|
"?\u{17f}\u{17f}", // prefix + long-s run
|
||||||
|
" ",
|
||||||
|
" x",
|
||||||
|
"\r",
|
||||||
|
"\n \n",
|
||||||
|
];
|
||||||
|
|
||||||
|
fn check_differential(enc: Encoding, reference: &tiktoken_rs::CoreBPE, seed: u64) {
|
||||||
|
let corpus: Vec<String> =
|
||||||
|
serde_json::from_str(include_str!("../../../fixtures/corpus.json")).unwrap();
|
||||||
|
// Multi-seed sweep: broader codepoint coverage against Unicode-table
|
||||||
|
// skew between the scanner's class tables and the reference engine.
|
||||||
|
let texts: Vec<String> = corpus
|
||||||
|
.into_iter()
|
||||||
|
.chain(TRICKY.iter().map(|s| s.to_string()))
|
||||||
|
.chain((0..4).flat_map(|k| random_strings(seed.wrapping_add(k * 0x9e37), 150)))
|
||||||
|
.collect();
|
||||||
|
for text in &texts {
|
||||||
|
let want: Vec<u32> = reference.encode_ordinary(text);
|
||||||
|
let got = enc.encode(text.as_str()).unwrap();
|
||||||
|
assert_eq!(got, want, "{enc:?} differential ids mismatch on {text:?}");
|
||||||
|
assert_eq!(
|
||||||
|
enc.count(text.as_str()),
|
||||||
|
want.len() as u32,
|
||||||
|
"{enc:?} differential count mismatch on {text:?}"
|
||||||
|
);
|
||||||
|
|
||||||
|
// UTF flavor parity: same ids/counts from native u16/u32 scans.
|
||||||
|
let u16s: Vec<u16> = text.encode_utf16().collect();
|
||||||
|
let u32s: Vec<u32> = text.chars().map(|c| c as u32).collect();
|
||||||
|
assert_eq!(enc.encode(&u16s).unwrap(), want, "{enc:?} utf16 parity mismatch on {text:?}");
|
||||||
|
assert_eq!(enc.count(&u16s), want.len() as u32, "{enc:?} utf16 count parity on {text:?}");
|
||||||
|
assert_eq!(enc.encode(&u32s).unwrap(), want, "{enc:?} utf32 parity mismatch on {text:?}");
|
||||||
|
assert_eq!(enc.count(&u32s), want.len() as u32, "{enc:?} utf32 count parity on {text:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
// Ill-formed UTF-16 (lone surrogates) must behave like a lossy JS
|
||||||
|
// crossing: identical to the replacement-char string, never panic.
|
||||||
|
let mut rng = Rng(seed ^ 0x5107);
|
||||||
|
for _ in 0..100 {
|
||||||
|
let len = (rng.next() % 40 + 1) as usize;
|
||||||
|
let raw: Vec<u16> = (0..len)
|
||||||
|
.map(|_| match rng.next() % 4 {
|
||||||
|
0 => 0xd800 + (rng.next() % 0x800) as u16, // surrogate soup
|
||||||
|
1 => (rng.next() % 0x80) as u16,
|
||||||
|
_ => rng.next() as u16,
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let lossy = String::from_utf16_lossy(&raw);
|
||||||
|
let want: Vec<u32> = reference.encode_ordinary(&lossy);
|
||||||
|
assert_eq!(enc.encode(&raw).unwrap(), want, "{enc:?} lossy utf16 mismatch on {raw:x?}");
|
||||||
|
assert_eq!(enc.count(&raw), want.len() as u32, "{enc:?} lossy utf16 count on {raw:x?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn o200k_differential() {
|
||||||
|
check_differential(Encoding::O200kBase, &tiktoken_rs::o200k_base().unwrap(), 0x0200f00d);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn cl100k_differential() {
|
||||||
|
check_differential(Encoding::Cl100kBase, &tiktoken_rs::cl100k_base().unwrap(), 0x0100beef);
|
||||||
|
}
|
||||||
@@ -0,0 +1,78 @@
|
|||||||
|
//! Golden-fixture test: Qwen3 vs reference HF tokenizers encode
|
||||||
|
//! (add_special_tokens=false), including NFC normalization and the
|
||||||
|
//! dead-rank (merge-unreachable vocab entry) regressions.
|
||||||
|
|
||||||
|
use crate::utok::Encoding;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn qwen3_matches_reference() {
|
||||||
|
let raw = include_str!("../../../fixtures/qwen3.json");
|
||||||
|
let fixture: serde_json::Value = serde_json::from_str(raw).expect("fixture parses");
|
||||||
|
let cases = fixture["cases"].as_array().expect("cases array");
|
||||||
|
assert!(!cases.is_empty());
|
||||||
|
for case in cases {
|
||||||
|
let text = case["text"].as_str().expect("text");
|
||||||
|
let want: Vec<u32> = case["ids"]
|
||||||
|
.as_array()
|
||||||
|
.expect("ids")
|
||||||
|
.iter()
|
||||||
|
.map(|v| v.as_u64().expect("id") as u32)
|
||||||
|
.collect();
|
||||||
|
let count = case["count"].as_u64().expect("count") as u32;
|
||||||
|
assert_eq!(count as usize, want.len(), "fixture self-consistency: {text:?}");
|
||||||
|
|
||||||
|
let got = Encoding::Qwen3.encode(text).expect("qwen is a BPE family");
|
||||||
|
assert_eq!(got, want, "encode mismatch on {text:?}");
|
||||||
|
assert_eq!(Encoding::Qwen3.count(text), count, "count mismatch on {text:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every fixture case must produce identical ids/counts when the input is
|
||||||
|
/// re-encoded as UTF-16 or UTF-32 units (flavor invariance for valid text).
|
||||||
|
/// Qwen additionally normalizes NFC, so non-u8 flavors exercise the
|
||||||
|
/// codepoint-level normalization path (NFD extras below).
|
||||||
|
#[test]
|
||||||
|
fn qwen3_utf16_utf32_parity() {
|
||||||
|
let raw = include_str!("../../../fixtures/qwen3.json");
|
||||||
|
let fixture: serde_json::Value = serde_json::from_str(raw).expect("fixture parses");
|
||||||
|
let cases = fixture["cases"].as_array().expect("cases array");
|
||||||
|
// NFD text (normalization must fire in every flavor), astral CJK,
|
||||||
|
// ZWJ emoji, and single-digit runs.
|
||||||
|
let extra = "cafe\u{301} A\u{30a}ngstro\u{308}m \u{1112}\u{1161}\u{11ab} 𝕏𠀀中文👨👩👧👦 12345";
|
||||||
|
let texts = cases
|
||||||
|
.iter()
|
||||||
|
.map(|c| c["text"].as_str().expect("text"))
|
||||||
|
.chain(std::iter::once(extra));
|
||||||
|
for text in texts {
|
||||||
|
let want = Encoding::Qwen3.encode(text).expect("qwen is a BPE family");
|
||||||
|
let u16s: Vec<u16> = text.encode_utf16().collect();
|
||||||
|
let u32s: Vec<u32> = text.chars().map(|c| c as u32).collect();
|
||||||
|
assert_eq!(
|
||||||
|
Encoding::Qwen3.encode(u16s.as_slice()),
|
||||||
|
Some(want.clone()),
|
||||||
|
"utf16 encode mismatch on {text:?}"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
Encoding::Qwen3.encode(u32s.as_slice()),
|
||||||
|
Some(want.clone()),
|
||||||
|
"utf32 encode mismatch on {text:?}"
|
||||||
|
);
|
||||||
|
let n = want.len() as u32;
|
||||||
|
assert_eq!(Encoding::Qwen3.count(u16s.as_slice()), n, "utf16 count mismatch on {text:?}");
|
||||||
|
assert_eq!(Encoding::Qwen3.count(u32s.as_slice()), n, "utf32 count mismatch on {text:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Merge-unreachable vocab entries (dead ranks) must never be emitted:
|
||||||
|
/// the pack blanks them, so whole-piece short-circuits cannot resolve to
|
||||||
|
/// them. Guarded by fixtures too; this pins the ids explicitly.
|
||||||
|
#[test]
|
||||||
|
fn qwen3_dead_ranks_not_emitted() {
|
||||||
|
// "毛泽东" is vocab id 105115 but unreachable via merges; HF emits
|
||||||
|
// [97008, 98340, 96265].
|
||||||
|
let got = Encoding::Qwen3
|
||||||
|
.encode("毛泽东")
|
||||||
|
.expect("qwen is a BPE family");
|
||||||
|
assert_eq!(got, vec![97008, 98340, 96265]);
|
||||||
|
assert!(!got.contains(&105115));
|
||||||
|
}
|
||||||
@@ -0,0 +1,227 @@
|
|||||||
|
//! Encoding-generic text input: UTF-8 / UTF-16 / UTF-32, xutf-style.
|
||||||
|
//!
|
||||||
|
//! No transcoding, no scratch buffers. The pipeline runs natively in the
|
||||||
|
//! input's own code units: the pre-tokenizer scans a codepoint cursor over
|
||||||
|
//! `&[U]`, and the BPE stage looks ranks up in a lazily-expanded per-flavor
|
||||||
|
//! table view (see `bpe.rs`). A JS UTF-16 string is tokenized directly.
|
||||||
|
//!
|
||||||
|
//! Decoding is permissive (xutf semantics): malformed sequences and lone
|
||||||
|
//! surrogates decode as U+FFFD and consume minimally. Valid text behaves
|
||||||
|
//! identically across flavors, so counts/ids are flavor-invariant.
|
||||||
|
|
||||||
|
use std::hash::Hash;
|
||||||
|
|
||||||
|
/// One code unit: `u8` (UTF-8), `u16` (UTF-16 native-endian), `u32` (UTF-32).
|
||||||
|
pub trait Unit: Copy + Eq + Ord + Hash + 'static {
|
||||||
|
/// Decode the codepoint starting at `units[i]`.
|
||||||
|
/// Returns `(codepoint, units_consumed)`; permissive on malformed input.
|
||||||
|
fn decode(units: &[Self], i: usize) -> (char, usize);
|
||||||
|
|
||||||
|
/// Encode `cp` into `out`, returning the unit count written.
|
||||||
|
/// `out` must have room for 4 units.
|
||||||
|
fn encode(cp: char, out: &mut [Self]) -> usize;
|
||||||
|
|
||||||
|
/// Identity byte view when this flavor already is UTF-8 (`u8` only).
|
||||||
|
/// Lets the engine skip per-piece re-encoding for `str` input.
|
||||||
|
fn as_utf8(units: &[Self]) -> Option<&[u8]>;
|
||||||
|
|
||||||
|
/// The unit as an ASCII byte when it encodes one (`< 0x80`).
|
||||||
|
fn ascii(self) -> Option<u8>;
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Unit for u8 {
|
||||||
|
#[inline]
|
||||||
|
fn decode(units: &[Self], i: usize) -> (char, usize) {
|
||||||
|
let b = units[i];
|
||||||
|
if b < 0x80 {
|
||||||
|
return (b as char, 1);
|
||||||
|
}
|
||||||
|
// Permissive multi-byte decode: on malformed input yield U+FFFD and
|
||||||
|
// consume one unit.
|
||||||
|
let need = match b {
|
||||||
|
0xc0..=0xdf => 2,
|
||||||
|
0xe0..=0xef => 3,
|
||||||
|
0xf0..=0xf7 => 4,
|
||||||
|
_ => return (char::REPLACEMENT_CHARACTER, 1),
|
||||||
|
};
|
||||||
|
if i + need > units.len() {
|
||||||
|
return (char::REPLACEMENT_CHARACTER, 1);
|
||||||
|
}
|
||||||
|
let mut cp = (b as u32) & (0x7f >> need);
|
||||||
|
for k in 1..need {
|
||||||
|
let c = units[i + k];
|
||||||
|
if c & 0xc0 != 0x80 {
|
||||||
|
return (char::REPLACEMENT_CHARACTER, 1);
|
||||||
|
}
|
||||||
|
cp = cp << 6 | (c & 0x3f) as u32;
|
||||||
|
}
|
||||||
|
match char::from_u32(cp) {
|
||||||
|
Some(c) => (c, need),
|
||||||
|
None => (char::REPLACEMENT_CHARACTER, need),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn encode(cp: char, out: &mut [Self]) -> usize {
|
||||||
|
cp.encode_utf8(out).len()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn as_utf8(units: &[Self]) -> Option<&[u8]> {
|
||||||
|
Some(units)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn ascii(self) -> Option<u8> {
|
||||||
|
(self < 0x80).then_some(self)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Unit for u16 {
|
||||||
|
#[inline]
|
||||||
|
fn decode(units: &[Self], i: usize) -> (char, usize) {
|
||||||
|
let u = units[i];
|
||||||
|
if !(0xd800..=0xdfff).contains(&u) {
|
||||||
|
// SAFETY-free: non-surrogate u16 is always a valid scalar.
|
||||||
|
return (char::from_u32(u as u32).unwrap_or(char::REPLACEMENT_CHARACTER), 1);
|
||||||
|
}
|
||||||
|
if u < 0xdc00
|
||||||
|
&& let Some(&lo) = units.get(i + 1)
|
||||||
|
&& (0xdc00..=0xdfff).contains(&lo)
|
||||||
|
{
|
||||||
|
let cp = 0x10000 + (((u as u32 - 0xd800) << 10) | (lo as u32 - 0xdc00));
|
||||||
|
return (char::from_u32(cp).unwrap_or(char::REPLACEMENT_CHARACTER), 2);
|
||||||
|
}
|
||||||
|
(char::REPLACEMENT_CHARACTER, 1) // lone surrogate
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn encode(cp: char, out: &mut [Self]) -> usize {
|
||||||
|
cp.encode_utf16(out).len()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn as_utf8(_units: &[Self]) -> Option<&[u8]> {
|
||||||
|
None
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn ascii(self) -> Option<u8> {
|
||||||
|
(self < 0x80).then_some(self as u8)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Unit for u32 {
|
||||||
|
#[inline]
|
||||||
|
fn decode(units: &[Self], i: usize) -> (char, usize) {
|
||||||
|
(char::from_u32(units[i]).unwrap_or(char::REPLACEMENT_CHARACTER), 1)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn encode(cp: char, out: &mut [Self]) -> usize {
|
||||||
|
out[0] = cp as u32;
|
||||||
|
1
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn as_utf8(_units: &[Self]) -> Option<&[u8]> {
|
||||||
|
None
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn ascii(self) -> Option<u8> {
|
||||||
|
(self < 0x80).then_some(self as u8)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Borrowable text in any flavor. Public entry type for
|
||||||
|
/// [`Encoding::count`](crate::utok::Encoding::count) / `encode`.
|
||||||
|
pub trait Utf {
|
||||||
|
type Unit: Unit;
|
||||||
|
fn units(&self) -> &[Self::Unit];
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Utf for str {
|
||||||
|
type Unit = u8;
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn units(&self) -> &[u8] {
|
||||||
|
self.as_bytes()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Utf for String {
|
||||||
|
type Unit = u8;
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn units(&self) -> &[u8] {
|
||||||
|
self.as_bytes()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Utf for [u16] {
|
||||||
|
type Unit = u16;
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn units(&self) -> &[u16] {
|
||||||
|
self
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Utf for Vec<u16> {
|
||||||
|
type Unit = u16;
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn units(&self) -> &[u16] {
|
||||||
|
self
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Utf for [u32] {
|
||||||
|
type Unit = u32;
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn units(&self) -> &[u32] {
|
||||||
|
self
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Utf for Vec<u32> {
|
||||||
|
type Unit = u32;
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
fn units(&self) -> &[u32] {
|
||||||
|
self
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Codepoint cursor over units — the pre-tokenizer's scan primitive.
|
||||||
|
pub struct Cursor<'a, U: Unit> {
|
||||||
|
pub units: &'a [U],
|
||||||
|
pub pos: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'a, U: Unit> Cursor<'a, U> {
|
||||||
|
#[inline]
|
||||||
|
pub fn new(units: &'a [U]) -> Self {
|
||||||
|
Self { units, pos: 0 }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Codepoint at the cursor without advancing.
|
||||||
|
#[inline]
|
||||||
|
pub fn peek(&self) -> Option<(char, usize)> {
|
||||||
|
(self.pos < self.units.len()).then(|| U::decode(self.units, self.pos))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Codepoint after `(cp, len)` from `peek` (one-codepoint lookahead).
|
||||||
|
#[inline]
|
||||||
|
pub fn peek2(&self, first_len: usize) -> Option<(char, usize)> {
|
||||||
|
let j = self.pos + first_len;
|
||||||
|
(j < self.units.len()).then(|| U::decode(self.units, j))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline]
|
||||||
|
pub fn advance(&mut self, n: usize) {
|
||||||
|
self.pos += n;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
# Downloaded raw tokenizer sources; generated data/*.bin.zst is committed.
|
||||||
|
cache/
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
// End-to-end N-API token-count throughput probe.
|
||||||
|
//
|
||||||
|
// Build the host addon first, then run from crates/pi-natives:
|
||||||
|
// bun --cwd ../../packages/natives run build
|
||||||
|
// bun tools/bench-natives.ts
|
||||||
|
|
||||||
|
import { countTokens, Encoding } from "../../../packages/natives/native/index.js";
|
||||||
|
|
||||||
|
const WINDOW_MS = 300;
|
||||||
|
|
||||||
|
const CASES: Record<string, string> = {
|
||||||
|
english: "The quick brown fox jumps over the lazy dog. It is a small corpus for tokenizer throughput. ".repeat(512),
|
||||||
|
code: "function count<T>(items: readonly T[]): number { return items.length; }\n".repeat(2_000),
|
||||||
|
cjk: "东京は日本の首都であり、世界で最も人口の多い都市圏の一つです。深度求索发布了新一代基座模型。".repeat(500),
|
||||||
|
};
|
||||||
|
const ENCODINGS = [
|
||||||
|
Encoding.O200kBase,
|
||||||
|
Encoding.Cl100kBase,
|
||||||
|
Encoding.ClaudeV3,
|
||||||
|
Encoding.ClaudeV47,
|
||||||
|
Encoding.ClaudeV5,
|
||||||
|
Encoding.ClaudeV5Sonnet,
|
||||||
|
Encoding.Qwen3,
|
||||||
|
Encoding.DeepSeekV3,
|
||||||
|
Encoding.KimiK2,
|
||||||
|
Encoding.Glm5,
|
||||||
|
];
|
||||||
|
|
||||||
|
console.log("pi-natives countTokens (JS string → UTF-16 → native)");
|
||||||
|
for (const encoding of ENCODINGS) {
|
||||||
|
for (const name in CASES) {
|
||||||
|
const text = CASES[name];
|
||||||
|
const tokens = countTokens(text, encoding);
|
||||||
|
countTokens(text, encoding); // Warm the lazy table.
|
||||||
|
const start = Bun.nanoseconds();
|
||||||
|
let runs = 0;
|
||||||
|
while ((Bun.nanoseconds() - start) / 1e6 < WINDOW_MS) {
|
||||||
|
countTokens(text, encoding);
|
||||||
|
runs++;
|
||||||
|
}
|
||||||
|
const seconds = (Bun.nanoseconds() - start) / 1e9;
|
||||||
|
const megabytesPerSecond = (new TextEncoder().encode(text).byteLength * runs) / seconds / 1e6;
|
||||||
|
console.log(encoding.padEnd(20) + name.padStart(10) + megabytesPerSecond.toFixed(1).padStart(12) + ` ${tokens}`);
|
||||||
|
}
|
||||||
|
}
|
||||||
+100256
File diff suppressed because it is too large
Load Diff
Vendored
+263174
File diff suppressed because it is too large
Load Diff
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+163584
File diff suppressed because it is too large
Load Diff
+199998
File diff suppressed because it is too large
Load Diff
File diff suppressed because one or more lines are too long
+408
@@ -0,0 +1,408 @@
|
|||||||
|
import os
|
||||||
|
from logging import getLogger
|
||||||
|
from pathlib import Path
|
||||||
|
from shutil import copyfile
|
||||||
|
from typing import Dict, Iterator, List, Optional, Tuple, Union, cast
|
||||||
|
|
||||||
|
import tiktoken
|
||||||
|
from tiktoken.load import load_tiktoken_bpe
|
||||||
|
from tokenizers import AddedToken
|
||||||
|
from transformers.convert_slow_tokenizer import bytes_to_unicode
|
||||||
|
from transformers.tokenization_utils import PreTrainedTokenizer
|
||||||
|
|
||||||
|
try:
|
||||||
|
from .encoding_k3 import build_chat_segments, is_batched_conversation
|
||||||
|
except ImportError: # pragma: no cover - supports direct file execution/import.
|
||||||
|
from encoding_k3 import build_chat_segments, is_batched_conversation
|
||||||
|
|
||||||
|
logger = getLogger(__name__)
|
||||||
|
VOCAB_FILES_NAMES = {"vocab_file": "tiktoken.model"}
|
||||||
|
|
||||||
|
|
||||||
|
class TikTokenTokenizer(PreTrainedTokenizer):
|
||||||
|
"""
|
||||||
|
Tokenizing and encoding/decoding text using the Tiktoken tokenizer. See megatron/tokenizer/tiktoken_tokenizer.py.
|
||||||
|
|
||||||
|
This tokenizer inherits from [`PreTrainedTokenizer`] which contains most of the main methods. Users should refer to
|
||||||
|
this superclass for more information regarding those methods.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
vocab_file (`str`):
|
||||||
|
The path to the Tiktoken model file.
|
||||||
|
bos_token (`str` or `tokenizers.AddedToken`, *optional*, defaults to `"<|begin_of_text|>",`):
|
||||||
|
The beginning of sequence token that was used during pretraining. Can be used a sequence classifier token.
|
||||||
|
eos_token (`str` or `tokenizers.AddedToken`, *optional*, defaults to `"<|end_of_text|>"`):
|
||||||
|
The end of sequence token.
|
||||||
|
unk_token (`str` or `tokenizers.AddedToken`, *optional*, defaults to `"<|reserved_special_token_249|>"`):
|
||||||
|
The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
|
||||||
|
token instead. The second to last item in special_tokens.
|
||||||
|
pad_token (`str` or `tokenizers.AddedToken`, *optional*, defaults to `"<|reserved_special_token_250|>"`):
|
||||||
|
The token used for padding, for example when batching sequences of different lengths.
|
||||||
|
additional_special_tokens (list of `str`, *optional*):
|
||||||
|
A tuple or a list of additional tokens, which will be marked as `special`, meaning that they will be
|
||||||
|
skipped when decoding if `skip_special_tokens` is set to `True`.
|
||||||
|
"""
|
||||||
|
|
||||||
|
vocab_files_names = VOCAB_FILES_NAMES
|
||||||
|
|
||||||
|
model_input_names = ["input_ids", "attention_mask"]
|
||||||
|
|
||||||
|
special_tokens: Dict[str, int]
|
||||||
|
|
||||||
|
num_reserved_special_tokens = 256
|
||||||
|
|
||||||
|
pat_str = "|".join([
|
||||||
|
r"""[\p{Han}]+""",
|
||||||
|
r"""[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]*[\p{Ll}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]+(?i:'s|'t|'re|'ve|'m|'ll|'d)?""",
|
||||||
|
r"""[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]+[\p{Ll}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]*(?i:'s|'t|'re|'ve|'m|'ll|'d)?""",
|
||||||
|
r"""\p{N}{1,3}""",
|
||||||
|
r""" ?[^\s\p{L}\p{N}]+[\r\n]*""",
|
||||||
|
r"""\s*[\r\n]+""",
|
||||||
|
r"""\s+(?!\S)""",
|
||||||
|
r"""\s+""",
|
||||||
|
])
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
vocab_file,
|
||||||
|
bos_token: Union[str, AddedToken] = "[BOS]",
|
||||||
|
eos_token: Union[str, AddedToken] = "[EOS]",
|
||||||
|
unk_token: Union[str, AddedToken, None] = None,
|
||||||
|
pad_token: Union[str, AddedToken, None] = None,
|
||||||
|
additional_special_tokens: List[str] = None,
|
||||||
|
added_tokens_decoder: Optional[dict] = None,
|
||||||
|
**kwargs,
|
||||||
|
):
|
||||||
|
assert os.path.isfile(vocab_file), vocab_file
|
||||||
|
|
||||||
|
if additional_special_tokens is None:
|
||||||
|
additional_special_tokens = [
|
||||||
|
"<|im_end|>",
|
||||||
|
"<|im_user|>",
|
||||||
|
"<|im_assistant|>",
|
||||||
|
"<|start_header_id|>",
|
||||||
|
"<|end_header_id|>",
|
||||||
|
"[EOT]",
|
||||||
|
"<|im_system|>",
|
||||||
|
"<|im_middle|>",
|
||||||
|
]
|
||||||
|
|
||||||
|
if added_tokens_decoder:
|
||||||
|
special_tokens_mapping = {
|
||||||
|
i: added_tokens_decoder[i].content
|
||||||
|
for i in added_tokens_decoder
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
special_tokens_mapping = {}
|
||||||
|
|
||||||
|
self.vocab_file = vocab_file
|
||||||
|
mergeable_ranks = load_tiktoken_bpe(vocab_file)
|
||||||
|
num_base_tokens = len(mergeable_ranks)
|
||||||
|
self.special_tokens = {
|
||||||
|
special_tokens_mapping.get(i, f"<|reserved_token_{i}|>"): i
|
||||||
|
for i in range(num_base_tokens, num_base_tokens +
|
||||||
|
self.num_reserved_special_tokens)
|
||||||
|
}
|
||||||
|
|
||||||
|
self.model = tiktoken.Encoding(
|
||||||
|
name=Path(vocab_file).name,
|
||||||
|
pat_str=self.pat_str,
|
||||||
|
mergeable_ranks=mergeable_ranks,
|
||||||
|
special_tokens=self.special_tokens,
|
||||||
|
)
|
||||||
|
logger.info(f"Reloaded tiktoken model from {vocab_file}")
|
||||||
|
|
||||||
|
self.n_words: int = self.model.n_vocab
|
||||||
|
# BOS / EOS token IDs
|
||||||
|
self.bos_id: int = self.special_tokens[str(bos_token)]
|
||||||
|
self.eos_id: int = self.special_tokens[str(eos_token)]
|
||||||
|
logger.info(
|
||||||
|
f"#words: {self.n_words} - BOS ID: {self.bos_id} - EOS ID: {self.eos_id}"
|
||||||
|
)
|
||||||
|
|
||||||
|
self.pad_id: int = self.special_tokens[str(pad_token)]
|
||||||
|
self.unk_id: int = self.special_tokens[str(unk_token)]
|
||||||
|
|
||||||
|
self.byte_encoder = bytes_to_unicode()
|
||||||
|
self.byte_decoder = {v: k for k, v in self.byte_encoder.items()}
|
||||||
|
|
||||||
|
self.decoder = {}
|
||||||
|
for i in range(self.n_words):
|
||||||
|
# Taken from https://gist.github.com/xenova/a452a6474428de0182b17605a98631ee
|
||||||
|
decoding = ''.join([
|
||||||
|
self.byte_encoder[ord(char)] for char in
|
||||||
|
self.model.decode_single_token_bytes(i).decode('latin-1')
|
||||||
|
])
|
||||||
|
self.decoder[i] = decoding
|
||||||
|
|
||||||
|
self.encoder = {}
|
||||||
|
for i in range(self.n_words):
|
||||||
|
if i in self.decoder:
|
||||||
|
self.encoder[self.decoder[i]] = i
|
||||||
|
|
||||||
|
super().__init__(
|
||||||
|
bos_token=bos_token,
|
||||||
|
eos_token=eos_token,
|
||||||
|
unk_token=unk_token,
|
||||||
|
pad_token=pad_token,
|
||||||
|
additional_special_tokens=additional_special_tokens,
|
||||||
|
added_tokens_decoder=added_tokens_decoder,
|
||||||
|
**kwargs,
|
||||||
|
)
|
||||||
|
self.all_special_ids_set = set(self.all_special_ids)
|
||||||
|
|
||||||
|
def _encode_text_piece(self, text: str,
|
||||||
|
allow_special_tokens: bool = True) -> List[int]:
|
||||||
|
# The tiktoken tokenizer can handle <=400k chars without
|
||||||
|
# pyo3_runtime.PanicException.
|
||||||
|
TIKTOKEN_MAX_ENCODE_CHARS = 400_000
|
||||||
|
|
||||||
|
# https://github.com/openai/tiktoken/issues/195
|
||||||
|
# Here we iterate over subsequences and split if we exceed the limit
|
||||||
|
# of max consecutive non-whitespace or whitespace characters.
|
||||||
|
MAX_NO_WHITESPACES_CHARS = 25_000
|
||||||
|
|
||||||
|
t: List[int] = []
|
||||||
|
for i in range(0, len(text), TIKTOKEN_MAX_ENCODE_CHARS):
|
||||||
|
for substr in self._split_whitespaces_or_nonwhitespaces(
|
||||||
|
text[i:i + TIKTOKEN_MAX_ENCODE_CHARS],
|
||||||
|
MAX_NO_WHITESPACES_CHARS,
|
||||||
|
):
|
||||||
|
if allow_special_tokens:
|
||||||
|
t.extend(
|
||||||
|
# structural markers: encode <|...|> as their special token IDs
|
||||||
|
self.model.encode(
|
||||||
|
substr,
|
||||||
|
allowed_special="all",
|
||||||
|
))
|
||||||
|
else:
|
||||||
|
t.extend(
|
||||||
|
# user/tool text: encode any <|...|> as ordinary BPE tokens (never as control tokens)
|
||||||
|
self.model.encode(
|
||||||
|
substr,
|
||||||
|
disallowed_special=(),
|
||||||
|
))
|
||||||
|
|
||||||
|
return t
|
||||||
|
|
||||||
|
def encode(self,
|
||||||
|
text: str,
|
||||||
|
allow_special_tokens: bool = True,
|
||||||
|
**kwargs) -> List[int]:
|
||||||
|
"""
|
||||||
|
Encodes a string into a list of token IDs.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
text (str): The input string to be encoded.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
list[int]: A list of token IDs.
|
||||||
|
"""
|
||||||
|
# If there are other args, we should call super().encode because there are a lot of code
|
||||||
|
# to handle those args. supper().encode finally will call _tokenize and _convert_token_to_id.
|
||||||
|
# NOTE: our encode method is not compatible with the super().encode method,
|
||||||
|
# e.g. split_special_tokens' default is True in our encode method.
|
||||||
|
if len(kwargs) > 0:
|
||||||
|
logger.warning(f"Calling super().encode with {kwargs}")
|
||||||
|
return super().encode(text, **kwargs)
|
||||||
|
|
||||||
|
assert type(text) is str
|
||||||
|
return self._encode_text_piece(text,
|
||||||
|
allow_special_tokens=allow_special_tokens)
|
||||||
|
|
||||||
|
def decode(self, token_ids: Union[int, List[int]], **kwargs) -> str:
|
||||||
|
"""
|
||||||
|
Decodes a list of token IDs into a string.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
token_ids (List[int]): The list of token IDs to be decoded.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: The decoded string.
|
||||||
|
"""
|
||||||
|
# If there are other args, we should call super().decode because there are a lot of code
|
||||||
|
# to handle those args. supper().encode finally will call convert_tokens_to_string and _convert_id_to_token.
|
||||||
|
if len(kwargs) > 0:
|
||||||
|
return super().decode(token_ids, **kwargs)
|
||||||
|
|
||||||
|
if type(token_ids) is int:
|
||||||
|
token_ids = [token_ids]
|
||||||
|
|
||||||
|
return self.model.decode(cast(List[int], token_ids))
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _split_whitespaces_or_nonwhitespaces(
|
||||||
|
s: str, max_consecutive_slice_len: int) -> Iterator[str]:
|
||||||
|
"""
|
||||||
|
Splits the string `s` so that each substring contains no more than `max_consecutive_slice_len`
|
||||||
|
consecutive whitespaces or consecutive non-whitespaces.
|
||||||
|
"""
|
||||||
|
current_slice_len = 0
|
||||||
|
current_slice_is_space = s[0].isspace() if len(s) > 0 else False
|
||||||
|
slice_start = 0
|
||||||
|
|
||||||
|
for i in range(len(s)):
|
||||||
|
is_now_space = s[i].isspace()
|
||||||
|
|
||||||
|
if current_slice_is_space ^ is_now_space:
|
||||||
|
current_slice_len = 1
|
||||||
|
current_slice_is_space = is_now_space
|
||||||
|
else:
|
||||||
|
current_slice_len += 1
|
||||||
|
if current_slice_len > max_consecutive_slice_len:
|
||||||
|
yield s[slice_start:i]
|
||||||
|
slice_start = i
|
||||||
|
current_slice_len = 1
|
||||||
|
yield s[slice_start:]
|
||||||
|
|
||||||
|
def _encode_chat_segments(self, segments) -> List[int]:
|
||||||
|
token_ids: List[int] = []
|
||||||
|
for segment in segments:
|
||||||
|
token_ids.extend(
|
||||||
|
self._encode_text_piece(
|
||||||
|
segment.text,
|
||||||
|
allow_special_tokens=segment.allow_special,
|
||||||
|
))
|
||||||
|
return token_ids
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _truncate(ids: List[int],
|
||||||
|
truncation: bool = False,
|
||||||
|
max_length: Optional[int] = None) -> List[int]:
|
||||||
|
if truncation and max_length is not None:
|
||||||
|
return ids[:max_length]
|
||||||
|
return ids
|
||||||
|
|
||||||
|
def _format_chat_token_output(self,
|
||||||
|
encoded_inputs: List[List[int]],
|
||||||
|
*,
|
||||||
|
is_batched: bool,
|
||||||
|
padding=False,
|
||||||
|
truncation: bool = False,
|
||||||
|
max_length: Optional[int] = None,
|
||||||
|
return_tensors=None,
|
||||||
|
return_dict: bool = False):
|
||||||
|
encoded_inputs = [
|
||||||
|
self._truncate(ids, truncation=truncation, max_length=max_length)
|
||||||
|
for ids in encoded_inputs
|
||||||
|
]
|
||||||
|
|
||||||
|
needs_batch_encoding = (
|
||||||
|
is_batched or padding or return_tensors is not None or return_dict)
|
||||||
|
if not needs_batch_encoding:
|
||||||
|
return encoded_inputs[0]
|
||||||
|
|
||||||
|
features = [{
|
||||||
|
"input_ids": ids,
|
||||||
|
"attention_mask": [1] * len(ids)
|
||||||
|
} for ids in encoded_inputs]
|
||||||
|
batch = self.pad(features,
|
||||||
|
padding=padding,
|
||||||
|
max_length=max_length if padding else None,
|
||||||
|
return_attention_mask=True,
|
||||||
|
return_tensors=return_tensors)
|
||||||
|
|
||||||
|
if return_dict:
|
||||||
|
return batch
|
||||||
|
if is_batched:
|
||||||
|
return batch["input_ids"]
|
||||||
|
return batch["input_ids"][0] if return_tensors is None else batch[
|
||||||
|
"input_ids"]
|
||||||
|
|
||||||
|
""" ----- Below are the abstract methods required by PreTrainedTokenizer ----- """
|
||||||
|
|
||||||
|
@property
|
||||||
|
def vocab_size(self) -> int:
|
||||||
|
return self.n_words
|
||||||
|
|
||||||
|
def get_vocab(self) -> Dict[str, int]:
|
||||||
|
return self.encoder
|
||||||
|
|
||||||
|
def _tokenize(self, text: str, **kwargs) -> List[str]:
|
||||||
|
return [self.decoder[t] for t in self.encode(text)]
|
||||||
|
|
||||||
|
def _convert_token_to_id(self, token: str) -> int:
|
||||||
|
return self.encoder.get(token, self.unk_id)
|
||||||
|
|
||||||
|
def _convert_id_to_token(self, index: int) -> str:
|
||||||
|
return self.decoder.get(index)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def clean_up_tokenization(out_string: str) -> str:
|
||||||
|
return out_string
|
||||||
|
|
||||||
|
def convert_tokens_to_string(self, tokens: List[str]) -> str:
|
||||||
|
text = ''.join(tokens)
|
||||||
|
text = bytearray([self.byte_decoder[c]
|
||||||
|
for c in text]).decode('utf-8', 'replace')
|
||||||
|
return text
|
||||||
|
|
||||||
|
def save_vocabulary(self,
|
||||||
|
save_directory: str,
|
||||||
|
filename_prefix: Optional[str] = None) -> Tuple[str]:
|
||||||
|
if not os.path.isdir(save_directory):
|
||||||
|
raise ValueError(
|
||||||
|
f"vocabulary path ({save_directory}) should be a directory")
|
||||||
|
out_vocab_file = os.path.join(
|
||||||
|
save_directory,
|
||||||
|
(filename_prefix + "-" if filename_prefix else "") +
|
||||||
|
VOCAB_FILES_NAMES["vocab_file"])
|
||||||
|
|
||||||
|
if os.path.abspath(self.vocab_file) != os.path.abspath(
|
||||||
|
out_vocab_file) and os.path.isfile(self.vocab_file):
|
||||||
|
copyfile(self.vocab_file, out_vocab_file)
|
||||||
|
|
||||||
|
return (out_vocab_file, )
|
||||||
|
|
||||||
|
def apply_chat_template(self,
|
||||||
|
conversation,
|
||||||
|
tools: Optional[list[dict]] = None,
|
||||||
|
tokenize: bool = False,
|
||||||
|
add_generation_prompt: bool = True,
|
||||||
|
thinking: bool = True,
|
||||||
|
padding=False,
|
||||||
|
truncation: bool = False,
|
||||||
|
max_length: Optional[int] = None,
|
||||||
|
return_tensors=None,
|
||||||
|
return_dict: bool = False,
|
||||||
|
**kwargs):
|
||||||
|
# Tokenizer-level rendering reorders tool result messages to match
|
||||||
|
# assistant tool_calls, normalizes per-call arguments and response
|
||||||
|
# schema, then encodes the resulting XTML structure segment-by-segment.
|
||||||
|
is_batched = is_batched_conversation(conversation)
|
||||||
|
conversations = conversation if is_batched else [conversation]
|
||||||
|
image_prompts = kwargs.pop("image_prompts", None)
|
||||||
|
if is_batched and image_prompts is not None:
|
||||||
|
raise ValueError("image_prompts is only supported for one chat.")
|
||||||
|
|
||||||
|
# by default set thinking effort to max
|
||||||
|
kwargs.setdefault("thinking_effort", "max")
|
||||||
|
|
||||||
|
segment_batches = [
|
||||||
|
build_chat_segments(
|
||||||
|
messages,
|
||||||
|
tools=tools,
|
||||||
|
add_generation_prompt=add_generation_prompt,
|
||||||
|
thinking=thinking,
|
||||||
|
image_prompts=image_prompts,
|
||||||
|
**kwargs,
|
||||||
|
) for messages in conversations
|
||||||
|
]
|
||||||
|
|
||||||
|
if not tokenize:
|
||||||
|
rendered = ["".join(segment.text for segment in segments)
|
||||||
|
for segments in segment_batches]
|
||||||
|
return rendered if is_batched else rendered[0]
|
||||||
|
|
||||||
|
encoded_inputs = [
|
||||||
|
self._encode_chat_segments(segments) for segments in segment_batches
|
||||||
|
]
|
||||||
|
return self._format_chat_token_output(
|
||||||
|
encoded_inputs,
|
||||||
|
is_batched=is_batched,
|
||||||
|
padding=padding,
|
||||||
|
truncation=truncation,
|
||||||
|
max_length=max_length,
|
||||||
|
return_tensors=return_tensors,
|
||||||
|
return_dict=return_dict,
|
||||||
|
)
|
||||||
@@ -0,0 +1,74 @@
|
|||||||
|
# One-off: does tiktoken rank-based byte_pair_merge match HF merges-list
|
||||||
|
# BPE for GLM-5 on non-whole-piece inputs? Compares a simulated rank
|
||||||
|
# merge against the reference on adversarial and random pieces.
|
||||||
|
# Usage: uv run --with tokenizers tools/check-glm-rankmerge.py
|
||||||
|
|
||||||
|
import json
|
||||||
|
import random
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from tokenizers import Tokenizer
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parent.parent
|
||||||
|
tok = Tokenizer.from_file(str(ROOT / "tools/cache/glm-5.tokenizer.json"))
|
||||||
|
tj = json.loads((ROOT / "tools/cache/glm-5.tokenizer.json").read_text())
|
||||||
|
|
||||||
|
|
||||||
|
def unicode_to_bytes():
|
||||||
|
bs = list(range(ord("!"), ord("~") + 1)) + list(range(0xA1, 0xAD)) + list(range(0xAE, 0x100))
|
||||||
|
cs = bs[:]
|
||||||
|
n = 0
|
||||||
|
for b in range(256):
|
||||||
|
if b not in bs:
|
||||||
|
bs.append(b)
|
||||||
|
cs.append(256 + n)
|
||||||
|
n += 1
|
||||||
|
return {chr(c): b for c, b in zip(cs, bs)}
|
||||||
|
|
||||||
|
|
||||||
|
INV = unicode_to_bytes()
|
||||||
|
ranks = {}
|
||||||
|
for key, rank in tj["model"]["vocab"].items():
|
||||||
|
ranks[bytes(INV[c] for c in key)] = rank
|
||||||
|
|
||||||
|
|
||||||
|
def rank_encode_piece(piece: bytes) -> list[int]:
|
||||||
|
if piece in ranks:
|
||||||
|
return [ranks[piece]]
|
||||||
|
parts = list(range(len(piece) + 1))
|
||||||
|
def pr(i):
|
||||||
|
if i + 2 >= len(parts):
|
||||||
|
return 1 << 60
|
||||||
|
return ranks.get(piece[parts[i]:parts[i + 2]], 1 << 60)
|
||||||
|
while len(parts) > 2:
|
||||||
|
best, bi = 1 << 60, -1
|
||||||
|
for i in range(len(parts) - 2):
|
||||||
|
r = ranks.get(piece[parts[i]:parts[i + 2]], 1 << 60)
|
||||||
|
if r < best:
|
||||||
|
best, bi = r, i
|
||||||
|
if bi < 0:
|
||||||
|
break
|
||||||
|
del parts[bi + 1]
|
||||||
|
return [ranks[piece[parts[i]:parts[i + 1]]] for i in range(len(parts) - 1)]
|
||||||
|
|
||||||
|
|
||||||
|
# Adversarial: probe tokens extended so the whole piece is OOV.
|
||||||
|
rare = "龘"
|
||||||
|
adversarial = [" 参考" + rare, " 参考资料" + rare, " 而" + rare, " 者" + rare, " 王" + rare,
|
||||||
|
"参考文献列表", " 参考文献综述汇编", rare + " 参考"]
|
||||||
|
random.seed(42)
|
||||||
|
cjk = [chr(c) for c in range(0x4E00, 0x9FFF, 7)]
|
||||||
|
rand = ["".join(random.choices(cjk, k=random.randint(2, 8))) for _ in range(3000)]
|
||||||
|
words = ["Übermensch", "naïveté", "переосмысление", "🎉🎊", "ffiffl", "supercalifragilistic"]
|
||||||
|
|
||||||
|
bad = 0
|
||||||
|
for text in adversarial + rand + words:
|
||||||
|
ref = tok.encode(text, add_special_tokens=False).ids
|
||||||
|
# simulate: pretokenize via the real tokenizer's offsets? use single-piece
|
||||||
|
# texts only (pure CJK/letter runs stay one piece under the GLM regex).
|
||||||
|
sim = rank_encode_piece(text.encode("utf-8"))
|
||||||
|
if sim != ref:
|
||||||
|
bad += 1
|
||||||
|
if bad <= 10:
|
||||||
|
print(f"MISMATCH {text!r}\n ref={ref}\n sim={sim}")
|
||||||
|
print(f"checked {len(adversarial) + len(rand) + len(words)}, mismatches: {bad}")
|
||||||
+11
-10
@@ -1,17 +1,18 @@
|
|||||||
/**
|
/**
|
||||||
* Regenerates the compact ctok vocabulary data embedded by the Rust ctok port
|
* Regenerates compact ctok vocabulary data for `src/utok/claude/`
|
||||||
* (crates/pi-natives/src/ctok/data/ctok_*.bin).
|
* (`tools/cache/ctok_*.bin`).
|
||||||
*
|
*
|
||||||
* Source of truth is the measured vocabulary of sanderland/ctok (MIT), pinned
|
* Source of truth is the measured vocabulary of sanderland/ctok (MIT), pinned
|
||||||
* to a release revision. Compaction drops the per-piece witness metadata,
|
* to a release revision. Compaction drops the per-piece witness metadata,
|
||||||
* parses the public `⟨bow⟩the⟨eow⟩` notation into the single-glyph internal
|
* parses the public `⟨bow⟩the⟨eow⟩` notation into C0 marker bytes, adds the
|
||||||
* form, adds the glued contraction spellings, and emits the front-coded
|
* glued contraction spellings, and emits the version-2 front-coded binary
|
||||||
* binary format below — cutting ~4.7 MB of upstream JSON to ~350 KB of
|
* format below — cutting ~4.7 MB of upstream JSON to ~254 KB of embedded
|
||||||
* embedded data. If the pin moves, also regenerate
|
* data. If the pin moves, also regenerate
|
||||||
* crates/pi-natives/src/ctok/testdata/fixtures.json against the same ctok
|
* `src/utok/claude/testdata/fixtures.json` against the same ctok release
|
||||||
* release (`uv run --with ctok …`; see the fixture doc in ctok/mod.rs).
|
* (`uv run --with ctok …`; see the fixture doc in `src/utok/claude/mod.rs`).
|
||||||
*
|
*
|
||||||
* Format (little-endian; parsed by `VocabCore::parse` in ctok/engine.rs):
|
* Format (little-endian; parsed by `VocabCore::parse` in
|
||||||
|
* `src/utok/claude/engine.rs`):
|
||||||
*
|
*
|
||||||
* magic b"CTOK"
|
* magic b"CTOK"
|
||||||
* version u8 = 2
|
* version u8 = 2
|
||||||
@@ -35,7 +36,7 @@ import * as path from "node:path";
|
|||||||
const CTOK_REV = "df3b59b5e645289a5eadc8e24036b99d39c333c4";
|
const CTOK_REV = "df3b59b5e645289a5eadc8e24036b99d39c333c4";
|
||||||
const UPSTREAM = `https://raw.githubusercontent.com/sanderland/ctok/${CTOK_REV}/ctok/data`;
|
const UPSTREAM = `https://raw.githubusercontent.com/sanderland/ctok/${CTOK_REV}/ctok/data`;
|
||||||
|
|
||||||
const DATA_DIR = path.join(import.meta.dir, "../../../crates/pi-natives/src/ctok/data");
|
const DATA_DIR = path.join(import.meta.dir, "cache");
|
||||||
|
|
||||||
/** Marker glyphs of ctok's internal marked form, keyed by public atom. */
|
/** Marker glyphs of ctok's internal marked form, keyed by public atom. */
|
||||||
const ATOMS: Record<string, string> = {
|
const ATOMS: Record<string, string> = {
|
||||||
@@ -0,0 +1,87 @@
|
|||||||
|
# Generate fixtures/deepseek3.json from the cached HF tokenizer.
|
||||||
|
#
|
||||||
|
# Usage: uv run --with tokenizers tools/gen-deepseek-fixtures.py
|
||||||
|
import json
|
||||||
|
import pathlib
|
||||||
|
|
||||||
|
from tokenizers import Tokenizer
|
||||||
|
|
||||||
|
ROOT = pathlib.Path(__file__).resolve().parent.parent
|
||||||
|
tok = Tokenizer.from_file(str(ROOT / "tools/cache/deepseek-v4.tokenizer.json"))
|
||||||
|
# encode_ordinary semantics: special added tokens present verbatim in the
|
||||||
|
# input must be split as plain text (pure BPE), never emitted as their ids.
|
||||||
|
# Required for the dead-entry probes below; a no-op for every other case.
|
||||||
|
tok.encode_special_tokens = True
|
||||||
|
|
||||||
|
corpus = json.loads((ROOT / "fixtures/corpus.json").read_text())
|
||||||
|
|
||||||
|
edge_cases = [
|
||||||
|
# Split-chain interaction: digits then CJK then latin.
|
||||||
|
"abc123def一二三ghi",
|
||||||
|
# 4+ digit numbers straddle the \p{N}{1,3} boundary.
|
||||||
|
"1234",
|
||||||
|
"12345 678901 3.14159265358979",
|
||||||
|
"2024-08-19T12:34:56.789Z",
|
||||||
|
"10000000 tokens cost $0.00042",
|
||||||
|
"一2三45六789零 第123章 第1234章",
|
||||||
|
# CJK runs incl. hiragana/katakana block edges (-ゟ, ゠-ヿ).
|
||||||
|
"深度求索发布了新一代基座模型,性能大幅提升。",
|
||||||
|
"こんにちは世界!カタカナ・テストです。",
|
||||||
|
"ゟ゠ヿ",
|
||||||
|
"中文English日本語한국어mixed",
|
||||||
|
# Punctuation-prefix-letters alternate ([!"#$%&'()*+,\-./...][A-Za-z]+).
|
||||||
|
".NET",
|
||||||
|
".NET Framework 4.8",
|
||||||
|
"(foo)",
|
||||||
|
"(int)x + (float)y",
|
||||||
|
"#include <stdio.h>",
|
||||||
|
"#pragma once",
|
||||||
|
"[foo]bar {baz}qux",
|
||||||
|
"@user mentioned ~home and $PATH",
|
||||||
|
"a.b.c e.g. i.e. etc.",
|
||||||
|
"'quoted' \"double\" `backtick`",
|
||||||
|
"C++ -O2 --flag=value",
|
||||||
|
"_underscore __dunder__",
|
||||||
|
# Whitespace lookahead \s+(?!\S) edges.
|
||||||
|
"word \nnext ",
|
||||||
|
" leading and trailing ",
|
||||||
|
# Dead-entry probes: the three merge-unreachable sentinels (ids 0..2)
|
||||||
|
# written out verbatim must tokenize as plain text — never as their
|
||||||
|
# ids (they are blanked in the packed table).
|
||||||
|
"<|begin▁of▁sentence|>",
|
||||||
|
"<|end▁of▁sentence|>hello<|▁pad▁|>",
|
||||||
|
]
|
||||||
|
|
||||||
|
texts = corpus + edge_cases
|
||||||
|
cases = []
|
||||||
|
for text in texts:
|
||||||
|
ids = tok.encode(text, add_special_tokens=False).ids
|
||||||
|
assert not any(i < 3 for i in ids), f"sentinel id leaked into reference: {text!r}"
|
||||||
|
cases.append({"text": text, "ids": ids, "count": len(ids)})
|
||||||
|
|
||||||
|
# V3 parity: encode one sample with the actual DeepSeek-V3 tokenizer
|
||||||
|
# (downloaded once into tools/cache/) and assert it matches V4 — the base
|
||||||
|
# BPE is identical across V3..V4.
|
||||||
|
PARITY_TEXT = "DeepSeek V3参数量6710亿, released 2024-12-26. (fn)main一二三"
|
||||||
|
v3_path = ROOT / "tools/cache/deepseek-v3.tokenizer.json"
|
||||||
|
if not v3_path.exists():
|
||||||
|
import urllib.request
|
||||||
|
|
||||||
|
url = "https://huggingface.co/deepseek-ai/DeepSeek-V3/resolve/main/tokenizer.json"
|
||||||
|
with urllib.request.urlopen(url) as resp:
|
||||||
|
v3_path.write_bytes(resp.read())
|
||||||
|
v3 = Tokenizer.from_file(str(v3_path))
|
||||||
|
v3.encode_special_tokens = True
|
||||||
|
v3_ids = v3.encode(PARITY_TEXT, add_special_tokens=False).ids
|
||||||
|
v4_ids = tok.encode(PARITY_TEXT, add_special_tokens=False).ids
|
||||||
|
assert v3_ids == v4_ids, f"V3/V4 drift on parity sample: {v3_ids} != {v4_ids}"
|
||||||
|
|
||||||
|
out = {
|
||||||
|
"generator": "uv run --with tokenizers tools/gen-deepseek-fixtures.py (tokenizers, cache/deepseek-v4.tokenizer.json, add_special_tokens=False, encode_special_tokens=True)",
|
||||||
|
"cases": cases,
|
||||||
|
"v3_parity": {"text": PARITY_TEXT, "ids": v3_ids, "count": len(v3_ids)},
|
||||||
|
}
|
||||||
|
(ROOT / "fixtures/deepseek3.json").write_text(
|
||||||
|
json.dumps(out, ensure_ascii=False, indent=1) + "\n"
|
||||||
|
)
|
||||||
|
print(f"wrote {len(cases)} cases")
|
||||||
@@ -0,0 +1,93 @@
|
|||||||
|
# Generate fixtures/glm5.json from the reference HF tokenizers runtime.
|
||||||
|
# Also hunts ignore_merges divergence probes: vocab tokens whose plain
|
||||||
|
# merge-loop encode (ignore_merges=False) differs from the whole-piece
|
||||||
|
# vocab hit (ignore_merges=True), proving the short-circuit is load-bearing.
|
||||||
|
#
|
||||||
|
# Usage: uv run --with tokenizers tools/gen-glm-fixtures.py
|
||||||
|
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from tokenizers import Tokenizer
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parent.parent
|
||||||
|
TOK_JSON = ROOT / "tools/cache/glm-5.tokenizer.json"
|
||||||
|
|
||||||
|
tok = Tokenizer.from_file(str(TOK_JSON))
|
||||||
|
|
||||||
|
# Variant with ignore_merges disabled, for probe hunting only.
|
||||||
|
tj = json.loads(TOK_JSON.read_text())
|
||||||
|
assert tj["model"]["ignore_merges"] is True
|
||||||
|
tj["model"]["ignore_merges"] = False
|
||||||
|
noim_path = ROOT / "tools/cache/glm-5.no-ignore-merges.json"
|
||||||
|
noim_path.write_text(json.dumps(tj))
|
||||||
|
tok_noim = Tokenizer.from_file(str(noim_path))
|
||||||
|
|
||||||
|
# GPT-2 byte-level alphabet, inverted (unicode char -> byte).
|
||||||
|
def unicode_to_bytes():
|
||||||
|
bs = list(range(ord("!"), ord("~") + 1)) + list(range(0xA1, 0xAD)) + list(range(0xAE, 0x100))
|
||||||
|
cs = bs[:]
|
||||||
|
n = 0
|
||||||
|
for b in range(256):
|
||||||
|
if b not in bs:
|
||||||
|
bs.append(b)
|
||||||
|
cs.append(256 + n)
|
||||||
|
n += 1
|
||||||
|
return {chr(c): b for c, b in zip(cs, bs)}
|
||||||
|
|
||||||
|
INV = unicode_to_bytes()
|
||||||
|
vocab = tj["model"]["vocab"]
|
||||||
|
|
||||||
|
# Hunt probes: multi-char vocab tokens that decode to valid UTF-8 text,
|
||||||
|
# survive pretokenization as a single piece (encode length 1 under
|
||||||
|
# ignore_merges), but merge to something else without the flag.
|
||||||
|
probes = []
|
||||||
|
for key, rank in vocab.items():
|
||||||
|
if len(key) < 2:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
text = bytes(INV[c] for c in key).decode("utf-8")
|
||||||
|
except (KeyError, UnicodeDecodeError):
|
||||||
|
continue
|
||||||
|
ids = tok.encode(text, add_special_tokens=False).ids
|
||||||
|
if ids != [rank]:
|
||||||
|
continue # pretokenizer splits it; not a whole-piece case
|
||||||
|
ids_noim = tok_noim.encode(text, add_special_tokens=False).ids
|
||||||
|
if ids_noim != ids:
|
||||||
|
probes.append((text, rank, ids_noim))
|
||||||
|
if len(probes) >= 5:
|
||||||
|
break
|
||||||
|
|
||||||
|
print(f"ignore_merges probes found: {len(probes)}")
|
||||||
|
for text, rank, noim in probes:
|
||||||
|
print(f" {text!r}: with={rank} without={noim}")
|
||||||
|
assert probes, "no ignore_merges divergence found — short-circuit unproven"
|
||||||
|
|
||||||
|
corpus = json.loads((ROOT / "fixtures/corpus.json").read_text())
|
||||||
|
extra = [
|
||||||
|
# Chinese samples
|
||||||
|
"智谱清言是由北京智谱华章科技有限公司开发的大语言模型。",
|
||||||
|
"你好,世界!这是一个测试。",
|
||||||
|
"人工智能正在改变世界,深度学习模型的参数规模不断增长。",
|
||||||
|
"中英文混排 mixed CJK and English 123 数字。",
|
||||||
|
" 全角空格和标点符号:《引号》、【括号】——破折号……省略号",
|
||||||
|
]
|
||||||
|
# Probes verbatim, plus OOV extensions: the piece is no longer a whole-vocab
|
||||||
|
# hit, so the merge loop must run and still match the reference around the
|
||||||
|
# unreachable substrings.
|
||||||
|
probe_texts = [t for t, _, _ in probes]
|
||||||
|
probe_texts += [t + "龘" for t, _, _ in probes]
|
||||||
|
probe_texts += ["龘" + t for t, _, _ in probes]
|
||||||
|
|
||||||
|
cases = []
|
||||||
|
for text in corpus + extra + probe_texts:
|
||||||
|
enc = tok.encode(text, add_special_tokens=False)
|
||||||
|
cases.append({"text": text, "ids": enc.ids, "count": len(enc.ids)})
|
||||||
|
|
||||||
|
out = {
|
||||||
|
"generator": "uv run --with tokenizers tools/gen-glm-fixtures.py (tokenizers reference, add_special_tokens=False)",
|
||||||
|
"cases": cases,
|
||||||
|
}
|
||||||
|
(ROOT / "fixtures/glm5.json").write_text(json.dumps(out, ensure_ascii=False, indent=1) + "\n")
|
||||||
|
noim_path.unlink()
|
||||||
|
print(f"wrote {len(cases)} cases to fixtures/glm5.json")
|
||||||
@@ -0,0 +1,69 @@
|
|||||||
|
# Generate fixtures/kimi_k2.json with reference tiktoken.
|
||||||
|
# Usage: uv run --with tiktoken --with blobfile tools/gen-kimi-fixtures.py
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import tiktoken
|
||||||
|
from tiktoken.load import load_tiktoken_bpe
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parent.parent
|
||||||
|
|
||||||
|
pat_str = "|".join([
|
||||||
|
r"""[\p{Han}]+""",
|
||||||
|
r"""[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]*[\p{Ll}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]+(?i:'s|'t|'re|'ve|'m|'ll|'d)?""",
|
||||||
|
r"""[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]+[\p{Ll}\p{Lm}\p{Lo}\p{M}&&[^\p{Han}]]*(?i:'s|'t|'re|'ve|'m|'ll|'d)?""",
|
||||||
|
r"""\p{N}{1,3}""",
|
||||||
|
r""" ?[^\s\p{L}\p{N}]+[\r\n]*""",
|
||||||
|
r"""\s*[\r\n]+""",
|
||||||
|
r"""\s+(?!\S)""",
|
||||||
|
r"""\s+""",
|
||||||
|
])
|
||||||
|
|
||||||
|
ranks = load_tiktoken_bpe(str(ROOT / "tools/cache/kimi.tiktoken.model"))
|
||||||
|
assert len(ranks) == 163_584, len(ranks)
|
||||||
|
enc = tiktoken.Encoding(name="kimi", pat_str=pat_str, mergeable_ranks=ranks, special_tokens={})
|
||||||
|
|
||||||
|
corpus = json.loads((ROOT / "fixtures/corpus.json").read_text())
|
||||||
|
|
||||||
|
extra = [
|
||||||
|
# Han-heavy text.
|
||||||
|
"中文分词是自然语言处理的基础任务之一。月之暗面发布了千亿参数模型。",
|
||||||
|
"汉字漢字汉字漢字",
|
||||||
|
# Mixed Han/Latin/digits with Han-adjacent case turns.
|
||||||
|
"中文English中文",
|
||||||
|
"中文english中文ENGLISH中文",
|
||||||
|
"GPT4发布于2023年3月14日,共有1750亿个参数。",
|
||||||
|
"深度学习deep learning模型model需要大量GPU资源,如A100或H100。",
|
||||||
|
"价格是99.99元,折扣为8.5折。",
|
||||||
|
# Han next to apostrophe contractions and case boundaries.
|
||||||
|
"他说:'It's fine'然后离开了。",
|
||||||
|
"中文Word中文WORD中文word",
|
||||||
|
# Kana/Hangul (non-Han CJK) beside Han.
|
||||||
|
"日本語テスト中文한국어",
|
||||||
|
# Han with whitespace runs and newlines.
|
||||||
|
"第一行\n第二行\r\n 第三行\t结束",
|
||||||
|
"中文 English 中文 English 中文",
|
||||||
|
# Long digit runs (\p{N}{1,3} chunking) and non-ASCII digits.
|
||||||
|
"12345678901234567890",
|
||||||
|
"١٢٣٤٥٦٧٨٩٠ ๑๒๓ 一二三",
|
||||||
|
# Marks (\p{M}) adjacent to Han and Latin.
|
||||||
|
"éé中文éé e\u0301\u0301中文",
|
||||||
|
# Punctuation runs absorbing trailing newlines.
|
||||||
|
"foo!!!\n\nbar???\r\n",
|
||||||
|
# Leading-space letter runs and lookahead tail.
|
||||||
|
" trailing spaces ",
|
||||||
|
" 中文 a 中文 A1中文",
|
||||||
|
]
|
||||||
|
|
||||||
|
texts = corpus + extra
|
||||||
|
cases = []
|
||||||
|
for text in texts:
|
||||||
|
ids = enc.encode_ordinary(text)
|
||||||
|
cases.append({"text": text, "ids": ids, "count": len(ids)})
|
||||||
|
|
||||||
|
out = {
|
||||||
|
"generator": f"tiktoken {tiktoken.__version__} Encoding(kimi, tokenization_kimi.py pat_str) encode_ordinary",
|
||||||
|
"cases": cases,
|
||||||
|
}
|
||||||
|
(ROOT / "fixtures/kimi_k2.json").write_text(json.dumps(out, ensure_ascii=False, indent=1) + "\n")
|
||||||
|
print(f"{len(cases)} cases, total {sum(c['count'] for c in cases)} tokens")
|
||||||
@@ -0,0 +1,77 @@
|
|||||||
|
# Generate golden fixtures for o200k_base / cl100k_base with Python tiktoken.
|
||||||
|
#
|
||||||
|
# uv run --with tiktoken python tools/gen-openai-fixtures.py
|
||||||
|
#
|
||||||
|
# Emits fixtures/{o200k_base,cl100k_base}.json:
|
||||||
|
# { "generator": str, "cases": [{ "text", "ids", "count" }] }
|
||||||
|
|
||||||
|
import json
|
||||||
|
import pathlib
|
||||||
|
|
||||||
|
import tiktoken
|
||||||
|
|
||||||
|
root = pathlib.Path(__file__).resolve().parent.parent
|
||||||
|
corpus = json.loads((root / "fixtures" / "corpus.json").read_text())
|
||||||
|
|
||||||
|
edge_cases = [
|
||||||
|
# very long single piece (one letter run stresses the merge loop)
|
||||||
|
"a" * 20000,
|
||||||
|
"z" + "a" * 8191,
|
||||||
|
# all 256 byte-ish codepoints U+0000..U+00FF (valid UTF-8 both sides)
|
||||||
|
"".join(chr(i) for i in range(256)),
|
||||||
|
# contraction casing (cl100k has case-insensitive suffix group up front)
|
||||||
|
"It'S ODD THAT'S y'ALL'VE dOn'T CAN'T won'T",
|
||||||
|
"'s 't 're 've 'm 'll 'd 'S 'T 'RE 'VE 'M 'LL 'D",
|
||||||
|
# digit grouping \p{N}{1,3}
|
||||||
|
"1 12 123 1234 12345 123456 1234567890123456789",
|
||||||
|
"٠١٢٣٤٥٦٧٨٩ ०१२३४५६७८९", # non-ASCII decimal digits
|
||||||
|
# o200k punctuation rule swallows trailing slashes: [\r\n/]*
|
||||||
|
"http:// a//b ///// -/\n\r\n//",
|
||||||
|
"path/to/file.txt // comment /* block */",
|
||||||
|
# whitespace boundary torture for \s+(?!\S) vs \s+
|
||||||
|
"x y",
|
||||||
|
"x \t y ",
|
||||||
|
" \t\u000b\u000c\u00a0\u2028\u2029\u3000tail",
|
||||||
|
"end ",
|
||||||
|
"\n\n\n",
|
||||||
|
"\r\r\r\n\n \n\t\r\n x",
|
||||||
|
# leading-symbol letter runs: [^\r\n\p{L}\p{N}]?\p{L}+
|
||||||
|
"@word #tag $var %pct & *star",
|
||||||
|
"_underscore __dunder__ mixed_Case_Words",
|
||||||
|
# marks and titlecase (o200k [\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}] classes)
|
||||||
|
"Džungla DŽ Dž dž İstanbul ff fi",
|
||||||
|
"e\u0301le\u0300ve a\u0308\u0301 x\u0e48\u0e49",
|
||||||
|
# CJK / mixed scripts
|
||||||
|
"中文English混排テスト한글1234",
|
||||||
|
# emoji + ZWJ + variation selectors
|
||||||
|
"👍🏽👨👩👧👦🇹🇷\ufe0f\u200d",
|
||||||
|
# single chars
|
||||||
|
"a",
|
||||||
|
" ",
|
||||||
|
"\t",
|
||||||
|
"'",
|
||||||
|
"\u00e9",
|
||||||
|
"𝕏",
|
||||||
|
# repeated punctuation runs
|
||||||
|
"!!!???...,,,;;;:::" * 40,
|
||||||
|
"=" * 3000,
|
||||||
|
# long whitespace run (merge loop over space tokens)
|
||||||
|
" " * 5000 + "x",
|
||||||
|
" " * 4097,
|
||||||
|
]
|
||||||
|
|
||||||
|
texts = corpus + edge_cases
|
||||||
|
|
||||||
|
for name in ("o200k_base", "cl100k_base"):
|
||||||
|
enc = tiktoken.get_encoding(name)
|
||||||
|
cases = []
|
||||||
|
for text in texts:
|
||||||
|
ids = enc.encode_ordinary(text)
|
||||||
|
cases.append({"text": text, "ids": ids, "count": len(ids)})
|
||||||
|
out = {
|
||||||
|
"generator": f"python tiktoken {tiktoken.__version__} {name} encode_ordinary",
|
||||||
|
"cases": cases,
|
||||||
|
}
|
||||||
|
path = root / "fixtures" / f"{name}.json"
|
||||||
|
path.write_text(json.dumps(out, ensure_ascii=False, indent=1) + "\n")
|
||||||
|
print(f"{name}: {len(cases)} cases -> {path}")
|
||||||
@@ -0,0 +1,94 @@
|
|||||||
|
# Generate fixtures/qwen3.json from the reference HF tokenizer.
|
||||||
|
# Run: uv run --with tokenizers tools/gen-qwen-fixtures.py
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import unicodedata
|
||||||
|
|
||||||
|
from tokenizers import Tokenizer
|
||||||
|
|
||||||
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||||
|
tok = Tokenizer.from_file(os.path.join(ROOT, "tools/cache/qwen3.8.tokenizer.json"))
|
||||||
|
|
||||||
|
with open(os.path.join(ROOT, "fixtures/corpus.json")) as f:
|
||||||
|
texts = json.load(f)
|
||||||
|
|
||||||
|
# Family-specific edge cases.
|
||||||
|
texts += [
|
||||||
|
# Merge-unreachable vocab entries as exact whole pieces (dead-rank
|
||||||
|
# regression: HF never emits these ids; a naive rank table would).
|
||||||
|
"毛泽东",
|
||||||
|
"俱乐部",
|
||||||
|
"全心全意为人民",
|
||||||
|
"承担一切因您的行为而直接或间接",
|
||||||
|
"足球俱乐部",
|
||||||
|
"材料", # reachable counterpart control
|
||||||
|
# ...embedded mid-text
|
||||||
|
"他研究毛泽东思想,加入足球俱乐部,去过新加坡和加拿大。",
|
||||||
|
"众所周知,勤勤恳恳、兢兢业业,跃跃欲试。",
|
||||||
|
"матри материал експерт",
|
||||||
|
" експерт",
|
||||||
|
"สังหาริมทรัพย์ มิถุนายน",
|
||||||
|
"بسبب الأسبوع سبب",
|
||||||
|
"Selanjutnya masyarakat terdapat",
|
||||||
|
# Chinese-heavy
|
||||||
|
"深度学习模型的训练需要大量的计算资源和高质量的数据集。近年来,随着硬件技术的飞速发展,大规模预训练语言模型在自然语言处理领域取得了突破性进展。",
|
||||||
|
"白日依山尽,黄河入海流。欲穷千里目,更上一层楼。",
|
||||||
|
"中华人民共和国全国人民代表大会常务委员会",
|
||||||
|
"你好,世界!这是一个测试。2024年(全角数字)",
|
||||||
|
# Digit runs: \p{N} is single-digit for Qwen (unlike cl100k's {1,3})
|
||||||
|
"1234567890",
|
||||||
|
"3.14159265358979",
|
||||||
|
"電話番号は0123456789です",
|
||||||
|
"١٢٣٤٥ ௧௨௩ ৪৫৬", # Arabic-Indic, Tamil, Bengali digits
|
||||||
|
"Ⅻ Ⅷ ½ ⅓ ①②③", # Nl / No categories also match \p{N}
|
||||||
|
"42nd 100th x1 x22 x333",
|
||||||
|
# NFC regression: NFD inputs must normalize before splitting
|
||||||
|
unicodedata.normalize("NFD", "naïve café résumé"),
|
||||||
|
unicodedata.normalize("NFD", "한국어 텍스트"),
|
||||||
|
unicodedata.normalize("NFD", "Ångström ế ộ"),
|
||||||
|
"e\u0301\u0301clair", # double combining acute (not fully composable)
|
||||||
|
"\u1e0b\u0323 \u0064\u0323\u0307", # ḋ+dot-below vs d+dot-below+dot-above (NFC reorders)
|
||||||
|
# Contractions with (?i:...)
|
||||||
|
"DON'T I'LL HE'S WE'RE THEY'VE I'M YOU'D",
|
||||||
|
"don't i'll he's we're they've i'm you'd",
|
||||||
|
"Mixed'S cAsE'Ll",
|
||||||
|
"it'\u017f IT'S don'T x'Ll they'RE we'VE i'M you'D", # U+017F long s folds into (?i:'s)
|
||||||
|
"can't've y'all'll've",
|
||||||
|
# Whitespace lookahead \s+(?!\S) boundaries
|
||||||
|
"a b c d",
|
||||||
|
"end ",
|
||||||
|
"tabs\t\t\tthen spaces \n newline",
|
||||||
|
"\n\n\n",
|
||||||
|
" ",
|
||||||
|
# Marks: [^\r\n\p{L}\p{N}]?[\p{L}\p{M}]+ takes leading non-letter
|
||||||
|
"$var _under #tag @user",
|
||||||
|
"«guillemets» “curly” ‘quotes’",
|
||||||
|
"ab\u0301c \u0301x combining", # marks ride letter runs; lone mark after space-prefix
|
||||||
|
"。汉字,测试!Qwen全角fifl",
|
||||||
|
# \s*[\r\n]+ eats through the LAST newline of a whitespace run
|
||||||
|
"x \r\n \n y",
|
||||||
|
"a\r\nb\rc\nd",
|
||||||
|
"para.\n\n Indented after blank.\r\n\r\nEnd",
|
||||||
|
]
|
||||||
|
|
||||||
|
# Dedup, preserve order.
|
||||||
|
seen = set()
|
||||||
|
ordered = []
|
||||||
|
for t in texts:
|
||||||
|
if t not in seen:
|
||||||
|
seen.add(t)
|
||||||
|
ordered.append(t)
|
||||||
|
|
||||||
|
cases = []
|
||||||
|
for text in ordered:
|
||||||
|
ids = tok.encode(text, add_special_tokens=False).ids
|
||||||
|
cases.append({"text": text, "ids": ids, "count": len(ids)})
|
||||||
|
|
||||||
|
out = {
|
||||||
|
"generator": "tools/gen-qwen-fixtures.py: HF tokenizers Tokenizer.from_file(tools/cache/qwen3.8.tokenizer.json).encode(text, add_special_tokens=False)",
|
||||||
|
"cases": cases,
|
||||||
|
}
|
||||||
|
with open(os.path.join(ROOT, "fixtures/qwen3.json"), "w") as f:
|
||||||
|
json.dump(out, f, ensure_ascii=False, indent=1)
|
||||||
|
f.write("\n")
|
||||||
|
print(f"{len(cases)} cases, total {sum(c['count'] for c in cases)} tokens")
|
||||||
Executable
+32
@@ -0,0 +1,32 @@
|
|||||||
|
#!/usr/bin/env bun
|
||||||
|
// Pack the ctok vocabulary blobs: zstd -19 compress the front-coded binaries
|
||||||
|
// produced by tools/gen-ctok-vocab.ts (upstream ctok df3b59b data).
|
||||||
|
//
|
||||||
|
// Sources (first hit wins): $CTOK_SRC, tools/cache/ (gen-ctok-vocab.ts output).
|
||||||
|
// Output: data/ctok_v3.bin.zst, data/ctok_v4_7.bin.zst — consumed by
|
||||||
|
// include_bytes! + zstd::decode_all in src/utok/claude/mod.rs.
|
||||||
|
|
||||||
|
import { existsSync } from "node:fs";
|
||||||
|
import * as path from "node:path";
|
||||||
|
|
||||||
|
const root = path.resolve(import.meta.dir, "..");
|
||||||
|
const candidates = [process.env.CTOK_SRC, path.join(root, "tools/cache")].filter(
|
||||||
|
(d): d is string => !!d,
|
||||||
|
);
|
||||||
|
|
||||||
|
const MAGIC = "CTOK"; // container magic written by gen-ctok-vocab.ts
|
||||||
|
const VERSION = 2; // format version byte (compact C0 marker alphabet)
|
||||||
|
|
||||||
|
for (const name of ["ctok_v3.bin", "ctok_v4_7.bin"]) {
|
||||||
|
const dir = candidates.find((d) => existsSync(path.join(d, name)));
|
||||||
|
if (!dir) throw new Error(`${name}: not found in ${candidates.join(", ")}`);
|
||||||
|
const raw = new Uint8Array(await Bun.file(path.join(dir, name)).arrayBuffer());
|
||||||
|
const head = new TextDecoder().decode(raw.subarray(0, MAGIC.length));
|
||||||
|
if (head !== MAGIC) throw new Error(`${name}: bad magic ${JSON.stringify(head)}`);
|
||||||
|
if (raw[MAGIC.length] !== VERSION)
|
||||||
|
throw new Error(`${name}: unsupported version ${raw[MAGIC.length]}, want ${VERSION}`);
|
||||||
|
const packed = Bun.zstdCompressSync(raw, { level: 19 });
|
||||||
|
const out = path.join(root, "data", `${name}.zst`);
|
||||||
|
await Bun.write(out, packed);
|
||||||
|
console.log(`${out}: ${raw.length} -> ${packed.length} bytes`);
|
||||||
|
}
|
||||||
@@ -0,0 +1,130 @@
|
|||||||
|
// Pack DeepSeek V3..V4 base vocabulary (128,000 entries) into UTOK1 + zstd -19.
|
||||||
|
//
|
||||||
|
// Source: tools/cache/deepseek-v4.tokenizer.json (HF tokenizers format).
|
||||||
|
// model.vocab keys are GPT-2 byte-level alphabet strings; the three
|
||||||
|
// sentinel specials at ids 0..2 live inside model.vocab (not byte-level
|
||||||
|
// decodable) and are merge-unreachable, so they are packed as EMPTY byte
|
||||||
|
// strings per fleet protocol (rank contiguity kept; RankTable::parse
|
||||||
|
// skips zero-length entries). The 1,283 added_tokens are excluded per
|
||||||
|
// encode_ordinary semantics.
|
||||||
|
//
|
||||||
|
// Usage: bun tools/pack-deepseek.ts
|
||||||
|
|
||||||
|
const ROOT = new URL("..", import.meta.url).pathname;
|
||||||
|
const SRC = `${ROOT}tools/cache/deepseek-v4.tokenizer.json`;
|
||||||
|
const OUT = `${ROOT}data/deepseek3.bin.zst`;
|
||||||
|
const VOCAB_SIZE = 128_000;
|
||||||
|
|
||||||
|
// GPT-2 bytes_to_unicode, inverted: alphabet char -> original byte.
|
||||||
|
function unicodeToByte(): Map<string, number> {
|
||||||
|
const bs: number[] = [];
|
||||||
|
for (let i = 0x21; i <= 0x7e; i++) bs.push(i);
|
||||||
|
for (let i = 0xa1; i <= 0xac; i++) bs.push(i);
|
||||||
|
for (let i = 0xae; i <= 0xff; i++) bs.push(i);
|
||||||
|
const cs = bs.slice();
|
||||||
|
let n = 0;
|
||||||
|
for (let b = 0; b < 256; b++) {
|
||||||
|
if (!bs.includes(b)) {
|
||||||
|
bs.push(b);
|
||||||
|
cs.push(256 + n);
|
||||||
|
n++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const m = new Map<string, number>();
|
||||||
|
for (let i = 0; i < bs.length; i++) m.set(String.fromCodePoint(cs[i]!), bs[i]!);
|
||||||
|
return m;
|
||||||
|
}
|
||||||
|
|
||||||
|
const u2b = unicodeToByte();
|
||||||
|
|
||||||
|
function decodeToken(tok: string): Uint8Array | null {
|
||||||
|
const out: number[] = [];
|
||||||
|
for (const ch of tok) {
|
||||||
|
const b = u2b.get(ch);
|
||||||
|
if (b === undefined) return null;
|
||||||
|
out.push(b);
|
||||||
|
}
|
||||||
|
return Uint8Array.from(out);
|
||||||
|
}
|
||||||
|
|
||||||
|
const json = await Bun.file(SRC).json();
|
||||||
|
const vocab: Record<string, number> = json.model.vocab;
|
||||||
|
const added: { id: number; content: string }[] = json.added_tokens;
|
||||||
|
|
||||||
|
if (added.length !== 1283) throw new Error(`expected 1283 added_tokens, got ${added.length}`);
|
||||||
|
|
||||||
|
// rank -> token bytes, asserting contiguity 0..127999.
|
||||||
|
const byRank: (Uint8Array | undefined)[] = new Array(VOCAB_SIZE);
|
||||||
|
let entries = 0;
|
||||||
|
for (const tok in vocab) {
|
||||||
|
const id = vocab[tok]!;
|
||||||
|
if (id < 0 || id >= VOCAB_SIZE) throw new Error(`vocab id ${id} out of range for "${tok}"`);
|
||||||
|
if (byRank[id] !== undefined) throw new Error(`duplicate rank ${id}`);
|
||||||
|
const decoded = decodeToken(tok);
|
||||||
|
if (decoded === null && id > 2) {
|
||||||
|
throw new Error(`non-byte-level token at unexpected rank ${id}: "${tok}"`);
|
||||||
|
}
|
||||||
|
// Dead sentinel specials -> empty (reachability asserted below).
|
||||||
|
byRank[id] = decoded ?? new Uint8Array(0);
|
||||||
|
entries++;
|
||||||
|
}
|
||||||
|
if (entries !== VOCAB_SIZE) throw new Error(`expected ${VOCAB_SIZE} vocab entries, got ${entries}`);
|
||||||
|
for (let r = 0; r < VOCAB_SIZE; r++) {
|
||||||
|
if (byRank[r] === undefined) throw new Error(`rank ${r} missing — vocab not contiguous`);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Non-empty byte keys must be unique or RankTable lookups are ambiguous.
|
||||||
|
const seen = new Set<string>();
|
||||||
|
for (const bytes of byRank as Uint8Array[]) {
|
||||||
|
if (bytes.length === 0) continue;
|
||||||
|
const key = Buffer.from(bytes).toString("latin1");
|
||||||
|
if (seen.has(key)) throw new Error(`duplicate token byte sequence: ${JSON.stringify(key)}`);
|
||||||
|
seen.add(key);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Merge reachability: a rank-table engine can whole-piece-match any vocab
|
||||||
|
// entry, but HF (ignore_merges=false) only ever emits alphabet chars and
|
||||||
|
// merge products. Assert the ONLY unreachable entries are the three
|
||||||
|
// sentinel specials (ids 0..2), which the pretokenizer can never yield as
|
||||||
|
// a whole piece — so no drift is possible.
|
||||||
|
{
|
||||||
|
// Merges are "left right" strings; byte-level alphabet never contains
|
||||||
|
// a raw space (space maps to Ġ), so a single split is unambiguous.
|
||||||
|
const merges: string[] = json.model.merges;
|
||||||
|
if (merges.length !== 127_741) throw new Error(`expected 127741 merges, got ${merges.length}`);
|
||||||
|
const reachable = new Set<string>();
|
||||||
|
for (const tok in vocab) if ([...tok].length === 1 && u2b.has(tok)) reachable.add(tok);
|
||||||
|
if (reachable.size !== 256) throw new Error(`expected 256 alphabet entries, got ${reachable.size}`);
|
||||||
|
for (const m of merges) {
|
||||||
|
const parts = m.split(" ");
|
||||||
|
if (parts.length !== 2) throw new Error(`malformed merge: ${JSON.stringify(m)}`);
|
||||||
|
reachable.add(parts[0]! + parts[1]!);
|
||||||
|
}
|
||||||
|
const dead: number[] = [];
|
||||||
|
for (const tok in vocab) if (!reachable.has(tok)) dead.push(vocab[tok]!);
|
||||||
|
dead.sort((x, y) => x - y);
|
||||||
|
if (dead.length !== 3 || dead[0] !== 0 || dead[1] !== 1 || dead[2] !== 2) {
|
||||||
|
throw new Error(`unexpected merge-unreachable ranks: ${dead.slice(0, 20).join(",")}`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// UTOK1: magic, u32le count, per entry LEB128(len) + raw bytes.
|
||||||
|
const parts: Uint8Array[] = [new TextEncoder().encode("UTOK1\n")];
|
||||||
|
const count = new Uint8Array(4);
|
||||||
|
new DataView(count.buffer).setUint32(0, VOCAB_SIZE, true);
|
||||||
|
parts.push(count);
|
||||||
|
for (const bytes of byRank as Uint8Array[]) {
|
||||||
|
let len = bytes.length;
|
||||||
|
const varint: number[] = [];
|
||||||
|
do {
|
||||||
|
let b = len & 0x7f;
|
||||||
|
len >>>= 7;
|
||||||
|
if (len > 0) b |= 0x80;
|
||||||
|
varint.push(b);
|
||||||
|
} while (len > 0);
|
||||||
|
parts.push(Uint8Array.from(varint), bytes);
|
||||||
|
}
|
||||||
|
const blob = Buffer.concat(parts);
|
||||||
|
const zst = Bun.zstdCompressSync(blob, { level: 19 });
|
||||||
|
await Bun.write(OUT, zst);
|
||||||
|
console.log(`packed ${VOCAB_SIZE} entries: ${blob.length} raw -> ${zst.length} zst -> ${OUT}`);
|
||||||
@@ -0,0 +1,86 @@
|
|||||||
|
// Pack GLM-5 vocab (zai-org/GLM-5 tokenizer.json) into UTOK1 + zstd -19.
|
||||||
|
// Base vocab only: 154,820 entries, ids 0..154819 contiguous; the 36
|
||||||
|
// added_tokens all sit above the base vocab (154820+) and are excluded.
|
||||||
|
// Vocab keys use the GPT-2 byte-level alphabet; decode back to raw bytes
|
||||||
|
// and assert every char maps (alphabet-decode cleanliness).
|
||||||
|
//
|
||||||
|
// Usage: bun tools/pack-glm.ts
|
||||||
|
|
||||||
|
import { zstdCompressSync } from "bun";
|
||||||
|
|
||||||
|
const EXPECTED = 154_820;
|
||||||
|
|
||||||
|
// GPT-2 bytes_to_unicode, inverted.
|
||||||
|
function unicodeToBytes(): Map<number, number> {
|
||||||
|
const bs: number[] = [];
|
||||||
|
for (let i = "!".charCodeAt(0); i <= "~".charCodeAt(0); i++) bs.push(i);
|
||||||
|
for (let i = 0xa1; i <= 0xac; i++) bs.push(i);
|
||||||
|
for (let i = 0xae; i <= 0xff; i++) bs.push(i);
|
||||||
|
const cs = bs.slice();
|
||||||
|
let n = 0;
|
||||||
|
for (let b = 0; b < 256; b++) {
|
||||||
|
if (!bs.includes(b)) {
|
||||||
|
bs.push(b);
|
||||||
|
cs.push(256 + n);
|
||||||
|
n++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const inv = new Map<number, number>();
|
||||||
|
for (let i = 0; i < bs.length; i++) inv.set(cs[i], bs[i]);
|
||||||
|
return inv;
|
||||||
|
}
|
||||||
|
|
||||||
|
const inv = unicodeToBytes();
|
||||||
|
const tj = await Bun.file(new URL("cache/glm-5.tokenizer.json", import.meta.url)).json();
|
||||||
|
const vocab: Record<string, number> = tj.model.vocab;
|
||||||
|
|
||||||
|
// added_tokens must all be out-of-vocab (above base range).
|
||||||
|
for (const t of tj.added_tokens) {
|
||||||
|
if (t.id < EXPECTED) throw new Error(`added token '${t.content}' (id ${t.id}) inside base vocab`);
|
||||||
|
if (vocab[t.content] !== undefined) throw new Error(`added token '${t.content}' also in model.vocab`);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Decode each key to raw bytes; assert contiguity + alphabet cleanliness.
|
||||||
|
const byRank: Uint8Array[] = new Array(EXPECTED);
|
||||||
|
let seen = 0;
|
||||||
|
for (const key in vocab) {
|
||||||
|
const rank = vocab[key];
|
||||||
|
seen++;
|
||||||
|
if (!Number.isInteger(rank) || rank < 0 || rank >= EXPECTED) throw new Error(`rank ${rank} out of range for '${key}'`);
|
||||||
|
if (byRank[rank] !== undefined) throw new Error(`duplicate rank ${rank}`);
|
||||||
|
const bytes = new Uint8Array(key.length);
|
||||||
|
let n = 0;
|
||||||
|
for (const ch of key) {
|
||||||
|
const b = inv.get(ch.codePointAt(0)!);
|
||||||
|
if (b === undefined) throw new Error(`rank ${rank}: char U+${ch.codePointAt(0)!.toString(16)} not in GPT-2 byte alphabet ('${key}')`);
|
||||||
|
bytes[n++] = b;
|
||||||
|
}
|
||||||
|
byRank[rank] = bytes.subarray(0, n);
|
||||||
|
}
|
||||||
|
if (seen !== EXPECTED) throw new Error(`vocab size ${seen} != ${EXPECTED}`);
|
||||||
|
for (let i = 0; i < EXPECTED; i++) if (byRank[i] === undefined) throw new Error(`missing rank ${i}`);
|
||||||
|
|
||||||
|
// UTOK1: magic, u32le count, per entry varint(len)+bytes.
|
||||||
|
const parts: Uint8Array[] = [new TextEncoder().encode("UTOK1\n")];
|
||||||
|
const count = new Uint8Array(4);
|
||||||
|
new DataView(count.buffer).setUint32(0, EXPECTED, true);
|
||||||
|
parts.push(count);
|
||||||
|
for (const tok of byRank) {
|
||||||
|
let len = tok.length;
|
||||||
|
const hdr: number[] = [];
|
||||||
|
do {
|
||||||
|
hdr.push(len >= 0x80 ? (len & 0x7f) | 0x80 : len);
|
||||||
|
len >>>= 7;
|
||||||
|
} while (len > 0);
|
||||||
|
parts.push(new Uint8Array(hdr), tok);
|
||||||
|
}
|
||||||
|
const raw = new Uint8Array(parts.reduce((a, p) => a + p.length, 0));
|
||||||
|
let off = 0;
|
||||||
|
for (const p of parts) {
|
||||||
|
raw.set(p, off);
|
||||||
|
off += p.length;
|
||||||
|
}
|
||||||
|
|
||||||
|
const zst = zstdCompressSync(raw, { level: 19 });
|
||||||
|
await Bun.write(new URL("../data/glm5.bin.zst", import.meta.url), zst);
|
||||||
|
console.log(`glm5: ${EXPECTED} tokens, raw ${raw.length} B, zst ${zst.length} B`);
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
// Pack Kimi K2/K3 base vocab (native tiktoken format) into UTOK1 + zstd.
|
||||||
|
// Usage: bun tools/pack-kimi.ts
|
||||||
|
// Input: tools/cache/kimi.tiktoken.model — lines of "<base64 token> <rank>".
|
||||||
|
// Specials live at 163584+ and are absent from the file.
|
||||||
|
|
||||||
|
const EXPECTED = 163_584;
|
||||||
|
|
||||||
|
const src = await Bun.file(new URL("cache/kimi.tiktoken.model", import.meta.url)).text();
|
||||||
|
const lines = src.split("\n").filter((l) => l.length > 0);
|
||||||
|
if (lines.length !== EXPECTED) throw new Error(`expected ${EXPECTED} entries, got ${lines.length}`);
|
||||||
|
|
||||||
|
const tokens: Uint8Array[] = new Array(lines.length);
|
||||||
|
for (const line of lines) {
|
||||||
|
const sp = line.indexOf(" ");
|
||||||
|
if (sp < 0) throw new Error(`malformed line: ${JSON.stringify(line)}`);
|
||||||
|
const rank = Number(line.slice(sp + 1));
|
||||||
|
if (!Number.isInteger(rank) || rank < 0 || rank >= EXPECTED) throw new Error(`bad rank ${rank}`);
|
||||||
|
if (tokens[rank] !== undefined) throw new Error(`duplicate rank ${rank}`);
|
||||||
|
tokens[rank] = Uint8Array.from(atob(line.slice(0, sp)), (c) => c.charCodeAt(0));
|
||||||
|
}
|
||||||
|
// Contiguity: every rank 0..EXPECTED-1 present exactly once.
|
||||||
|
for (let r = 0; r < EXPECTED; r++) if (tokens[r] === undefined) throw new Error(`missing rank ${r}`);
|
||||||
|
|
||||||
|
// UTOK1: magic 'UTOK1\n', u32le count, per entry varint(len)+bytes.
|
||||||
|
const parts: Uint8Array[] = [];
|
||||||
|
parts.push(new TextEncoder().encode("UTOK1\n"));
|
||||||
|
const cnt = new Uint8Array(4);
|
||||||
|
new DataView(cnt.buffer).setUint32(0, EXPECTED, true);
|
||||||
|
parts.push(cnt);
|
||||||
|
for (const tok of tokens) {
|
||||||
|
let n = tok.length;
|
||||||
|
const v: number[] = [];
|
||||||
|
while (n >= 0x80) {
|
||||||
|
v.push((n & 0x7f) | 0x80);
|
||||||
|
n >>>= 7;
|
||||||
|
}
|
||||||
|
v.push(n);
|
||||||
|
parts.push(new Uint8Array(v), tok);
|
||||||
|
}
|
||||||
|
const raw = new Uint8Array(parts.reduce((s, p) => s + p.length, 0));
|
||||||
|
let off = 0;
|
||||||
|
for (const p of parts) {
|
||||||
|
raw.set(p, off);
|
||||||
|
off += p.length;
|
||||||
|
}
|
||||||
|
|
||||||
|
const zst = Bun.zstdCompressSync(raw, { level: 19 });
|
||||||
|
await Bun.write(new URL("../data/kimi_k2.bin.zst", import.meta.url), zst);
|
||||||
|
console.log(`kimi_k2: ${EXPECTED} entries, raw ${raw.length} B, zst ${zst.length} B`);
|
||||||
@@ -0,0 +1,57 @@
|
|||||||
|
// Pack OpenAI tiktoken rank files into UTOK1 + zstd -19 blobs.
|
||||||
|
//
|
||||||
|
// bun tools/pack-openai.ts
|
||||||
|
//
|
||||||
|
// Reads tools/cache/{o200k_base,cl100k_base}.tiktoken (base64-token + rank
|
||||||
|
// per line), asserts rank contiguity, writes data/<name>.bin.zst.
|
||||||
|
|
||||||
|
const root = new URL("..", import.meta.url).pathname;
|
||||||
|
|
||||||
|
function varint(n: number): number[] {
|
||||||
|
const out: number[] = [];
|
||||||
|
while (n >= 0x80) {
|
||||||
|
out.push((n & 0x7f) | 0x80);
|
||||||
|
n >>>= 7;
|
||||||
|
}
|
||||||
|
out.push(n);
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
async function pack(name: string, expected: number) {
|
||||||
|
const text = await Bun.file(`${root}tools/cache/${name}.tiktoken`).text();
|
||||||
|
const lines = text.split("\n").filter((l) => l.length > 0);
|
||||||
|
if (lines.length !== expected) {
|
||||||
|
throw new Error(`${name}: expected ${expected} entries, got ${lines.length}`);
|
||||||
|
}
|
||||||
|
const chunks: Uint8Array[] = [];
|
||||||
|
let total = 0;
|
||||||
|
const push = (b: Uint8Array) => {
|
||||||
|
chunks.push(b);
|
||||||
|
total += b.length;
|
||||||
|
};
|
||||||
|
const header = new Uint8Array(10);
|
||||||
|
header.set(new TextEncoder().encode("UTOK1\n"), 0);
|
||||||
|
new DataView(header.buffer).setUint32(6, lines.length, true);
|
||||||
|
push(header);
|
||||||
|
for (let rank = 0; rank < lines.length; rank++) {
|
||||||
|
const [b64, rankStr] = lines[rank].split(" ");
|
||||||
|
if (Number(rankStr) !== rank) {
|
||||||
|
throw new Error(`${name}: rank discontinuity at line ${rank}: got ${rankStr}`);
|
||||||
|
}
|
||||||
|
const token = Uint8Array.fromBase64(b64);
|
||||||
|
push(new Uint8Array(varint(token.length)));
|
||||||
|
push(token);
|
||||||
|
}
|
||||||
|
const raw = new Uint8Array(total);
|
||||||
|
let off = 0;
|
||||||
|
for (const c of chunks) {
|
||||||
|
raw.set(c, off);
|
||||||
|
off += c.length;
|
||||||
|
}
|
||||||
|
const zst = Bun.zstdCompressSync(raw, { level: 19 });
|
||||||
|
await Bun.write(`${root}data/${name}.bin.zst`, zst);
|
||||||
|
console.log(`${name}: ${lines.length} entries, ${raw.length} raw -> ${zst.length} zst`);
|
||||||
|
}
|
||||||
|
|
||||||
|
await pack("o200k_base", 199998);
|
||||||
|
await pack("cl100k_base", 100256);
|
||||||
@@ -0,0 +1,110 @@
|
|||||||
|
// Pack the Qwen3 (3.5/3.6/3.8) vocab into data/qwen3.bin.zst (UTOK1 + zstd -19).
|
||||||
|
//
|
||||||
|
// Source: tools/cache/qwen3.8.tokenizer.json (HF tokenizers format).
|
||||||
|
// The vocab keys are plain GPT-2 byte-level alphabet strings (the
|
||||||
|
// families.json note about id 0 = '|' was a misdiagnosis; id 0 is '!').
|
||||||
|
//
|
||||||
|
// Real trap handled here: 201 vocab entries are unreachable via the merges
|
||||||
|
// list (len(vocab) - 256 byte tokens - len(merges)). With ignore_merges=false
|
||||||
|
// the HF tokenizer can never emit them, but a rank-table engine's whole-piece
|
||||||
|
// short-circuit would. We keep their rank slots (UTOK1 requires rank = index)
|
||||||
|
// but emit them as EMPTY byte strings: the splitter never produces empty
|
||||||
|
// pieces, so they become unmatchable — verified to reproduce reference ids
|
||||||
|
// exactly (fixtures/qwen3.json).
|
||||||
|
//
|
||||||
|
// Run: bun tools/pack-qwen.ts
|
||||||
|
|
||||||
|
const SRC = new URL("cache/qwen3.8.tokenizer.json", import.meta.url).pathname;
|
||||||
|
const OUT = new URL("../data/qwen3.bin.zst", import.meta.url).pathname;
|
||||||
|
|
||||||
|
const VOCAB_SIZE = 248_044; // base vocab; 33 added tokens (248044-248076) excluded
|
||||||
|
const ALPHABET_SIZE = 256;
|
||||||
|
|
||||||
|
const tj = await Bun.file(SRC).json();
|
||||||
|
const model = tj.model;
|
||||||
|
if (model.type !== "BPE") throw new Error(`unexpected model.type ${model.type}`);
|
||||||
|
if (model.byte_fallback || model.ignore_merges) throw new Error("unexpected model flags");
|
||||||
|
if (tj.normalizer?.type !== "NFC") throw new Error("expected NFC normalizer");
|
||||||
|
|
||||||
|
const vocab: Record<string, number> = model.vocab;
|
||||||
|
const merges: (string | [string, string])[] = model.merges;
|
||||||
|
|
||||||
|
// GPT-2 byte-level alphabet: unicode char -> original byte.
|
||||||
|
const u2b: Record<string, number> = {};
|
||||||
|
{
|
||||||
|
const bs: number[] = [];
|
||||||
|
for (let b = 0x21; b <= 0x7e; b++) bs.push(b);
|
||||||
|
for (let b = 0xa1; b <= 0xac; b++) bs.push(b);
|
||||||
|
for (let b = 0xae; b <= 0xff; b++) bs.push(b);
|
||||||
|
const seen = new Set(bs);
|
||||||
|
const cs = bs.slice();
|
||||||
|
let n = 0;
|
||||||
|
for (let b = 0; b < 256; b++) {
|
||||||
|
if (!seen.has(b)) {
|
||||||
|
bs.push(b);
|
||||||
|
cs.push(256 + n++);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (let i = 0; i < bs.length; i++) u2b[String.fromCodePoint(cs[i])] = bs[i];
|
||||||
|
}
|
||||||
|
|
||||||
|
// Merge-reachable token strings.
|
||||||
|
const reachable = new Set<string>();
|
||||||
|
for (const m of merges) {
|
||||||
|
const [a, b] = typeof m === "string" ? [m.slice(0, m.indexOf(" ")), m.slice(m.indexOf(" ") + 1)] : m;
|
||||||
|
reachable.add(a + b);
|
||||||
|
}
|
||||||
|
|
||||||
|
// rank -> raw bytes (empty for merge-unreachable multi-char entries).
|
||||||
|
const entries: (Uint8Array | null)[] = new Array(Object.keys(vocab).length).fill(null);
|
||||||
|
let dead = 0;
|
||||||
|
for (const tok in vocab) {
|
||||||
|
const rank = vocab[tok];
|
||||||
|
if (entries[rank] !== null) throw new Error(`duplicate rank ${rank}`);
|
||||||
|
const chars = [...tok];
|
||||||
|
if (chars.length > 1 && !reachable.has(tok)) {
|
||||||
|
dead++;
|
||||||
|
entries[rank] = new Uint8Array(0);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
const bytes = new Uint8Array(chars.length);
|
||||||
|
for (let i = 0; i < chars.length; i++) {
|
||||||
|
const b = u2b[chars[i]];
|
||||||
|
if (b === undefined) throw new Error(`non-alphabet char in vocab entry ${rank}: ${tok}`);
|
||||||
|
bytes[i] = b;
|
||||||
|
}
|
||||||
|
entries[rank] = bytes;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Assertions: size + rank contiguity (no null slot).
|
||||||
|
if (entries.length !== VOCAB_SIZE) throw new Error(`vocab size ${entries.length}, expected ${VOCAB_SIZE}`);
|
||||||
|
for (let r = 0; r < entries.length; r++) {
|
||||||
|
if (entries[r] === null) throw new Error(`rank gap at ${r}: ranks not contiguous`);
|
||||||
|
}
|
||||||
|
|
||||||
|
// UTOK1: magic, u32le count, per entry varint(len) + bytes.
|
||||||
|
const parts: Uint8Array[] = [new TextEncoder().encode("UTOK1\n")];
|
||||||
|
const count = new Uint8Array(4);
|
||||||
|
new DataView(count.buffer).setUint32(0, entries.length, true);
|
||||||
|
parts.push(count);
|
||||||
|
for (const bytes of entries as Uint8Array[]) {
|
||||||
|
let len = bytes.length;
|
||||||
|
const varint: number[] = [];
|
||||||
|
do {
|
||||||
|
varint.push(len >= 0x80 ? (len & 0x7f) | 0x80 : len);
|
||||||
|
len >>>= 7;
|
||||||
|
} while (len > 0);
|
||||||
|
parts.push(new Uint8Array(varint), bytes);
|
||||||
|
}
|
||||||
|
const raw = new Uint8Array(parts.reduce((n, p) => n + p.length, 0));
|
||||||
|
{
|
||||||
|
let off = 0;
|
||||||
|
for (const p of parts) {
|
||||||
|
raw.set(p, off);
|
||||||
|
off += p.length;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const packed = Bun.zstdCompressSync(raw, { level: 19 });
|
||||||
|
await Bun.write(OUT, packed);
|
||||||
|
console.log(`qwen3: ${entries.length} entries (${dead} dead slots emptied), raw ${raw.length} B -> ${packed.length} B zstd`);
|
||||||
+2
-1
@@ -103,6 +103,7 @@ providers:
|
|||||||
- `discovery.type`: `ollama`, `llama.cpp`, `lm-studio`, `openai-models-list`, `proxy`, or `litellm`
|
- `discovery.type`: `ollama`, `llama.cpp`, `lm-studio`, `openai-models-list`, `proxy`, or `litellm`
|
||||||
- `transport`: `pi-native` only. When set, every model under that provider is sent to an `omp auth-gateway` compatible `baseUrl` via `POST /v1/pi/stream`; `apiKey` is the gateway bearer.
|
- `transport`: `pi-native` only. When set, every model under that provider is sent to an `omp auth-gateway` compatible `baseUrl` via `POST /v1/pi/stream`; `apiKey` is the gateway bearer.
|
||||||
- `imageInputDecoder`: `stb` only. Set this on a custom model or `modelOverrides` entry when the serving backend uses an STB-compatible image decoder that cannot accept WebP; OMP converts attached and historical WebP images before provider dispatch.
|
- `imageInputDecoder`: `stb` only. Set this on a custom model or `modelOverrides` entry when the serving backend uses an STB-compatible image decoder that cannot accept WebP; OMP converts attached and historical WebP images before provider dispatch.
|
||||||
|
- `tokenizer`: opt into a specific embedded local tokenizer when a proxy's model id is ambiguous or noncanonical. Allowed values: `claude-v3`, `claude-v47`, `claude-v5`, `claude-v5-sonnet`, `qwen3`, `deepseek-v3`, `kimi-k2`, and `glm5`. Omit it to use catalog identity policy; unknown models retain the fast local estimate.
|
||||||
|
|
||||||
## Validation rules (current)
|
## Validation rules (current)
|
||||||
|
|
||||||
@@ -193,7 +194,7 @@ Provider defaults vs per-model overrides:
|
|||||||
- Provider `headers`, `compat`, and `remoteCompaction` are baselines.
|
- Provider `headers`, `compat`, and `remoteCompaction` are baselines.
|
||||||
- Model `headers` override provider header keys.
|
- Model `headers` override provider header keys.
|
||||||
- `modelOverrides` can override model metadata (`name`, `reasoning`, `thinking`, `input`, `imageInputDecoder`,
|
- `modelOverrides` can override model metadata (`name`, `reasoning`, `thinking`, `input`, `imageInputDecoder`,
|
||||||
`supportsTools`, `cost`, `premiumMultiplier`, `contextWindow`, `maxTokens`,
|
`tokenizer`, `supportsTools`, `cost`, `premiumMultiplier`, `contextWindow`, `maxTokens`,
|
||||||
`omitMaxOutputTokens`, `headers`, `compat`, `contextPromotionTarget`, `compactionModel`, and
|
`omitMaxOutputTokens`, `headers`, `compat`, `contextPromotionTarget`, `compactionModel`, and
|
||||||
`remoteCompaction`).
|
`remoteCompaction`).
|
||||||
- `compat` is deep-merged for nested routing blocks (`openRouterRouting`, `vercelGatewayRouting`,
|
- `compat` is deep-merged for nested routing blocks (`openRouterRouting`, `vercelGatewayRouting`,
|
||||||
|
|||||||
@@ -14,7 +14,7 @@
|
|||||||
|
|
||||||
### Changed
|
### Changed
|
||||||
|
|
||||||
- Claude models get exact native ctok counts instead of the bytes/4 estimate: the tokenizer family is resolved per model (v3 for Claude 3 … Opus 4.6, v4.7 for Opus 4.7–4.9, v5 for Opus 5+, the sonnet-5 frame variant for the non-opus 5-series). Non-Claude models keep the fast estimate (or o200k with `PI_TOKENIZER_ACCURATE=1`).
|
- Catalog-resolved tokenizer families now drive exact native counts: Claude, Qwen 3.5+, DeepSeek V3/V4/R1, Kimi K2/K3, and GLM-5+ use their matching embedded tokenizer; unknown models retain the fast estimate (or o200k with `PI_TOKENIZER_ACCURATE=1`). `Tokenizer` now takes the resolved catalog `Model`, never a raw model id.
|
||||||
|
|
||||||
## [17.3.8] - 2026-08-19
|
## [17.3.8] - 2026-08-19
|
||||||
|
|
||||||
|
|||||||
@@ -37,7 +37,7 @@ import {
|
|||||||
} from "./agent-loop";
|
} from "./agent-loop";
|
||||||
import type { AppendOnlyContextManager } from "./append-only-context";
|
import type { AppendOnlyContextManager } from "./append-only-context";
|
||||||
import { isProviderRefusalMessage } from "./replay-policy";
|
import { isProviderRefusalMessage } from "./replay-policy";
|
||||||
import { claudeEncodingForModel, Tokenizer } from "./tokenizer";
|
import { tokenizerEncodingForModel, Tokenizer } from "./tokenizer";
|
||||||
import type {
|
import type {
|
||||||
AgentBeforeModelCall,
|
AgentBeforeModelCall,
|
||||||
AgentContext,
|
AgentContext,
|
||||||
@@ -364,7 +364,7 @@ export class Agent {
|
|||||||
pendingToolCalls: new Set<string>(),
|
pendingToolCalls: new Set<string>(),
|
||||||
error: undefined,
|
error: undefined,
|
||||||
};
|
};
|
||||||
#tokenizer = new Tokenizer(this.#state.model?.id);
|
#tokenizer = new Tokenizer(this.#state.model);
|
||||||
#listeners = new Set<(e: AgentEvent) => void>();
|
#listeners = new Set<(e: AgentEvent) => void>();
|
||||||
#abortController?: AbortController;
|
#abortController?: AbortController;
|
||||||
#convertToLlm: (messages: AgentMessage[]) => Message[] | Promise<Message[]>;
|
#convertToLlm: (messages: AgentMessage[]) => Message[] | Promise<Message[]>;
|
||||||
@@ -460,7 +460,7 @@ export class Agent {
|
|||||||
if (opts.initialState?.messages) this.#state.messages = opts.initialState.messages.slice();
|
if (opts.initialState?.messages) this.#state.messages = opts.initialState.messages.slice();
|
||||||
if (opts.initialState?.pendingToolCalls)
|
if (opts.initialState?.pendingToolCalls)
|
||||||
this.#state.pendingToolCalls = new Set(opts.initialState.pendingToolCalls);
|
this.#state.pendingToolCalls = new Set(opts.initialState.pendingToolCalls);
|
||||||
this.#syncTokenizer(this.#state.model?.id);
|
this.#syncTokenizer(this.#state.model);
|
||||||
this.#convertToLlm = opts.convertToLlm || defaultConvertToLlm;
|
this.#convertToLlm = opts.convertToLlm || defaultConvertToLlm;
|
||||||
this.#transformContext = opts.transformContext;
|
this.#transformContext = opts.transformContext;
|
||||||
this.#steeringMode = opts.steeringMode || "one-at-a-time";
|
this.#steeringMode = opts.steeringMode || "one-at-a-time";
|
||||||
@@ -739,9 +739,9 @@ export class Agent {
|
|||||||
* Swap the tokenizer only when the encoding actually changes, so the warm
|
* Swap the tokenizer only when the encoding actually changes, so the warm
|
||||||
* per-message memo survives same-encoding model switches.
|
* per-message memo survives same-encoding model switches.
|
||||||
*/
|
*/
|
||||||
#syncTokenizer(modelId: string | null | undefined): void {
|
#syncTokenizer(model: Model | null | undefined): void {
|
||||||
if ((modelId ? claudeEncodingForModel(modelId) : null) !== this.#tokenizer.encoding) {
|
if (tokenizerEncodingForModel(model) !== this.#tokenizer.encoding) {
|
||||||
this.#tokenizer = new Tokenizer(modelId);
|
this.#tokenizer = new Tokenizer(model);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -922,9 +922,9 @@ export class Agent {
|
|||||||
this.#state.systemPrompt = typeof v === "string" ? [v] : v;
|
this.#state.systemPrompt = typeof v === "string" ? [v] : v;
|
||||||
}
|
}
|
||||||
|
|
||||||
setModel(m: Model) {
|
setModel(model: Model) {
|
||||||
this.#state.model = m;
|
this.#state.model = model;
|
||||||
this.#syncTokenizer(m?.id);
|
this.#syncTokenizer(model);
|
||||||
}
|
}
|
||||||
|
|
||||||
setThinkingLevel(l: Effort | undefined) {
|
setThinkingLevel(l: Effort | undefined) {
|
||||||
|
|||||||
@@ -313,7 +313,7 @@ export async function generateBranchSummary(
|
|||||||
// Token budget = context window minus reserved space for prompt + response
|
// Token budget = context window minus reserved space for prompt + response
|
||||||
const contextWindow = model.contextWindow || 128000;
|
const contextWindow = model.contextWindow || 128000;
|
||||||
const tokenBudget = contextWindow - reserveTokens;
|
const tokenBudget = contextWindow - reserveTokens;
|
||||||
const tokenizer = new Tokenizer(model.id);
|
const tokenizer = new Tokenizer(model);
|
||||||
|
|
||||||
const { messages, fileOps } = prepareBranchEntries(entries, tokenizer, tokenBudget);
|
const { messages, fileOps } = prepareBranchEntries(entries, tokenizer, tokenBudget);
|
||||||
|
|
||||||
|
|||||||
@@ -860,7 +860,7 @@ export async function generateSummary(
|
|||||||
// Convert to LLM messages first (handles custom app messages when caller provides a transformer).
|
// Convert to LLM messages first (handles custom app messages when caller provides a transformer).
|
||||||
const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(currentMessages);
|
const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(currentMessages);
|
||||||
const dialect = preferredDialect(model.id);
|
const dialect = preferredDialect(model.id);
|
||||||
const tokenizer = new Tokenizer(model.id);
|
const tokenizer = new Tokenizer(model);
|
||||||
const wholeConversation = serializeConversationForSummary(llmMessages, dialect);
|
const wholeConversation = serializeConversationForSummary(llmMessages, dialect);
|
||||||
const budgetTokens = summaryInputBudgetTokens(model, maxTokens);
|
const budgetTokens = summaryInputBudgetTokens(model, maxTokens);
|
||||||
// A span that outgrew the summarizer's window is summarized as a fold: each
|
// A span that outgrew the summarizer's window is summarized as a fold: each
|
||||||
@@ -1316,7 +1316,7 @@ export function prepareCompaction(
|
|||||||
pathEntries: SessionEntry[],
|
pathEntries: SessionEntry[],
|
||||||
settings: CompactionSettings,
|
settings: CompactionSettings,
|
||||||
activeModel?: Model,
|
activeModel?: Model,
|
||||||
tokenizer: Tokenizer = new Tokenizer(activeModel?.id),
|
tokenizer: Tokenizer = new Tokenizer(activeModel),
|
||||||
): CompactionPreparation | undefined {
|
): CompactionPreparation | undefined {
|
||||||
if (pathEntries.length > 0 && pathEntries[pathEntries.length - 1].type === "compaction") {
|
if (pathEntries.length > 0 && pathEntries[pathEntries.length - 1].type === "compaction") {
|
||||||
return undefined;
|
return undefined;
|
||||||
@@ -1596,7 +1596,7 @@ export async function compact(
|
|||||||
: undefined;
|
: undefined;
|
||||||
const trimmed = trimRemoteCompactionInputToContextWindow(
|
const trimmed = trimRemoteCompactionInputToContextWindow(
|
||||||
remoteHistory,
|
remoteHistory,
|
||||||
new Tokenizer(model.id),
|
new Tokenizer(model),
|
||||||
model.contextWindow,
|
model.contextWindow,
|
||||||
instructions,
|
instructions,
|
||||||
tools,
|
tools,
|
||||||
|
|||||||
@@ -791,7 +791,7 @@ export async function requestOpenAiRemoteCompaction(
|
|||||||
const requestModel = resolveOpenAiCompactModel(model);
|
const requestModel = resolveOpenAiCompactModel(model);
|
||||||
const trimmed = trimRemoteCompactionInputToContextWindow(
|
const trimmed = trimRemoteCompactionInputToContextWindow(
|
||||||
compactInput,
|
compactInput,
|
||||||
new Tokenizer(model.id),
|
new Tokenizer(model),
|
||||||
model.contextWindow,
|
model.contextWindow,
|
||||||
instructions,
|
instructions,
|
||||||
);
|
);
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import { bareModelId, parseAnthropicModel, semverGte } from "@oh-my-pi/pi-catalog/identity";
|
import type { Model } from "@oh-my-pi/pi-ai";
|
||||||
|
import type { ModelTokenizer } from "@oh-my-pi/pi-catalog/types";
|
||||||
import { countTokens as countTokensNat, Encoding } from "@oh-my-pi/pi-natives";
|
import { countTokens as countTokensNat, Encoding } from "@oh-my-pi/pi-natives";
|
||||||
import { stringifyJson } from "@oh-my-pi/pi-utils";
|
import { stringifyJson } from "@oh-my-pi/pi-utils";
|
||||||
import * as snapcompact from "@oh-my-pi/snapcompact";
|
import * as snapcompact from "@oh-my-pi/snapcompact";
|
||||||
@@ -8,38 +9,29 @@ import type { AgentMessage } from "./types";
|
|||||||
const testEnv = Bun.env.NODE_ENV === "test";
|
const testEnv = Bun.env.NODE_ENV === "test";
|
||||||
const accurate = process.env.PI_TOKENIZER_ACCURATE === "1" && !testEnv;
|
const accurate = process.env.PI_TOKENIZER_ACCURATE === "1" && !testEnv;
|
||||||
|
|
||||||
/**
|
const NATIVE_ENCODING: Record<ModelTokenizer, Encoding> = {
|
||||||
* ctok encoding for a Claude model id, or `null` for non-Claude models.
|
"claude-v3": Encoding.ClaudeV3,
|
||||||
*
|
"claude-v47": Encoding.ClaudeV47,
|
||||||
* Family routing mirrors the ctok reconstruction plus live measurement:
|
"claude-v5": Encoding.ClaudeV5,
|
||||||
* Claude 3 through Opus 4.6 (and every non-opus Claude below 5) count with
|
"claude-v5-sonnet": Encoding.ClaudeV5Sonnet,
|
||||||
* v3, Opus 4.7–4.9 with v4.7, Opus 5+ with v5, and the non-opus 5-series
|
qwen3: Encoding.Qwen3,
|
||||||
* (sonnet/fable/mythos) with the sonnet-5 frame variant. Ids
|
"deepseek-v3": Encoding.DeepSeekV3,
|
||||||
* `parseAnthropicModel` cannot classify (e.g. haiku) fall back to v3, which
|
"kimi-k2": Encoding.KimiK2,
|
||||||
* covers every such model shipped to date.
|
glm5: Encoding.Glm5,
|
||||||
*/
|
};
|
||||||
export function claudeEncodingForModel(modelId: string): Encoding | null {
|
|
||||||
const bare = bareModelId(modelId);
|
/** Maps the catalog-resolved tokenizer family to its native implementation. */
|
||||||
const parsed = parseAnthropicModel(bare);
|
export function tokenizerEncodingForModel(model: Pick<Model, "tokenizer"> | null | undefined): Encoding | null {
|
||||||
if (parsed) {
|
return model?.tokenizer ? NATIVE_ENCODING[model.tokenizer] : null;
|
||||||
if (parsed.kind === "opus") {
|
|
||||||
if (semverGte(parsed.version, "5")) return Encoding.ClaudeV5;
|
|
||||||
if (semverGte(parsed.version, "4.7")) return Encoding.ClaudeV47;
|
|
||||||
return Encoding.ClaudeV3;
|
|
||||||
}
|
|
||||||
// Sonnet, Fable, and Mythos: the 4.7 family is opus-only, and their
|
|
||||||
// 5-series message frame differs from opus-5's (measured live).
|
|
||||||
return semverGte(parsed.version, "5") ? Encoding.ClaudeV5Sonnet : Encoding.ClaudeV3;
|
|
||||||
}
|
|
||||||
return /(^|[-/.:])claude([-.:]|$)/i.test(bare) ? Encoding.ClaudeV3 : null;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* `strict` always pays for an exact native count (Claude ctok when the model
|
* `strict` always pays for an exact native count (the catalog-resolved
|
||||||
* is Claude, o200k_base otherwise). `approximate` and `upperbound` prefer the
|
* tokenizer when known, o200k_base otherwise). `approximate` and
|
||||||
* same exact count when a Claude encoding is known or `PI_TOKENIZER_ACCURATE=1`
|
* `upperbound` prefer the same exact count for known tokenizer families or
|
||||||
* is set, and otherwise fall back to a cheap heuristic: `approximate` a
|
* when `PI_TOKENIZER_ACCURATE=1` is set; otherwise they use a cheap heuristic:
|
||||||
* bytes/4 guess, `upperbound` the raw byte length (never undercounts).
|
* `approximate` a bytes/4 guess, `upperbound` the raw byte length (never
|
||||||
|
* undercounts).
|
||||||
*/
|
*/
|
||||||
export type TokenCountMode = "strict" | "approximate" | "upperbound";
|
export type TokenCountMode = "strict" | "approximate" | "upperbound";
|
||||||
|
|
||||||
@@ -103,13 +95,13 @@ interface MessageEstimate {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Model-aware local token counter. Immutable: the encoding is fixed at
|
* Model-aware local token counter. Immutable: the catalog-resolved encoding
|
||||||
* construction, so a cached count can never straddle two encodings. An `Agent`
|
* is fixed at construction, so a cached count can never straddle two
|
||||||
* owns one for its active model (swapping the instance when the model's
|
* encodings. An `Agent` owns one for its active model (swapping the instance
|
||||||
* encoding changes) and exposes it as `agent.tokenizer`; one-shot flows
|
* when the model's encoding changes); one-shot flows construct their own for
|
||||||
* (summarization, snapcompact sizing) construct their own for the model that
|
* the model that will be billed. Known tokenizer families use exact native
|
||||||
* will be billed. Claude models get exact native ctok counts; everything else
|
* counts; unknown models keep the fast byte estimate (or o200k when
|
||||||
* keeps the fast byte estimate (or o200k when `PI_TOKENIZER_ACCURATE=1`).
|
* `PI_TOKENIZER_ACCURATE=1`).
|
||||||
*/
|
*/
|
||||||
export class Tokenizer {
|
export class Tokenizer {
|
||||||
readonly #encoding: Encoding | null;
|
readonly #encoding: Encoding | null;
|
||||||
@@ -124,8 +116,8 @@ export class Tokenizer {
|
|||||||
*/
|
*/
|
||||||
#estimates = new WeakMap<AgentMessage, MessageEstimate>();
|
#estimates = new WeakMap<AgentMessage, MessageEstimate>();
|
||||||
|
|
||||||
constructor(modelId?: string | null) {
|
constructor(model?: Pick<Model, "tokenizer"> | null) {
|
||||||
this.#encoding = modelId ? claudeEncodingForModel(modelId) : null;
|
this.#encoding = tokenizerEncodingForModel(model);
|
||||||
}
|
}
|
||||||
|
|
||||||
get encoding(): Encoding | null {
|
get encoding(): Encoding | null {
|
||||||
|
|||||||
@@ -1,35 +1,25 @@
|
|||||||
import { describe, expect, test } from "bun:test";
|
import { describe, expect, test } from "bun:test";
|
||||||
import { Encoding } from "@oh-my-pi/pi-natives";
|
import { Encoding } from "@oh-my-pi/pi-natives";
|
||||||
import { claudeEncodingForModel, Tokenizer } from "../src/tokenizer";
|
import { tokenizerEncodingForModel, Tokenizer } from "../src/tokenizer";
|
||||||
|
|
||||||
// Contract: local token counting must pick the ctok family that matches the
|
// Contract: the catalog resolves model identity once as Model.tokenizer; the
|
||||||
// model's tokenizer generation (v3 for Claude 3 … Opus 4.6 and every
|
// agent maps that catalog property to the matching native counter. A wrong
|
||||||
// non-opus < 5, v4.7 for Opus 4.7–4.9, v5 for the 5-series). A wrong family
|
// row silently skews every context-budget and compaction decision.
|
||||||
// silently skews every context-budget and compaction decision for that model.
|
describe("tokenizerEncodingForModel", () => {
|
||||||
describe("claudeEncodingForModel", () => {
|
test("maps every catalog tokenizer family to its native counter", () => {
|
||||||
test("opus routes on the 4.7 and 5.0 version thresholds", () => {
|
expect(tokenizerEncodingForModel({ tokenizer: "claude-v3" })).toBe(Encoding.ClaudeV3);
|
||||||
expect(claudeEncodingForModel("claude-opus-4-5")).toBe(Encoding.ClaudeV3);
|
expect(tokenizerEncodingForModel({ tokenizer: "claude-v47" })).toBe(Encoding.ClaudeV47);
|
||||||
expect(claudeEncodingForModel("claude-opus-4-6")).toBe(Encoding.ClaudeV3);
|
expect(tokenizerEncodingForModel({ tokenizer: "claude-v5" })).toBe(Encoding.ClaudeV5);
|
||||||
expect(claudeEncodingForModel("claude-opus-4-7")).toBe(Encoding.ClaudeV47);
|
expect(tokenizerEncodingForModel({ tokenizer: "claude-v5-sonnet" })).toBe(Encoding.ClaudeV5Sonnet);
|
||||||
expect(claudeEncodingForModel("claude-opus-4-9-20260101")).toBe(Encoding.ClaudeV47);
|
expect(tokenizerEncodingForModel({ tokenizer: "qwen3" })).toBe(Encoding.Qwen3);
|
||||||
expect(claudeEncodingForModel("claude-opus-5")).toBe(Encoding.ClaudeV5);
|
expect(tokenizerEncodingForModel({ tokenizer: "deepseek-v3" })).toBe(Encoding.DeepSeekV3);
|
||||||
|
expect(tokenizerEncodingForModel({ tokenizer: "kimi-k2" })).toBe(Encoding.KimiK2);
|
||||||
|
expect(tokenizerEncodingForModel({ tokenizer: "glm5" })).toBe(Encoding.Glm5);
|
||||||
});
|
});
|
||||||
|
|
||||||
test("non-opus kinds skip the opus-only 4.7 family and use the sonnet-5 frame", () => {
|
test("leaves unknown catalog models on the estimate policy", () => {
|
||||||
expect(claudeEncodingForModel("claude-sonnet-4-5-20250929")).toBe(Encoding.ClaudeV3);
|
expect(tokenizerEncodingForModel({})).toBeNull();
|
||||||
expect(claudeEncodingForModel("claude-sonnet-5")).toBe(Encoding.ClaudeV5Sonnet);
|
expect(tokenizerEncodingForModel(undefined)).toBeNull();
|
||||||
expect(claudeEncodingForModel("claude-fable-5")).toBe(Encoding.ClaudeV5Sonnet);
|
|
||||||
});
|
|
||||||
test("unclassifiable claude ids fall back to v3; provider prefixes are stripped", () => {
|
|
||||||
expect(claudeEncodingForModel("claude-3-5-haiku-20241022")).toBe(Encoding.ClaudeV3);
|
|
||||||
expect(claudeEncodingForModel("claude-haiku-4-5")).toBe(Encoding.ClaudeV3);
|
|
||||||
expect(claudeEncodingForModel("anthropic/claude-opus-4-7")).toBe(Encoding.ClaudeV47);
|
|
||||||
});
|
|
||||||
|
|
||||||
test("non-claude models get no ctok encoding", () => {
|
|
||||||
expect(claudeEncodingForModel("gpt-5.4")).toBeNull();
|
|
||||||
expect(claudeEncodingForModel("gemini-3-pro")).toBeNull();
|
|
||||||
expect(claudeEncodingForModel("glm-4.7")).toBeNull();
|
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
@@ -40,26 +30,26 @@ describe("Tokenizer", () => {
|
|||||||
expect(tokenizer.countTokens("hello world")).toBe(3);
|
expect(tokenizer.countTokens("hello world")).toBe(3);
|
||||||
});
|
});
|
||||||
|
|
||||||
test("encoding is fixed at construction per model id", () => {
|
test("encoding is fixed at construction from the catalog model", () => {
|
||||||
expect(new Tokenizer("claude-opus-4-7").encoding).toBe(Encoding.ClaudeV47);
|
expect(new Tokenizer({ tokenizer: "claude-v47" }).encoding).toBe(Encoding.ClaudeV47);
|
||||||
expect(new Tokenizer("claude-opus-5").encoding).toBe(Encoding.ClaudeV5);
|
expect(new Tokenizer({ tokenizer: "claude-v5" }).encoding).toBe(Encoding.ClaudeV5);
|
||||||
expect(new Tokenizer("gpt-5.4").encoding).toBeNull();
|
expect(new Tokenizer({}).encoding).toBeNull();
|
||||||
expect(new Tokenizer(undefined).encoding).toBeNull();
|
expect(new Tokenizer(undefined).encoding).toBeNull();
|
||||||
});
|
});
|
||||||
|
|
||||||
test("separate instances do not interfere with each other", () => {
|
test("separate instances do not interfere with each other", () => {
|
||||||
const t1 = new Tokenizer("claude-opus-4-7");
|
const t1 = new Tokenizer({ tokenizer: "claude-v47" });
|
||||||
const t2 = new Tokenizer("claude-opus-5");
|
const t2 = new Tokenizer({ tokenizer: "qwen3" });
|
||||||
const t3 = new Tokenizer("gpt-5.4");
|
const t3 = new Tokenizer({});
|
||||||
|
|
||||||
expect(t1.encoding).toBe(Encoding.ClaudeV47);
|
expect(t1.encoding).toBe(Encoding.ClaudeV47);
|
||||||
expect(t2.encoding).toBe(Encoding.ClaudeV5);
|
expect(t2.encoding).toBe(Encoding.Qwen3);
|
||||||
expect(t3.encoding).toBeNull();
|
expect(t3.encoding).toBeNull();
|
||||||
|
|
||||||
const t4 = new Tokenizer("claude-sonnet-4-5-20250929");
|
const t4 = new Tokenizer({ tokenizer: "claude-v3" });
|
||||||
expect(t4.encoding).toBe(Encoding.ClaudeV3);
|
expect(t4.encoding).toBe(Encoding.ClaudeV3);
|
||||||
expect(t1.encoding).toBe(Encoding.ClaudeV47);
|
expect(t1.encoding).toBe(Encoding.ClaudeV47);
|
||||||
expect(t2.encoding).toBe(Encoding.ClaudeV5);
|
expect(t2.encoding).toBe(Encoding.Qwen3);
|
||||||
expect(t3.encoding).toBeNull();
|
expect(t3.encoding).toBeNull();
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
@@ -78,8 +68,7 @@ describe("countTokens with modes", () => {
|
|||||||
test("strict mode uses native counting regardless of encoding", () => {
|
test("strict mode uses native counting regardless of encoding", () => {
|
||||||
const noEncoding = new Tokenizer();
|
const noEncoding = new Tokenizer();
|
||||||
expect(noEncoding.countTokens("hello world", "strict")).toBe(2);
|
expect(noEncoding.countTokens("hello world", "strict")).toBe(2);
|
||||||
|
const claudeEncoding = new Tokenizer({ tokenizer: "claude-v47" });
|
||||||
const claudeEncoding = new Tokenizer("claude-opus-4-7");
|
|
||||||
expect(claudeEncoding.countTokens("hello world", "strict")).toBeGreaterThan(0);
|
expect(claudeEncoding.countTokens("hello world", "strict")).toBeGreaterThan(0);
|
||||||
});
|
});
|
||||||
|
|
||||||
@@ -87,8 +76,8 @@ describe("countTokens with modes", () => {
|
|||||||
// approximate/upperbound skip the encoding entirely under NODE_ENV=test
|
// approximate/upperbound skip the encoding entirely under NODE_ENV=test
|
||||||
// (fast estimate for a snappy suite); strict is testEnv-independent, so
|
// (fast estimate for a snappy suite); strict is testEnv-independent, so
|
||||||
// it is the mode that proves per-instance encoding isolation here.
|
// it is the mode that proves per-instance encoding isolation here.
|
||||||
const claude = new Tokenizer("claude-opus-4-7");
|
const claude = new Tokenizer({ tokenizer: "claude-v47" });
|
||||||
const generic = new Tokenizer("gpt-5.4");
|
const generic = new Tokenizer({});
|
||||||
expect(claude.countTokens("hello world", "strict")).not.toBe(generic.countTokens("hello world", "strict"));
|
expect(claude.countTokens("hello world", "strict")).not.toBe(generic.countTokens("hello world", "strict"));
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -2,6 +2,10 @@
|
|||||||
|
|
||||||
## [Unreleased]
|
## [Unreleased]
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- Models now materialize an optional `tokenizer` family in the catalog (`claude-v3`/`v47`/`v5`, Qwen 3.5+, DeepSeek V3/V4/R1, Kimi K2/K3, and GLM-5+). The field follows `requestModelId`, applies to bundled, discovered, and custom models, and can be explicitly overridden in model configuration.
|
||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
- Fixed `opencode-go/muse-spark-1.2` and `muse-spark-1.2-contributor` still failing every tool-call turn with `OpenAI completions stream closed before a finish_reason was received` on 17.3.8. The earlier pin only covered the models.dev resolver, but models.dev omits these ids under `opencode-go` entirely, so live `/zen/go/v1/models` discovery had no bundled reference and defaulted them to chat completions. The per-id API pins now also apply inside the discovery mapper, and pinned ids invalidate cached routes written before the pin ([#8957](https://github.com/can1357/oh-my-pi/issues/8957)).
|
- Fixed `opencode-go/muse-spark-1.2` and `muse-spark-1.2-contributor` still failing every tool-call turn with `OpenAI completions stream closed before a finish_reason was received` on 17.3.8. The earlier pin only covered the models.dev resolver, but models.dev omits these ids under `opencode-go` entirely, so live `/zen/go/v1/models` discovery had no bundled reference and defaulted them to chat completions. The per-id API pins now also apply inside the discovery mapper, and pinned ids invalidate cached routes written before the pin ([#8957](https://github.com/can1357/oh-my-pi/issues/8957)).
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ import { buildDevinCompat } from "./compat/devin";
|
|||||||
import { buildOpenAICompat, buildOpenAIResponsesCompat, buildOpenRouterCompat } from "./compat/openai";
|
import { buildOpenAICompat, buildOpenAIResponsesCompat, buildOpenRouterCompat } from "./compat/openai";
|
||||||
import { bareModelId, parseOpenAIModel, semverGte } from "./identity/classify";
|
import { bareModelId, parseOpenAIModel, semverGte } from "./identity/classify";
|
||||||
import { resolveModelThinking } from "./model-thinking";
|
import { resolveModelThinking } from "./model-thinking";
|
||||||
|
import { resolveModelTokenizer } from "./model-tokenizer";
|
||||||
import type { Api, CompatOf, Model, ModelSpec } from "./types";
|
import type { Api, CompatOf, Model, ModelSpec } from "./types";
|
||||||
import { cleanModelName } from "./utils";
|
import { cleanModelName } from "./utils";
|
||||||
|
|
||||||
@@ -64,6 +65,7 @@ export function buildModel<TApi extends Api>(spec: ModelSpec<TApi>): Model<TApi>
|
|||||||
return {
|
return {
|
||||||
...spec,
|
...spec,
|
||||||
name: cleanModelName(spec.name),
|
name: cleanModelName(spec.name),
|
||||||
|
tokenizer: spec.tokenizer ?? resolveModelTokenizer(spec.requestModelId ?? spec.id),
|
||||||
thinking: resolveModelThinking(spec, compat),
|
thinking: resolveModelThinking(spec, compat),
|
||||||
supportsComputerUse: supportsOpenAIGAComputerUse(spec, supportsComputerUseConfig),
|
supportsComputerUse: supportsOpenAIGAComputerUse(spec, supportsComputerUseConfig),
|
||||||
supportsComputerUseConfig,
|
supportsComputerUseConfig,
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ export * from "./identity";
|
|||||||
export * from "./model-cache";
|
export * from "./model-cache";
|
||||||
export * from "./model-manager";
|
export * from "./model-manager";
|
||||||
export * from "./model-thinking";
|
export * from "./model-thinking";
|
||||||
|
export * from "./model-tokenizer";
|
||||||
export * from "./models";
|
export * from "./models";
|
||||||
export * from "./provider-models";
|
export * from "./provider-models";
|
||||||
export * from "./types";
|
export * from "./types";
|
||||||
|
|||||||
@@ -0,0 +1,69 @@
|
|||||||
|
import { bareModelId, parseAnthropicModel, parseGlmModel, semverGte } from "./identity/classify";
|
||||||
|
import type { ModelTokenizer } from "./types";
|
||||||
|
|
||||||
|
const DEEPSEEK_V3_ALIASES: Record<string, true> = {
|
||||||
|
"deepseek-chat": true,
|
||||||
|
"deepseek-reasoner": true,
|
||||||
|
};
|
||||||
|
const KIMI_K2_ALIASES: Record<string, true> = {
|
||||||
|
"kimi-for-coding": true,
|
||||||
|
"kimi-for-coding-highspeed": true,
|
||||||
|
};
|
||||||
|
|
||||||
|
function claudeTokenizer(modelId: string): ModelTokenizer | undefined {
|
||||||
|
const parsed = parseAnthropicModel(modelId);
|
||||||
|
if (parsed) {
|
||||||
|
if (parsed.kind === "opus") {
|
||||||
|
if (semverGte(parsed.version, "5")) return "claude-v5";
|
||||||
|
if (semverGte(parsed.version, "4.7")) return "claude-v47";
|
||||||
|
return "claude-v3";
|
||||||
|
}
|
||||||
|
return semverGte(parsed.version, "5") ? "claude-v5-sonnet" : "claude-v3";
|
||||||
|
}
|
||||||
|
return /(^|[-/.:])claude([-.:]|$)/i.test(modelId) ? "claude-v3" : undefined;
|
||||||
|
}
|
||||||
|
|
||||||
|
function qwenTokenizer(modelId: string): ModelTokenizer | undefined {
|
||||||
|
const version = /qwen[-_ ]?(\d+)\.(\d+)(?![\dbB])/i.exec(modelId);
|
||||||
|
if (!version) return undefined;
|
||||||
|
const major = Number.parseInt(version[1], 10);
|
||||||
|
const minor = Number.parseInt(version[2], 10);
|
||||||
|
return major > 3 || (major === 3 && minor >= 5) ? "qwen3" : undefined;
|
||||||
|
}
|
||||||
|
|
||||||
|
function deepSeekTokenizer(modelId: string): ModelTokenizer | undefined {
|
||||||
|
const lower = modelId.toLowerCase();
|
||||||
|
if (lower.includes("distill")) return undefined;
|
||||||
|
if (DEEPSEEK_V3_ALIASES[lower]) return "deepseek-v3";
|
||||||
|
return /(?:^|[-_.:])(?:v?[34]|r1)(?:[-_.:]|$)/.test(lower) && lower.includes("deepseek")
|
||||||
|
? "deepseek-v3"
|
||||||
|
: undefined;
|
||||||
|
}
|
||||||
|
|
||||||
|
function kimiTokenizer(modelId: string): ModelTokenizer | undefined {
|
||||||
|
const lower = modelId.toLowerCase();
|
||||||
|
if (KIMI_K2_ALIASES[lower]) return "kimi-k2";
|
||||||
|
return /(?:^|[-_.:])kimi[-_.:]?k?[23](?:[-_.:]|$)/.test(lower) ? "kimi-k2" : undefined;
|
||||||
|
}
|
||||||
|
|
||||||
|
function glmTokenizer(modelId: string): ModelTokenizer | undefined {
|
||||||
|
const glm = parseGlmModel(modelId);
|
||||||
|
return glm && semverGte(glm.version, "5") ? "glm5" : undefined;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Resolve the exact locally embedded tokenizer for a canonical model id.
|
||||||
|
*
|
||||||
|
* This is catalog policy, not a runtime caller heuristic: [`buildModel`](./build.ts)
|
||||||
|
* materializes the result as `Model.tokenizer`; consumers read that property.
|
||||||
|
*/
|
||||||
|
export function resolveModelTokenizer(modelId: string): ModelTokenizer | undefined {
|
||||||
|
const bare = bareModelId(modelId);
|
||||||
|
return (
|
||||||
|
claudeTokenizer(bare) ??
|
||||||
|
qwenTokenizer(bare) ??
|
||||||
|
deepSeekTokenizer(bare) ??
|
||||||
|
kimiTokenizer(bare) ??
|
||||||
|
glmTokenizer(bare)
|
||||||
|
);
|
||||||
|
}
|
||||||
@@ -859,6 +859,24 @@ export interface ModelCost extends TokenCost {
|
|||||||
longContext?: LongContextTokenCost;
|
longContext?: LongContextTokenCost;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Exact local content tokenizer family for a model.
|
||||||
|
*
|
||||||
|
* Absent means no first-party local tokenizer is known and consumers retain
|
||||||
|
* their estimate/default-tokenizer policy. The values name tokenizer
|
||||||
|
* generations rather than providers: DeepSeek V3 through V4 share
|
||||||
|
* `"deepseek-v3"`; Kimi K2 through K3 share `"kimi-k2"`.
|
||||||
|
*/
|
||||||
|
export type ModelTokenizer =
|
||||||
|
| "claude-v3"
|
||||||
|
| "claude-v47"
|
||||||
|
| "claude-v5"
|
||||||
|
| "claude-v5-sonnet"
|
||||||
|
| "qwen3"
|
||||||
|
| "deepseek-v3"
|
||||||
|
| "kimi-k2"
|
||||||
|
| "glm5";
|
||||||
|
|
||||||
// Model interface for the unified model system
|
// Model interface for the unified model system
|
||||||
export interface Model<TApi extends Api = Api> {
|
export interface Model<TApi extends Api = Api> {
|
||||||
id: string;
|
id: string;
|
||||||
@@ -883,6 +901,12 @@ export interface Model<TApi extends Api = Api> {
|
|||||||
provider: Provider;
|
provider: Provider;
|
||||||
baseUrl: string;
|
baseUrl: string;
|
||||||
reasoning: boolean;
|
reasoning: boolean;
|
||||||
|
/**
|
||||||
|
* Exact local tokenizer family resolved from the model identity or supplied
|
||||||
|
* explicitly by a catalog/discovery source. Absent leaves local counting to
|
||||||
|
* the consumer's fallback policy.
|
||||||
|
*/
|
||||||
|
tokenizer?: ModelTokenizer;
|
||||||
input: ("text" | "image")[];
|
input: ("text" | "image")[];
|
||||||
/**
|
/**
|
||||||
* Decoder family used for image inputs when it has narrower format support
|
* Decoder family used for image inputs when it has narrower format support
|
||||||
|
|||||||
@@ -0,0 +1,40 @@
|
|||||||
|
import { describe, expect, test } from "bun:test";
|
||||||
|
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||||
|
import { resolveModelTokenizer } from "@oh-my-pi/pi-catalog/model-tokenizer";
|
||||||
|
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
|
||||||
|
|
||||||
|
function spec(id: string): ModelSpec<"openai-completions"> {
|
||||||
|
return {
|
||||||
|
id,
|
||||||
|
name: id,
|
||||||
|
api: "openai-completions",
|
||||||
|
provider: "custom",
|
||||||
|
baseUrl: "https://api.example.com/v1",
|
||||||
|
reasoning: false,
|
||||||
|
input: ["text"],
|
||||||
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||||
|
contextWindow: 128_000,
|
||||||
|
maxTokens: 8_192,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
describe("resolveModelTokenizer", () => {
|
||||||
|
test("routes only tokenizer generations covered by embedded vocabularies", () => {
|
||||||
|
expect(resolveModelTokenizer("anthropic/claude-opus-4-7")).toBe("claude-v47");
|
||||||
|
expect(resolveModelTokenizer("claude-sonnet-5")).toBe("claude-v5-sonnet");
|
||||||
|
expect(resolveModelTokenizer("Qwen/Qwen3.8-27B")).toBe("qwen3");
|
||||||
|
expect(resolveModelTokenizer("qwen3-32b")).toBeUndefined();
|
||||||
|
expect(resolveModelTokenizer("deepseek-chat")).toBe("deepseek-v3");
|
||||||
|
expect(resolveModelTokenizer("deepseek-r1-0528")).toBe("deepseek-v3");
|
||||||
|
expect(resolveModelTokenizer("deepseek-r1-distill-qwen-32b")).toBeUndefined();
|
||||||
|
expect(resolveModelTokenizer("moonshotai/Kimi-K3")).toBe("kimi-k2");
|
||||||
|
expect(resolveModelTokenizer("moonshot-v1-128k")).toBeUndefined();
|
||||||
|
expect(resolveModelTokenizer("glm-5.2")).toBe("glm5");
|
||||||
|
expect(resolveModelTokenizer("glm-4.7")).toBeUndefined();
|
||||||
|
});
|
||||||
|
test("buildModel materializes wire-model tokenizer policy and preserves explicit policy", () => {
|
||||||
|
expect(buildModel(spec("deepseek-v4-pro")).tokenizer).toBe("deepseek-v3");
|
||||||
|
expect(buildModel({ ...spec("alias"), requestModelId: "deepseek-v4-pro" }).tokenizer).toBe("deepseek-v3");
|
||||||
|
expect(buildModel({ ...spec("deepseek-v4-pro"), tokenizer: "qwen3" }).tokenizer).toBe("qwen3");
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -5,11 +5,12 @@
|
|||||||
### Added
|
### Added
|
||||||
|
|
||||||
- Added `qwenTemplateReasoningEffort` to the `models.yml` `compat` schema, so the auto-enabled Qwen 3.8+ template effort dialect (`chat_template_kwargs.reasoning_effort`) can be switched off per provider/model for strict local servers that reject unknown `chat_template_kwargs`.
|
- Added `qwenTemplateReasoningEffort` to the `models.yml` `compat` schema, so the auto-enabled Qwen 3.8+ template effort dialect (`chat_template_kwargs.reasoning_effort`) can be switched off per provider/model for strict local servers that reject unknown `chat_template_kwargs`.
|
||||||
|
- Added `tokenizer` to custom model and `modelOverrides` configuration. It overrides the catalog-resolved local tokenizer family for a model when a proxy serves a known model id with a different tokenizer.
|
||||||
|
|
||||||
### Changed
|
### Changed
|
||||||
|
|
||||||
- `/settings` rows can now carry a risk note: a warning glyph on the row plus a warning-colored line above the description. `External Thinking` (`externalThinking`, `--external-thinking`) is the first user — providers have flagged the request shape it produces as abuse, up to account-level enforcement, so both the settings entry and `--help` now say so.
|
- `/settings` rows can now carry a risk note: a warning glyph on the row plus a warning-colored line above the description. `External Thinking` (`externalThinking`, `--external-thinking`) is the first user — providers have flagged the request shape it produces as abuse, up to account-level enforcement, so both the settings entry and `--help` now say so.
|
||||||
- Token counting is now scoped to the model being billed rather than to a process-global tokenizer: session maintenance, stats, advisors, `/context`, snapcompact inline imaging, and `compress` each count through the owning agent's `Tokenizer` (`agent.tokenizer`). Message counting is `Tokenizer.countMessage`/`countMessages` (replacing the free `estimateTokens(message, tokenizer)` helper; the legacy shim keeps a compat `estimateTokens` export for legacy pi extensions). `estimateToolSchemaTokens`, `estimateSkillsTokens`, `computeNonMessageTokens`, and `computeNonMessageBreakdown` take an explicit tokenizer; `scripts/measure-prompt-tokens.ts` accepts an optional model id (argv) so its numbers match what that model is charged.
|
- Token counting is now scoped to the model being billed rather than to a process-global tokenizer: session maintenance, stats, advisors, `/context`, snapcompact inline imaging, and `compress` each count through the owning agent's `Tokenizer` (`agent.tokenizer`). Message counting is `Tokenizer.countMessage`/`countMessages` (replacing the free `estimateTokens(message, tokenizer)` helper; the legacy shim keeps a compat `estimateTokens` export for legacy pi extensions). `estimateToolSchemaTokens`, `estimateSkillsTokens`, `computeNonMessageTokens`, and `computeNonMessageBreakdown` take an explicit tokenizer; standalone prompt inspection intentionally keeps the default estimate because it has no resolved catalog model.
|
||||||
- The advisor runtime's `maintainContext` hook now receives the pending update as a message instead of a pre-computed token count — sizing it needs the advisor model's tokenizer, which the host owns.
|
- The advisor runtime's `maintainContext` hook now receives the pending update as a message instead of a pre-computed token count — sizing it needs the advisor model's tokenizer, which the host owns.
|
||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|||||||
@@ -14,10 +14,9 @@ function est(s: string): number {
|
|||||||
await Settings.init({ inMemory: true, cwd: process.cwd() });
|
await Settings.init({ inMemory: true, cwd: process.cwd() });
|
||||||
const settings = Settings.isolated({});
|
const settings = Settings.isolated({});
|
||||||
|
|
||||||
// Optional model id (argv[2]) scopes the counter to that model's tokenizer, so
|
// This standalone inspection script has no resolved catalog Model; its counts
|
||||||
// the numbers match what the agent will actually be charged. Without it the
|
// therefore intentionally use the runtime's default estimate policy.
|
||||||
// counts are the fast byte estimate the runtime uses for non-Claude models.
|
const tokenizer = new Tokenizer();
|
||||||
const tokenizer = new Tokenizer(process.argv[2]);
|
|
||||||
|
|
||||||
const session: ToolSession = {
|
const session: ToolSession = {
|
||||||
cwd: process.cwd(),
|
cwd: process.cwd(),
|
||||||
|
|||||||
@@ -75,13 +75,12 @@ export class CompressProtocol {
|
|||||||
#verdict: string | undefined;
|
#verdict: string | undefined;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* `modelId` scopes the token counter to the compressing model. Metrics are
|
* Metrics measure source-vs-draft ratios with the default estimate. The
|
||||||
* source-vs-draft ratios measured with one counter, so they stay coherent
|
* compress session resolves its model after this ledger is constructed, so
|
||||||
* even when the model is unknown at construction time (the session that
|
* no catalog model is available here.
|
||||||
* resolves it is built from this protocol).
|
|
||||||
*/
|
*/
|
||||||
constructor(source: string, modelId?: string | null) {
|
constructor(source: string) {
|
||||||
this.#tokenizer = new Tokenizer(modelId);
|
this.#tokenizer = new Tokenizer();
|
||||||
this.#sourceWords = words(source);
|
this.#sourceWords = words(source);
|
||||||
this.#sourceTokens = this.#tokenizer.countTokens(source);
|
this.#sourceTokens = this.#tokenizer.countTokens(source);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -86,6 +86,7 @@ export function buildCustomModelOverlay(
|
|||||||
thinking: modelDef.thinking,
|
thinking: modelDef.thinking,
|
||||||
input: modelDef.input,
|
input: modelDef.input,
|
||||||
imageInputDecoder: modelDef.imageInputDecoder,
|
imageInputDecoder: modelDef.imageInputDecoder,
|
||||||
|
tokenizer: modelDef.tokenizer,
|
||||||
supportsTools: modelDef.supportsTools,
|
supportsTools: modelDef.supportsTools,
|
||||||
cost: modelDef.cost,
|
cost: modelDef.cost,
|
||||||
contextWindow: modelDef.contextWindow,
|
contextWindow: modelDef.contextWindow,
|
||||||
@@ -136,6 +137,7 @@ export function finalizeCustomModel(model: CustomModelOverlay, options: CustomMo
|
|||||||
headers: resolvedModel.headers,
|
headers: resolvedModel.headers,
|
||||||
omitMaxOutputTokens: resolvedModel.omitMaxOutputTokens ?? reference?.omitMaxOutputTokens,
|
omitMaxOutputTokens: resolvedModel.omitMaxOutputTokens ?? reference?.omitMaxOutputTokens,
|
||||||
compat: mergeCompat(reference?.compatConfig, resolvedModel.compat),
|
compat: mergeCompat(reference?.compatConfig, resolvedModel.compat),
|
||||||
|
tokenizer: resolvedModel.tokenizer,
|
||||||
contextPromotionTarget: resolvedModel.contextPromotionTarget,
|
contextPromotionTarget: resolvedModel.contextPromotionTarget,
|
||||||
compactionModel: resolvedModel.compactionModel,
|
compactionModel: resolvedModel.compactionModel,
|
||||||
remoteCompaction: resolvedModel.remoteCompaction,
|
remoteCompaction: resolvedModel.remoteCompaction,
|
||||||
|
|||||||
@@ -184,6 +184,7 @@ export interface ModelPatch {
|
|||||||
thinking?: ThinkingConfig;
|
thinking?: ThinkingConfig;
|
||||||
input?: ("text" | "image")[];
|
input?: ("text" | "image")[];
|
||||||
imageInputDecoder?: Model<Api>["imageInputDecoder"];
|
imageInputDecoder?: Model<Api>["imageInputDecoder"];
|
||||||
|
tokenizer?: Model<Api>["tokenizer"];
|
||||||
supportsTools?: boolean;
|
supportsTools?: boolean;
|
||||||
cost?: Partial<Model<Api>["cost"]>;
|
cost?: Partial<Model<Api>["cost"]>;
|
||||||
contextWindow?: number;
|
contextWindow?: number;
|
||||||
@@ -211,6 +212,7 @@ export function applyModelPatch(base: Model<Api>, patch: ModelPatch, transport:
|
|||||||
if (patch.reasoning !== undefined) result.reasoning = patch.reasoning;
|
if (patch.reasoning !== undefined) result.reasoning = patch.reasoning;
|
||||||
if (patch.thinking !== undefined) result.thinking = patch.thinking;
|
if (patch.thinking !== undefined) result.thinking = patch.thinking;
|
||||||
if (patch.input !== undefined) result.input = patch.input;
|
if (patch.input !== undefined) result.input = patch.input;
|
||||||
|
if (patch.tokenizer !== undefined) result.tokenizer = patch.tokenizer;
|
||||||
if (patch.imageInputDecoder !== undefined) result.imageInputDecoder = patch.imageInputDecoder;
|
if (patch.imageInputDecoder !== undefined) result.imageInputDecoder = patch.imageInputDecoder;
|
||||||
if (patch.supportsTools !== undefined) result.supportsTools = patch.supportsTools;
|
if (patch.supportsTools !== undefined) result.supportsTools = patch.supportsTools;
|
||||||
if (patch.contextWindow !== undefined) result.contextWindow = patch.contextWindow;
|
if (patch.contextWindow !== undefined) result.contextWindow = patch.contextWindow;
|
||||||
|
|||||||
@@ -132,6 +132,10 @@ export const getModelsConfigSchemaBundle = once(() => {
|
|||||||
};
|
};
|
||||||
});
|
});
|
||||||
|
|
||||||
|
const ModelTokenizerSchema = type(
|
||||||
|
'"claude-v3" | "claude-v47" | "claude-v5" | "claude-v5-sonnet" | "qwen3" | "deepseek-v3" | "kimi-k2" | "glm5"',
|
||||||
|
);
|
||||||
|
|
||||||
const RemoteCompactionSchema = type({
|
const RemoteCompactionSchema = type({
|
||||||
"enabled?": "boolean",
|
"enabled?": "boolean",
|
||||||
"api?": ApiSchema,
|
"api?": ApiSchema,
|
||||||
@@ -169,6 +173,7 @@ export const getModelsConfigSchemaBundle = once(() => {
|
|||||||
"thinking?": ModelThinkingSchema,
|
"thinking?": ModelThinkingSchema,
|
||||||
"input?": '("text" | "image")[]',
|
"input?": '("text" | "image")[]',
|
||||||
"imageInputDecoder?": '"stb"',
|
"imageInputDecoder?": '"stb"',
|
||||||
|
"tokenizer?": ModelTokenizerSchema,
|
||||||
"supportsTools?": "boolean",
|
"supportsTools?": "boolean",
|
||||||
"cost?": {
|
"cost?": {
|
||||||
input: "number",
|
input: "number",
|
||||||
@@ -219,6 +224,7 @@ export const getModelsConfigSchemaBundle = once(() => {
|
|||||||
"thinking?": ModelThinkingSchema,
|
"thinking?": ModelThinkingSchema,
|
||||||
"input?": '("text" | "image")[]',
|
"input?": '("text" | "image")[]',
|
||||||
"imageInputDecoder?": '"stb"',
|
"imageInputDecoder?": '"stb"',
|
||||||
|
"tokenizer?": ModelTokenizerSchema,
|
||||||
"supportsTools?": "boolean",
|
"supportsTools?": "boolean",
|
||||||
"cost?": {
|
"cost?": {
|
||||||
"input?": "number",
|
"input?": "number",
|
||||||
|
|||||||
@@ -282,7 +282,7 @@ export function estimateInlineSavings(input: {
|
|||||||
}
|
}
|
||||||
|
|
||||||
const shape = snapcompact.resolveShape(model, options.shape);
|
const shape = snapcompact.resolveShape(model, options.shape);
|
||||||
const tokenizer = new Tokenizer(model.id);
|
const tokenizer = new Tokenizer(model);
|
||||||
let existingImages = 0;
|
let existingImages = 0;
|
||||||
for (const message of input.messages) {
|
for (const message of input.messages) {
|
||||||
if (!Array.isArray(message.content)) continue;
|
if (!Array.isArray(message.content)) continue;
|
||||||
@@ -421,7 +421,7 @@ export class SnapcompactInlineTransformer {
|
|||||||
if (!model.input.includes("image")) return context;
|
if (!model.input.includes("image")) return context;
|
||||||
|
|
||||||
const shape = snapcompact.resolveShape(model, this.options.shape);
|
const shape = snapcompact.resolveShape(model, this.options.shape);
|
||||||
const tokenizer = new Tokenizer(model.id);
|
const tokenizer = new Tokenizer(model);
|
||||||
const budget = snapcompact.providerImageBudget(model.provider) - countContextImages(context);
|
const budget = snapcompact.providerImageBudget(model.provider) - countContextImages(context);
|
||||||
if (budget <= 0) return context;
|
if (budget <= 0) return context;
|
||||||
|
|
||||||
|
|||||||
@@ -5,6 +5,7 @@
|
|||||||
### Added
|
### Added
|
||||||
|
|
||||||
- Added `ClaudeV3`/`ClaudeV47`/`ClaudeV5` encodings to `countTokens`: a Rust rewrite of [ctok](https://github.com/sanderland/ctok) by Sander Land (MIT), reconstructing Anthropic's `count_tokens` offline. Counts are exact on ctok's ~3.4M-response measurement corpora; the port is validated against 493 Python-ctok reference fixtures covering all three families. The pipeline is byte-level throughout — markers occupy one byte, normalization borrows text no rule touches, ASCII and ideographs skip the Unicode tables, and pieces are matched with one Aho-Corasick transition per byte instead of a per-position vocabulary descent — which counts English prose at 64 MiB/s, markdown at 73 MiB/s, source code at 35 MiB/s and CJK at 49 MiB/s per core: 1.5× (CJK, already cheap per byte) to 5.5× (prose, markdown, digits) a straightforward character-level implementation of the same model, which is held to byte-for-byte identical counts across 2.4M randomized differential comparisons.
|
- Added `ClaudeV3`/`ClaudeV47`/`ClaudeV5` encodings to `countTokens`: a Rust rewrite of [ctok](https://github.com/sanderland/ctok) by Sander Land (MIT), reconstructing Anthropic's `count_tokens` offline. Counts are exact on ctok's ~3.4M-response measurement corpora; the port is validated against 493 Python-ctok reference fixtures covering all three families. The pipeline is byte-level throughout — markers occupy one byte, normalization borrows text no rule touches, ASCII and ideographs skip the Unicode tables, and pieces are matched with one Aho-Corasick transition per byte instead of a per-position vocabulary descent — which counts English prose at 64 MiB/s, markdown at 73 MiB/s, source code at 35 MiB/s and CJK at 49 MiB/s per core: 1.5× (CJK, already cheap per byte) to 5.5× (prose, markdown, digits) a straightforward character-level implementation of the same model, which is held to byte-for-byte identical counts across 2.4M randomized differential comparisons.
|
||||||
|
- Added zstd-embedded exact content tokenizers for Qwen 3.5+/3.6+/3.8, DeepSeek V3/V4/R1, Kimi K2/K3, and GLM-5 alongside the rebuilt OpenAI o200k/cl100k and Claude reconstructions. `countTokens` now reads JavaScript strings through a reusable UTF-16 buffer, so native counting does not allocate a UTF-8 temporary.
|
||||||
|
|
||||||
## [17.3.8] - 2026-08-19
|
## [17.3.8] - 2026-08-19
|
||||||
|
|
||||||
|
|||||||
Vendored
+17
-8
@@ -630,16 +630,17 @@ export declare function cosineSimilarityPairs(vectors: Float64Array, count: numb
|
|||||||
* Count tokens in `input`.
|
* Count tokens in `input`.
|
||||||
*
|
*
|
||||||
* `input` may be a single string or an array of strings; an array returns
|
* `input` may be a single string or an array of strings; an array returns
|
||||||
* the sum across all elements (encoded in parallel via rayon when the global
|
* the sum across all elements (counted in parallel when the global rayon pool
|
||||||
* pool is available). Always returns a single token total — use this for any
|
* is available). Always returns a single token total — use this for any
|
||||||
* aggregate budget question without paying a per-element napi crossing.
|
* aggregate budget question without paying a per-element napi crossing.
|
||||||
*
|
*
|
||||||
* Measures user/model content, not wire-protocol tokens: BPE encodings use
|
* Measures user/model content, not wire-protocol tokens: BPE encodings
|
||||||
* ordinary encoding (no special-token handling) and the Claude encodings
|
* use ordinary encoding (no special-token handling) and the Claude
|
||||||
* count message content without the fixed per-message frame. Defaults to
|
* encodings count message content without the fixed per-message frame.
|
||||||
* `o200k_base`; pass a `Claude*` encoding for exact Claude counts.
|
* Defaults to `o200k_base`; pass a `Claude*` encoding for exact Claude
|
||||||
|
* counts, or the matching family encoding for Qwen/DeepSeek/Kimi/GLM.
|
||||||
*/
|
*/
|
||||||
export declare function countTokens(input: string | Array<string>, encoding?: Encoding | undefined | null): number
|
export declare function countTokens(input: string | string[], encoding?: Encoding | undefined | null): number
|
||||||
|
|
||||||
export interface DesktopCapabilities {
|
export interface DesktopCapabilities {
|
||||||
backend: string
|
backend: string
|
||||||
@@ -862,7 +863,15 @@ export declare enum Encoding {
|
|||||||
/** Claude Opus 5+ (ctok v5 reconstruction). */
|
/** Claude Opus 5+ (ctok v5 reconstruction). */
|
||||||
ClaudeV5 = 'ClaudeV5',
|
ClaudeV5 = 'ClaudeV5',
|
||||||
/** Claude Sonnet/Fable 5+ (live-measured non-opus v5 frame). */
|
/** Claude Sonnet/Fable 5+ (live-measured non-opus v5 frame). */
|
||||||
ClaudeV5Sonnet = 'ClaudeV5Sonnet'
|
ClaudeV5Sonnet = 'ClaudeV5Sonnet',
|
||||||
|
/** Qwen 3.5 / 3.6 / 3.8 (248k vocabulary). */
|
||||||
|
Qwen3 = 'Qwen3',
|
||||||
|
/** `DeepSeek` V3 … V4 (identical base BPE). */
|
||||||
|
DeepSeekV3 = 'DeepSeekV3',
|
||||||
|
/** Kimi K2 … K3. */
|
||||||
|
KimiK2 = 'KimiK2',
|
||||||
|
/** GLM-5.x exact; GLM-4.x near-exact. */
|
||||||
|
Glm5 = 'Glm5'
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -105,6 +105,10 @@ export const Encoding = {
|
|||||||
ClaudeV47: "ClaudeV47",
|
ClaudeV47: "ClaudeV47",
|
||||||
ClaudeV5: "ClaudeV5",
|
ClaudeV5: "ClaudeV5",
|
||||||
ClaudeV5Sonnet: "ClaudeV5Sonnet",
|
ClaudeV5Sonnet: "ClaudeV5Sonnet",
|
||||||
|
Qwen3: "Qwen3",
|
||||||
|
DeepSeekV3: "DeepSeekV3",
|
||||||
|
KimiK2: "KimiK2",
|
||||||
|
Glm5: "Glm5",
|
||||||
};
|
};
|
||||||
export const FileType = {
|
export const FileType = {
|
||||||
File: 1,
|
File: 1,
|
||||||
|
|||||||
@@ -37,7 +37,6 @@
|
|||||||
"test": "bun test --parallel",
|
"test": "bun test --parallel",
|
||||||
"fix": "biome check --write --unsafe .",
|
"fix": "biome check --write --unsafe .",
|
||||||
"fmt": "biome format --write .",
|
"fmt": "biome format --write .",
|
||||||
"gen:ctok": "bun scripts/gen-ctok-vocab.ts",
|
|
||||||
"gen:native": "bun scripts/embed-native.ts",
|
"gen:native": "bun scripts/embed-native.ts",
|
||||||
"gen:native:reset": "bun scripts/embed-native.ts --reset",
|
"gen:native:reset": "bun scripts/embed-native.ts --reset",
|
||||||
"gen:npm": "bun scripts/gen-npm-packages.ts",
|
"gen:npm": "bun scripts/gen-npm-packages.ts",
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user