feat(pi-natives): added token counting API with O200k and Cl100k encodings
- Added a new `tokens` module in `crates/pi-natives` using `tiktoken-rs`, exposing `count_tokens` and `count_tokens_batch` through N-API with `Encoding::O200kBase` as the default. - Exported the `Encoding` enum plus token counting functions in `packages/natives/native` TypeScript declarations and JS runtime bindings. - Registered `tiktoken-rs` in `crates/pi-natives/Cargo.toml` and updated `Cargo.lock` with its resolved dependency entries.
This commit is contained in:
Generated
+57
-9
@@ -144,7 +144,7 @@ version = "0.39.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "057ae90e7256ebf85f840b1638268df0142c9d19467d500b790631fd301acc27"
|
||||
dependencies = [
|
||||
"bit-set",
|
||||
"bit-set 0.8.0",
|
||||
"regex",
|
||||
"thiserror 2.0.18",
|
||||
"tree-sitter",
|
||||
@@ -215,15 +215,30 @@ dependencies = [
|
||||
"serde",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "bit-set"
|
||||
version = "0.5.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0700ddab506f33b20a03b13996eccd309a48e5ff77d0d95926aa0210fb4e95f1"
|
||||
dependencies = [
|
||||
"bit-vec 0.6.3",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "bit-set"
|
||||
version = "0.8.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "08807e080ed7f9d5433fa9b275196cfc35414f66a0c79d864dc51a0d825231a3"
|
||||
dependencies = [
|
||||
"bit-vec",
|
||||
"bit-vec 0.8.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "bit-vec"
|
||||
version = "0.6.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "349f9b6a179ed607305526ca489b34ad0a41aed5f7980fa90eb03160b69598fb"
|
||||
|
||||
[[package]]
|
||||
name = "bit-vec"
|
||||
version = "0.8.0"
|
||||
@@ -288,7 +303,7 @@ dependencies = [
|
||||
"cfg-if",
|
||||
"chrono",
|
||||
"clap",
|
||||
"fancy-regex",
|
||||
"fancy-regex 0.16.2",
|
||||
"futures",
|
||||
"itertools",
|
||||
"nix 0.30.1",
|
||||
@@ -315,7 +330,7 @@ dependencies = [
|
||||
"chrono",
|
||||
"clap",
|
||||
"command-fds",
|
||||
"fancy-regex",
|
||||
"fancy-regex 0.16.2",
|
||||
"futures",
|
||||
"getrandom 0.3.4",
|
||||
"homedir",
|
||||
@@ -794,13 +809,24 @@ version = "3.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dea2df4cf52843e0452895c455a1a2cfbb842a1e7329671acf418fdc53ed4c59"
|
||||
|
||||
[[package]]
|
||||
name = "fancy-regex"
|
||||
version = "0.13.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "531e46835a22af56d1e3b66f04844bed63158bc094a628bec1d321d9b4c44bf2"
|
||||
dependencies = [
|
||||
"bit-set 0.5.3",
|
||||
"regex-automata",
|
||||
"regex-syntax",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "fancy-regex"
|
||||
version = "0.16.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "998b056554fbe42e03ae0e152895cd1a7e1002aec800fdc6635d20270260c46f"
|
||||
dependencies = [
|
||||
"bit-set",
|
||||
"bit-set 0.8.0",
|
||||
"regex-automata",
|
||||
"regex-syntax",
|
||||
]
|
||||
@@ -885,7 +911,7 @@ dependencies = [
|
||||
"fluent-syntax",
|
||||
"intl-memoizer",
|
||||
"intl_pluralrules",
|
||||
"rustc-hash",
|
||||
"rustc-hash 2.1.2",
|
||||
"self_cell",
|
||||
"smallvec",
|
||||
"unic-langid",
|
||||
@@ -1663,7 +1689,7 @@ dependencies = [
|
||||
"napi-build",
|
||||
"napi-sys",
|
||||
"nohash-hasher",
|
||||
"rustc-hash",
|
||||
"rustc-hash 2.1.2",
|
||||
"tokio",
|
||||
]
|
||||
|
||||
@@ -2172,6 +2198,7 @@ dependencies = [
|
||||
"similar",
|
||||
"smallvec",
|
||||
"syntect",
|
||||
"tiktoken-rs",
|
||||
"tokio",
|
||||
"tokio-util",
|
||||
"toml",
|
||||
@@ -2589,6 +2616,12 @@ dependencies = [
|
||||
"archery",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustc-hash"
|
||||
version = "1.1.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "08d43f7aa6b08d49f382cde6a7982047c3426db949b1424bc4b7ec9ae12c6ce2"
|
||||
|
||||
[[package]]
|
||||
name = "rustc-hash"
|
||||
version = "2.1.2"
|
||||
@@ -2887,7 +2920,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "656b45c05d95a5704399aeef6bd0ddec7b2b3531b7c9e900abbf7c4d2190c925"
|
||||
dependencies = [
|
||||
"bincode",
|
||||
"fancy-regex",
|
||||
"fancy-regex 0.16.2",
|
||||
"flate2",
|
||||
"fnv",
|
||||
"once_cell",
|
||||
@@ -2990,6 +3023,21 @@ dependencies = [
|
||||
"zune-jpeg",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tiktoken-rs"
|
||||
version = "0.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "25563eeba904d770acf527e8b370fe9a5547bacd20ff84a0b6c3bc41288e5625"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"base64",
|
||||
"bstr",
|
||||
"fancy-regex 0.13.0",
|
||||
"lazy_static",
|
||||
"regex",
|
||||
"rustc-hash 1.1.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tiny-keccak"
|
||||
version = "2.0.2"
|
||||
@@ -3722,7 +3770,7 @@ version = "0.5.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cb30dbbd9036155e74adad6812e9898d03ec374946234fbcebd5dfc7b9187b90"
|
||||
dependencies = [
|
||||
"rustc-hash",
|
||||
"rustc-hash 2.1.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
||||
@@ -119,6 +119,7 @@ smallvec = { version = "1.15.1", features = [
|
||||
memmap2 = "0.9"
|
||||
xxhash-rust = { version = "0.8", features = ["xxh64"] }
|
||||
regex = "1"
|
||||
tiktoken-rs = "0.7"
|
||||
similar = "3.0.0"
|
||||
serde = { version = "1.0", features = ["derive"] }
|
||||
serde_json = { version = "1.0", features = ["preserve_order"] }
|
||||
|
||||
@@ -48,4 +48,5 @@ pub mod pty;
|
||||
pub mod shell;
|
||||
pub mod task;
|
||||
pub mod text;
|
||||
pub mod tokens;
|
||||
pub(crate) mod utils;
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
//! Token counting via tiktoken-rs.
|
||||
//!
|
||||
//! Two encodings are exposed:
|
||||
//!
|
||||
//! - `O200kBase` — GPT-4o / o1 / GPT-5 (the modern OpenAI default).
|
||||
//! - `Cl100kBase` — GPT-3.5 / GPT-4 / older models.
|
||||
//!
|
||||
//! `o200k_base` is the default. Anthropic doesn't publish their tokenizer, so
|
||||
//! either of these is an approximation for Claude (within ~5–10% across
|
||||
//! English/code text). `o200k_base` is closer to current frontier models'
|
||||
//! actual segmentation and is the right default for budget estimates.
|
||||
//!
|
||||
//! Both BPE tables are embedded in the binary; encoders are built once on
|
||||
//! first use and reused thereafter.
|
||||
|
||||
use std::sync::LazyLock;
|
||||
|
||||
use napi::bindgen_prelude::Either;
|
||||
use napi_derive::napi;
|
||||
use rayon::prelude::*;
|
||||
use tiktoken_rs::{CoreBPE, cl100k_base, o200k_base};
|
||||
|
||||
/// Tokenizer encoding to use.
|
||||
#[napi(string_enum)]
|
||||
pub enum Encoding {
|
||||
/// GPT-4o / o1 / GPT-5 (default).
|
||||
O200kBase,
|
||||
/// GPT-3.5 / GPT-4 / older.
|
||||
Cl100kBase,
|
||||
}
|
||||
|
||||
static O200K: LazyLock<CoreBPE> =
|
||||
LazyLock::new(|| o200k_base().expect("failed to initialize o200k_base BPE tables"));
|
||||
|
||||
static CL100K: LazyLock<CoreBPE> =
|
||||
LazyLock::new(|| cl100k_base().expect("failed to initialize cl100k_base BPE tables"));
|
||||
|
||||
fn encoder(encoding: Option<Encoding>) -> &'static CoreBPE {
|
||||
match encoding.unwrap_or(Encoding::O200kBase) {
|
||||
Encoding::O200kBase => &O200K,
|
||||
Encoding::Cl100kBase => &CL100K,
|
||||
}
|
||||
}
|
||||
|
||||
/// Count tokens in `input`.
|
||||
///
|
||||
/// `input` may be a single string or an array of strings; an array returns
|
||||
/// the sum across all elements (encoded in parallel via rayon). Always
|
||||
/// returns a single token total — use this for any aggregate budget question
|
||||
/// without paying a per-element napi crossing.
|
||||
///
|
||||
/// Uses ordinary encoding (no special-token handling), which is the right
|
||||
/// choice for measuring user/model content rather than wire-protocol tokens.
|
||||
/// Defaults to `o200k_base`; pass `Cl100kBase` for older OpenAI models.
|
||||
#[napi]
|
||||
pub fn count_tokens(input: Either<String, Vec<String>>, encoding: Option<Encoding>) -> u32 {
|
||||
let bpe = encoder(encoding);
|
||||
match input {
|
||||
Either::A(text) => bpe.encode_ordinary(&text).len() as u32,
|
||||
Either::B(texts) => texts
|
||||
.par_iter()
|
||||
.map(|s| bpe.encode_ordinary(s).len() as u32)
|
||||
.sum(),
|
||||
}
|
||||
}
|
||||
Vendored
+22
@@ -310,6 +310,20 @@ export interface ContextLine {
|
||||
*/
|
||||
export declare function copyToClipboard(text: string): void
|
||||
|
||||
/**
|
||||
* Count tokens in `input`.
|
||||
*
|
||||
* `input` may be a single string or an array of strings; an array returns
|
||||
* the sum across all elements (encoded in parallel via rayon). Always
|
||||
* returns a single token total — use this for any aggregate budget question
|
||||
* without paying a per-element napi crossing.
|
||||
*
|
||||
* Uses ordinary encoding (no special-token handling), which is the right
|
||||
* choice for measuring user/model content rather than wire-protocol tokens.
|
||||
* Defaults to `o200k_base`; pass `Cl100kBase` for older OpenAI models.
|
||||
*/
|
||||
export declare function countTokens(input: string | Array<string>, encoding?: Encoding | undefined | null): number
|
||||
|
||||
/**
|
||||
* Detect macOS system appearance via CoreFoundation.
|
||||
* Returns `"dark"` or `"light"` on macOS, `null` on other platforms.
|
||||
@@ -337,6 +351,14 @@ export declare enum Ellipsis {
|
||||
*/
|
||||
export declare function encodeSixel(bytes: Uint8Array, targetWidthPx: number, targetHeightPx: number): string
|
||||
|
||||
/** Tokenizer encoding to use. */
|
||||
export declare enum Encoding {
|
||||
/** GPT-4o / o1 / GPT-5 (default). */
|
||||
O200kBase = 'O200kBase',
|
||||
/** GPT-3.5 / GPT-4 / older. */
|
||||
Cl100kBase = 'Cl100kBase'
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute a brush shell command.
|
||||
*
|
||||
|
||||
@@ -238,6 +238,10 @@ module.exports.Ellipsis = {
|
||||
Ascii: 1,
|
||||
Omit: 2,
|
||||
};
|
||||
module.exports.Encoding = {
|
||||
O200kBase: 'O200kBase',
|
||||
Cl100kBase: 'Cl100kBase',
|
||||
};
|
||||
module.exports.FileType = {
|
||||
File: 1,
|
||||
Dir: 2,
|
||||
|
||||
Reference in New Issue
Block a user