diff --git a/Cargo.lock b/Cargo.lock index e6dc532b5..f097c88e9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -144,7 +144,7 @@ version = "0.39.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "057ae90e7256ebf85f840b1638268df0142c9d19467d500b790631fd301acc27" dependencies = [ - "bit-set", + "bit-set 0.8.0", "regex", "thiserror 2.0.18", "tree-sitter", @@ -215,15 +215,30 @@ dependencies = [ "serde", ] +[[package]] +name = "bit-set" +version = "0.5.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0700ddab506f33b20a03b13996eccd309a48e5ff77d0d95926aa0210fb4e95f1" +dependencies = [ + "bit-vec 0.6.3", +] + [[package]] name = "bit-set" version = "0.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "08807e080ed7f9d5433fa9b275196cfc35414f66a0c79d864dc51a0d825231a3" dependencies = [ - "bit-vec", + "bit-vec 0.8.0", ] +[[package]] +name = "bit-vec" +version = "0.6.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "349f9b6a179ed607305526ca489b34ad0a41aed5f7980fa90eb03160b69598fb" + [[package]] name = "bit-vec" version = "0.8.0" @@ -288,7 +303,7 @@ dependencies = [ "cfg-if", "chrono", "clap", - "fancy-regex", + "fancy-regex 0.16.2", "futures", "itertools", "nix 0.30.1", @@ -315,7 +330,7 @@ dependencies = [ "chrono", "clap", "command-fds", - "fancy-regex", + "fancy-regex 0.16.2", "futures", "getrandom 0.3.4", "homedir", @@ -794,13 +809,24 @@ version = "3.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "dea2df4cf52843e0452895c455a1a2cfbb842a1e7329671acf418fdc53ed4c59" +[[package]] +name = "fancy-regex" +version = "0.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "531e46835a22af56d1e3b66f04844bed63158bc094a628bec1d321d9b4c44bf2" +dependencies = [ + "bit-set 0.5.3", + "regex-automata", + "regex-syntax", +] + [[package]] name = "fancy-regex" version = "0.16.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "998b056554fbe42e03ae0e152895cd1a7e1002aec800fdc6635d20270260c46f" dependencies = [ - "bit-set", + "bit-set 0.8.0", "regex-automata", "regex-syntax", ] @@ -885,7 +911,7 @@ dependencies = [ "fluent-syntax", "intl-memoizer", "intl_pluralrules", - "rustc-hash", + "rustc-hash 2.1.2", "self_cell", "smallvec", "unic-langid", @@ -1663,7 +1689,7 @@ dependencies = [ "napi-build", "napi-sys", "nohash-hasher", - "rustc-hash", + "rustc-hash 2.1.2", "tokio", ] @@ -2172,6 +2198,7 @@ dependencies = [ "similar", "smallvec", "syntect", + "tiktoken-rs", "tokio", "tokio-util", "toml", @@ -2589,6 +2616,12 @@ dependencies = [ "archery", ] +[[package]] +name = "rustc-hash" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "08d43f7aa6b08d49f382cde6a7982047c3426db949b1424bc4b7ec9ae12c6ce2" + [[package]] name = "rustc-hash" version = "2.1.2" @@ -2887,7 +2920,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "656b45c05d95a5704399aeef6bd0ddec7b2b3531b7c9e900abbf7c4d2190c925" dependencies = [ "bincode", - "fancy-regex", + "fancy-regex 0.16.2", "flate2", "fnv", "once_cell", @@ -2990,6 +3023,21 @@ dependencies = [ "zune-jpeg", ] +[[package]] +name = "tiktoken-rs" +version = "0.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "25563eeba904d770acf527e8b370fe9a5547bacd20ff84a0b6c3bc41288e5625" +dependencies = [ + "anyhow", + "base64", + "bstr", + "fancy-regex 0.13.0", + "lazy_static", + "regex", + "rustc-hash 1.1.0", +] + [[package]] name = "tiny-keccak" version = "2.0.2" @@ -3722,7 +3770,7 @@ version = "0.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cb30dbbd9036155e74adad6812e9898d03ec374946234fbcebd5dfc7b9187b90" dependencies = [ - "rustc-hash", + "rustc-hash 2.1.2", ] [[package]] diff --git a/crates/pi-natives/Cargo.toml b/crates/pi-natives/Cargo.toml index 6156ac043..216f0ec45 100644 --- a/crates/pi-natives/Cargo.toml +++ b/crates/pi-natives/Cargo.toml @@ -119,6 +119,7 @@ smallvec = { version = "1.15.1", features = [ memmap2 = "0.9" xxhash-rust = { version = "0.8", features = ["xxh64"] } regex = "1" +tiktoken-rs = "0.7" similar = "3.0.0" serde = { version = "1.0", features = ["derive"] } serde_json = { version = "1.0", features = ["preserve_order"] } diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index d89dca487..53589d38a 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -48,4 +48,5 @@ pub mod pty; pub mod shell; pub mod task; pub mod text; +pub mod tokens; pub(crate) mod utils; diff --git a/crates/pi-natives/src/tokens.rs b/crates/pi-natives/src/tokens.rs new file mode 100644 index 000000000..03270451b --- /dev/null +++ b/crates/pi-natives/src/tokens.rs @@ -0,0 +1,65 @@ +//! Token counting via tiktoken-rs. +//! +//! Two encodings are exposed: +//! +//! - `O200kBase` — GPT-4o / o1 / GPT-5 (the modern OpenAI default). +//! - `Cl100kBase` — GPT-3.5 / GPT-4 / older models. +//! +//! `o200k_base` is the default. Anthropic doesn't publish their tokenizer, so +//! either of these is an approximation for Claude (within ~5–10% across +//! English/code text). `o200k_base` is closer to current frontier models' +//! actual segmentation and is the right default for budget estimates. +//! +//! Both BPE tables are embedded in the binary; encoders are built once on +//! first use and reused thereafter. + +use std::sync::LazyLock; + +use napi::bindgen_prelude::Either; +use napi_derive::napi; +use rayon::prelude::*; +use tiktoken_rs::{CoreBPE, cl100k_base, o200k_base}; + +/// Tokenizer encoding to use. +#[napi(string_enum)] +pub enum Encoding { + /// GPT-4o / o1 / GPT-5 (default). + O200kBase, + /// GPT-3.5 / GPT-4 / older. + Cl100kBase, +} + +static O200K: LazyLock = + LazyLock::new(|| o200k_base().expect("failed to initialize o200k_base BPE tables")); + +static CL100K: LazyLock = + LazyLock::new(|| cl100k_base().expect("failed to initialize cl100k_base BPE tables")); + +fn encoder(encoding: Option) -> &'static CoreBPE { + match encoding.unwrap_or(Encoding::O200kBase) { + Encoding::O200kBase => &O200K, + Encoding::Cl100kBase => &CL100K, + } +} + +/// Count tokens in `input`. +/// +/// `input` may be a single string or an array of strings; an array returns +/// the sum across all elements (encoded in parallel via rayon). Always +/// returns a single token total — use this for any aggregate budget question +/// without paying a per-element napi crossing. +/// +/// Uses ordinary encoding (no special-token handling), which is the right +/// choice for measuring user/model content rather than wire-protocol tokens. +/// Defaults to `o200k_base`; pass `Cl100kBase` for older OpenAI models. +#[napi] +pub fn count_tokens(input: Either>, encoding: Option) -> u32 { + let bpe = encoder(encoding); + match input { + Either::A(text) => bpe.encode_ordinary(&text).len() as u32, + Either::B(texts) => texts + .par_iter() + .map(|s| bpe.encode_ordinary(s).len() as u32) + .sum(), + } +} diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 3dca5c6c4..599cd2e8a 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -310,6 +310,20 @@ export interface ContextLine { */ export declare function copyToClipboard(text: string): void +/** + * Count tokens in `input`. + * + * `input` may be a single string or an array of strings; an array returns + * the sum across all elements (encoded in parallel via rayon). Always + * returns a single token total — use this for any aggregate budget question + * without paying a per-element napi crossing. + * + * Uses ordinary encoding (no special-token handling), which is the right + * choice for measuring user/model content rather than wire-protocol tokens. + * Defaults to `o200k_base`; pass `Cl100kBase` for older OpenAI models. + */ +export declare function countTokens(input: string | Array, encoding?: Encoding | undefined | null): number + /** * Detect macOS system appearance via CoreFoundation. * Returns `"dark"` or `"light"` on macOS, `null` on other platforms. @@ -337,6 +351,14 @@ export declare enum Ellipsis { */ export declare function encodeSixel(bytes: Uint8Array, targetWidthPx: number, targetHeightPx: number): string +/** Tokenizer encoding to use. */ +export declare enum Encoding { + /** GPT-4o / o1 / GPT-5 (default). */ + O200kBase = 'O200kBase', + /** GPT-3.5 / GPT-4 / older. */ + Cl100kBase = 'Cl100kBase' +} + /** * Execute a brush shell command. * diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index ef7b00235..3703ad5f6 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -238,6 +238,10 @@ module.exports.Ellipsis = { Ascii: 1, Omit: 2, }; +module.exports.Encoding = { + O200kBase: 'O200kBase', + Cl100kBase: 'Cl100kBase', +}; module.exports.FileType = { File: 1, Dir: 2,