feat(pi-natives/tools): implemented utok tokenizer for multiple models
- Replaced the `ctok` implementation with the `utok` universal tokenizer supporting multiple model families and UTF text encodings. - Added tokenizer support and embedding data for Qwen3, DeepSeek V3, Kimi K2, and GLM-5 model variants. - Added fixture generation scripts, vocabulary packers, and golden test suites for validating tokenization parity. - Updated dependency requirements and Bazel workspace definitions for new crates and tools.
This commit is contained in:
@@ -14,10 +14,9 @@ function est(s: string): number {
|
||||
await Settings.init({ inMemory: true, cwd: process.cwd() });
|
||||
const settings = Settings.isolated({});
|
||||
|
||||
// Optional model id (argv[2]) scopes the counter to that model's tokenizer, so
|
||||
// the numbers match what the agent will actually be charged. Without it the
|
||||
// counts are the fast byte estimate the runtime uses for non-Claude models.
|
||||
const tokenizer = new Tokenizer(process.argv[2]);
|
||||
// This standalone inspection script has no resolved catalog Model; its counts
|
||||
// therefore intentionally use the runtime's default estimate policy.
|
||||
const tokenizer = new Tokenizer();
|
||||
|
||||
const session: ToolSession = {
|
||||
cwd: process.cwd(),
|
||||
|
||||
Reference in New Issue
Block a user