feat(pi-natives/tools): implemented utok tokenizer for multiple models

- Replaced the `ctok` implementation with the `utok` universal tokenizer supporting multiple model families and UTF text encodings.
- Added tokenizer support and embedding data for Qwen3, DeepSeek V3, Kimi K2, and GLM-5 model variants.
- Added fixture generation scripts, vocabulary packers, and golden test suites for validating tokenization parity.
- Updated dependency requirements and Bazel workspace definitions for new crates and tools.
This commit is contained in:
can1357
2026-08-20 00:37:05 +02:00
parent 37fd2dbbe1
commit 0cdd37fc15
101 changed files with 746591 additions and 278 deletions
@@ -14,10 +14,9 @@ function est(s: string): number {
await Settings.init({ inMemory: true, cwd: process.cwd() });
const settings = Settings.isolated({});
// Optional model id (argv[2]) scopes the counter to that model's tokenizer, so
// the numbers match what the agent will actually be charged. Without it the
// counts are the fast byte estimate the runtime uses for non-Claude models.
const tokenizer = new Tokenizer(process.argv[2]);
// This standalone inspection script has no resolved catalog Model; its counts
// therefore intentionally use the runtime's default estimate policy.
const tokenizer = new Tokenizer();
const session: ToolSession = {
cwd: process.cwd(),