Files
oh-my-pi/packages/agent/test/tokenizer.test.ts
T
can1357 0cdd37fc15 feat(pi-natives/tools): implemented utok tokenizer for multiple models
- Replaced the `ctok` implementation with the `utok` universal tokenizer supporting multiple model families and UTF text encodings.
- Added tokenizer support and embedding data for Qwen3, DeepSeek V3, Kimi K2, and GLM-5 model variants.
- Added fixture generation scripts, vocabulary packers, and golden test suites for validating tokenization parity.
- Updated dependency requirements and Bazel workspace definitions for new crates and tools.
2026-08-20 01:45:04 +02:00

84 lines
3.7 KiB
TypeScript

import { describe, expect, test } from "bun:test";
import { Encoding } from "@oh-my-pi/pi-natives";
import { tokenizerEncodingForModel, Tokenizer } from "../src/tokenizer";
// Contract: the catalog resolves model identity once as Model.tokenizer; the
// agent maps that catalog property to the matching native counter. A wrong
// row silently skews every context-budget and compaction decision.
describe("tokenizerEncodingForModel", () => {
test("maps every catalog tokenizer family to its native counter", () => {
expect(tokenizerEncodingForModel({ tokenizer: "claude-v3" })).toBe(Encoding.ClaudeV3);
expect(tokenizerEncodingForModel({ tokenizer: "claude-v47" })).toBe(Encoding.ClaudeV47);
expect(tokenizerEncodingForModel({ tokenizer: "claude-v5" })).toBe(Encoding.ClaudeV5);
expect(tokenizerEncodingForModel({ tokenizer: "claude-v5-sonnet" })).toBe(Encoding.ClaudeV5Sonnet);
expect(tokenizerEncodingForModel({ tokenizer: "qwen3" })).toBe(Encoding.Qwen3);
expect(tokenizerEncodingForModel({ tokenizer: "deepseek-v3" })).toBe(Encoding.DeepSeekV3);
expect(tokenizerEncodingForModel({ tokenizer: "kimi-k2" })).toBe(Encoding.KimiK2);
expect(tokenizerEncodingForModel({ tokenizer: "glm5" })).toBe(Encoding.Glm5);
});
test("leaves unknown catalog models on the estimate policy", () => {
expect(tokenizerEncodingForModel({})).toBeNull();
expect(tokenizerEncodingForModel(undefined)).toBeNull();
});
});
describe("Tokenizer", () => {
test("defaults to null encoding and byte estimation", () => {
const tokenizer = new Tokenizer();
expect(tokenizer.encoding).toBeNull();
expect(tokenizer.countTokens("hello world")).toBe(3);
});
test("encoding is fixed at construction from the catalog model", () => {
expect(new Tokenizer({ tokenizer: "claude-v47" }).encoding).toBe(Encoding.ClaudeV47);
expect(new Tokenizer({ tokenizer: "claude-v5" }).encoding).toBe(Encoding.ClaudeV5);
expect(new Tokenizer({}).encoding).toBeNull();
expect(new Tokenizer(undefined).encoding).toBeNull();
});
test("separate instances do not interfere with each other", () => {
const t1 = new Tokenizer({ tokenizer: "claude-v47" });
const t2 = new Tokenizer({ tokenizer: "qwen3" });
const t3 = new Tokenizer({});
expect(t1.encoding).toBe(Encoding.ClaudeV47);
expect(t2.encoding).toBe(Encoding.Qwen3);
expect(t3.encoding).toBeNull();
const t4 = new Tokenizer({ tokenizer: "claude-v3" });
expect(t4.encoding).toBe(Encoding.ClaudeV3);
expect(t1.encoding).toBe(Encoding.ClaudeV47);
expect(t2.encoding).toBe(Encoding.Qwen3);
expect(t3.encoding).toBeNull();
});
});
describe("countTokens with modes", () => {
test("approximate mode uses fast estimation", () => {
const tokenizer = new Tokenizer();
expect(tokenizer.countTokens("hello world", "approximate")).toBe(3);
});
test("upperbound mode uses byte length", () => {
const tokenizer = new Tokenizer();
expect(tokenizer.countTokens("hello world", "upperbound")).toBe(11);
});
test("strict mode uses native counting regardless of encoding", () => {
const noEncoding = new Tokenizer();
expect(noEncoding.countTokens("hello world", "strict")).toBe(2);
const claudeEncoding = new Tokenizer({ tokenizer: "claude-v47" });
expect(claudeEncoding.countTokens("hello world", "strict")).toBeGreaterThan(0);
});
test("mode is per-call; encoding stays independently model-scoped in strict mode", () => {
// approximate/upperbound skip the encoding entirely under NODE_ENV=test
// (fast estimate for a snappy suite); strict is testEnv-independent, so
// it is the mode that proves per-instance encoding isolation here.
const claude = new Tokenizer({ tokenizer: "claude-v47" });
const generic = new Tokenizer({});
expect(claude.countTokens("hello world", "strict")).not.toBe(generic.countTokens("hello world", "strict"));
});
});