Files
oh-my-pi/crates/pi-natives/tools/pack-openai.ts
T
can1357 0cdd37fc15 feat(pi-natives/tools): implemented utok tokenizer for multiple models
- Replaced the `ctok` implementation with the `utok` universal tokenizer supporting multiple model families and UTF text encodings.
- Added tokenizer support and embedding data for Qwen3, DeepSeek V3, Kimi K2, and GLM-5 model variants.
- Added fixture generation scripts, vocabulary packers, and golden test suites for validating tokenization parity.
- Updated dependency requirements and Bazel workspace definitions for new crates and tools.
2026-08-20 01:45:04 +02:00

58 lines
1.7 KiB
TypeScript

// Pack OpenAI tiktoken rank files into UTOK1 + zstd -19 blobs.
//
// bun tools/pack-openai.ts
//
// Reads tools/cache/{o200k_base,cl100k_base}.tiktoken (base64-token + rank
// per line), asserts rank contiguity, writes data/<name>.bin.zst.
const root = new URL("..", import.meta.url).pathname;
function varint(n: number): number[] {
const out: number[] = [];
while (n >= 0x80) {
out.push((n & 0x7f) | 0x80);
n >>>= 7;
}
out.push(n);
return out;
}
async function pack(name: string, expected: number) {
const text = await Bun.file(`${root}tools/cache/${name}.tiktoken`).text();
const lines = text.split("\n").filter((l) => l.length > 0);
if (lines.length !== expected) {
throw new Error(`${name}: expected ${expected} entries, got ${lines.length}`);
}
const chunks: Uint8Array[] = [];
let total = 0;
const push = (b: Uint8Array) => {
chunks.push(b);
total += b.length;
};
const header = new Uint8Array(10);
header.set(new TextEncoder().encode("UTOK1\n"), 0);
new DataView(header.buffer).setUint32(6, lines.length, true);
push(header);
for (let rank = 0; rank < lines.length; rank++) {
const [b64, rankStr] = lines[rank].split(" ");
if (Number(rankStr) !== rank) {
throw new Error(`${name}: rank discontinuity at line ${rank}: got ${rankStr}`);
}
const token = Uint8Array.fromBase64(b64);
push(new Uint8Array(varint(token.length)));
push(token);
}
const raw = new Uint8Array(total);
let off = 0;
for (const c of chunks) {
raw.set(c, off);
off += c.length;
}
const zst = Bun.zstdCompressSync(raw, { level: 19 });
await Bun.write(`${root}data/${name}.bin.zst`, zst);
console.log(`${name}: ${lines.length} entries, ${raw.length} raw -> ${zst.length} zst`);
}
await pack("o200k_base", 199998);
await pack("cl100k_base", 100256);