3f304ee5a8
- Extracted native token counting into a new localized `tokenizer.ts` wrapping `@oh-my-pi/pi-natives`. - Introduced a lightning-fast byte-length estimation logic for token counting when accurate counting is disabled. - Diverted token calculations to the faster estimator during test environments and when `PI_TOKENIZER_ACCURATE` is falsy. - Updated agent base and coding-agent sessions to consume the new localized `countTokens` utility.
18 lines
513 B
TypeScript
18 lines
513 B
TypeScript
import { countTokens as countTokensNat } from "@oh-my-pi/pi-natives";
|
|
|
|
const accurate = process.env.PI_TOKENIZER_ACCURATE === "1" && Bun.env.NODE_ENV !== "test";
|
|
|
|
function estimateTokens(text: string) {
|
|
return (Buffer.byteLength(text, "utf-8") + 3) >> 2;
|
|
}
|
|
|
|
export function countTokens(text: string | string[]): number {
|
|
if (accurate) {
|
|
return countTokensNat(text);
|
|
} else if (Array.isArray(text)) {
|
|
return text.reduce((sum, t) => sum + estimateTokens(t), 0);
|
|
} else {
|
|
return estimateTokens(text);
|
|
}
|
|
}
|