feat(pi-natives/tools): implemented utok tokenizer for multiple models

- Replaced the `ctok` implementation with the `utok` universal tokenizer supporting multiple model families and UTF text encodings.
- Added tokenizer support and embedding data for Qwen3, DeepSeek V3, Kimi K2, and GLM-5 model variants.
- Added fixture generation scripts, vocabulary packers, and golden test suites for validating tokenization parity.
- Updated dependency requirements and Bazel workspace definitions for new crates and tools.
This commit is contained in:
can1357
2026-08-20 00:37:05 +02:00
parent 37fd2dbbe1
commit 0cdd37fc15
101 changed files with 746591 additions and 278 deletions
+2 -1
View File
@@ -5,11 +5,12 @@
### Added
- Added `qwenTemplateReasoningEffort` to the `models.yml` `compat` schema, so the auto-enabled Qwen 3.8+ template effort dialect (`chat_template_kwargs.reasoning_effort`) can be switched off per provider/model for strict local servers that reject unknown `chat_template_kwargs`.
- Added `tokenizer` to custom model and `modelOverrides` configuration. It overrides the catalog-resolved local tokenizer family for a model when a proxy serves a known model id with a different tokenizer.
### Changed
- `/settings` rows can now carry a risk note: a warning glyph on the row plus a warning-colored line above the description. `External Thinking` (`externalThinking`, `--external-thinking`) is the first user — providers have flagged the request shape it produces as abuse, up to account-level enforcement, so both the settings entry and `--help` now say so.
- Token counting is now scoped to the model being billed rather than to a process-global tokenizer: session maintenance, stats, advisors, `/context`, snapcompact inline imaging, and `compress` each count through the owning agent's `Tokenizer` (`agent.tokenizer`). Message counting is `Tokenizer.countMessage`/`countMessages` (replacing the free `estimateTokens(message, tokenizer)` helper; the legacy shim keeps a compat `estimateTokens` export for legacy pi extensions). `estimateToolSchemaTokens`, `estimateSkillsTokens`, `computeNonMessageTokens`, and `computeNonMessageBreakdown` take an explicit tokenizer; `scripts/measure-prompt-tokens.ts` accepts an optional model id (argv) so its numbers match what that model is charged.
- Token counting is now scoped to the model being billed rather than to a process-global tokenizer: session maintenance, stats, advisors, `/context`, snapcompact inline imaging, and `compress` each count through the owning agent's `Tokenizer` (`agent.tokenizer`). Message counting is `Tokenizer.countMessage`/`countMessages` (replacing the free `estimateTokens(message, tokenizer)` helper; the legacy shim keeps a compat `estimateTokens` export for legacy pi extensions). `estimateToolSchemaTokens`, `estimateSkillsTokens`, `computeNonMessageTokens`, and `computeNonMessageBreakdown` take an explicit tokenizer; standalone prompt inspection intentionally keeps the default estimate because it has no resolved catalog model.
- The advisor runtime's `maintainContext` hook now receives the pending update as a message instead of a pre-computed token count — sizing it needs the advisor model's tokenizer, which the host owns.
### Fixed
@@ -14,10 +14,9 @@ function est(s: string): number {
await Settings.init({ inMemory: true, cwd: process.cwd() });
const settings = Settings.isolated({});
// Optional model id (argv[2]) scopes the counter to that model's tokenizer, so
// the numbers match what the agent will actually be charged. Without it the
// counts are the fast byte estimate the runtime uses for non-Claude models.
const tokenizer = new Tokenizer(process.argv[2]);
// This standalone inspection script has no resolved catalog Model; its counts
// therefore intentionally use the runtime's default estimate policy.
const tokenizer = new Tokenizer();
const session: ToolSession = {
cwd: process.cwd(),
@@ -75,13 +75,12 @@ export class CompressProtocol {
#verdict: string | undefined;
/**
* `modelId` scopes the token counter to the compressing model. Metrics are
* source-vs-draft ratios measured with one counter, so they stay coherent
* even when the model is unknown at construction time (the session that
* resolves it is built from this protocol).
* Metrics measure source-vs-draft ratios with the default estimate. The
* compress session resolves its model after this ledger is constructed, so
* no catalog model is available here.
*/
constructor(source: string, modelId?: string | null) {
this.#tokenizer = new Tokenizer(modelId);
constructor(source: string) {
this.#tokenizer = new Tokenizer();
this.#sourceWords = words(source);
this.#sourceTokens = this.#tokenizer.countTokens(source);
}
@@ -86,6 +86,7 @@ export function buildCustomModelOverlay(
thinking: modelDef.thinking,
input: modelDef.input,
imageInputDecoder: modelDef.imageInputDecoder,
tokenizer: modelDef.tokenizer,
supportsTools: modelDef.supportsTools,
cost: modelDef.cost,
contextWindow: modelDef.contextWindow,
@@ -136,6 +137,7 @@ export function finalizeCustomModel(model: CustomModelOverlay, options: CustomMo
headers: resolvedModel.headers,
omitMaxOutputTokens: resolvedModel.omitMaxOutputTokens ?? reference?.omitMaxOutputTokens,
compat: mergeCompat(reference?.compatConfig, resolvedModel.compat),
tokenizer: resolvedModel.tokenizer,
contextPromotionTarget: resolvedModel.contextPromotionTarget,
compactionModel: resolvedModel.compactionModel,
remoteCompaction: resolvedModel.remoteCompaction,
@@ -184,6 +184,7 @@ export interface ModelPatch {
thinking?: ThinkingConfig;
input?: ("text" | "image")[];
imageInputDecoder?: Model<Api>["imageInputDecoder"];
tokenizer?: Model<Api>["tokenizer"];
supportsTools?: boolean;
cost?: Partial<Model<Api>["cost"]>;
contextWindow?: number;
@@ -211,6 +212,7 @@ export function applyModelPatch(base: Model<Api>, patch: ModelPatch, transport:
if (patch.reasoning !== undefined) result.reasoning = patch.reasoning;
if (patch.thinking !== undefined) result.thinking = patch.thinking;
if (patch.input !== undefined) result.input = patch.input;
if (patch.tokenizer !== undefined) result.tokenizer = patch.tokenizer;
if (patch.imageInputDecoder !== undefined) result.imageInputDecoder = patch.imageInputDecoder;
if (patch.supportsTools !== undefined) result.supportsTools = patch.supportsTools;
if (patch.contextWindow !== undefined) result.contextWindow = patch.contextWindow;
@@ -132,6 +132,10 @@ export const getModelsConfigSchemaBundle = once(() => {
};
});
const ModelTokenizerSchema = type(
'"claude-v3" | "claude-v47" | "claude-v5" | "claude-v5-sonnet" | "qwen3" | "deepseek-v3" | "kimi-k2" | "glm5"',
);
const RemoteCompactionSchema = type({
"enabled?": "boolean",
"api?": ApiSchema,
@@ -169,6 +173,7 @@ export const getModelsConfigSchemaBundle = once(() => {
"thinking?": ModelThinkingSchema,
"input?": '("text" | "image")[]',
"imageInputDecoder?": '"stb"',
"tokenizer?": ModelTokenizerSchema,
"supportsTools?": "boolean",
"cost?": {
input: "number",
@@ -219,6 +224,7 @@ export const getModelsConfigSchemaBundle = once(() => {
"thinking?": ModelThinkingSchema,
"input?": '("text" | "image")[]',
"imageInputDecoder?": '"stb"',
"tokenizer?": ModelTokenizerSchema,
"supportsTools?": "boolean",
"cost?": {
"input?": "number",
@@ -282,7 +282,7 @@ export function estimateInlineSavings(input: {
}
const shape = snapcompact.resolveShape(model, options.shape);
const tokenizer = new Tokenizer(model.id);
const tokenizer = new Tokenizer(model);
let existingImages = 0;
for (const message of input.messages) {
if (!Array.isArray(message.content)) continue;
@@ -421,7 +421,7 @@ export class SnapcompactInlineTransformer {
if (!model.input.includes("image")) return context;
const shape = snapcompact.resolveShape(model, this.options.shape);
const tokenizer = new Tokenizer(model.id);
const tokenizer = new Tokenizer(model);
const budget = snapcompact.providerImageBudget(model.provider) - countContextImages(context);
if (budget <= 0) return context;