feat(pi-natives/tools): implemented utok tokenizer for multiple models
- Replaced the `ctok` implementation with the `utok` universal tokenizer supporting multiple model families and UTF text encodings. - Added tokenizer support and embedding data for Qwen3, DeepSeek V3, Kimi K2, and GLM-5 model variants. - Added fixture generation scripts, vocabulary packers, and golden test suites for validating tokenization parity. - Updated dependency requirements and Bazel workspace definitions for new crates and tools.
This commit is contained in:
@@ -5,11 +5,12 @@
|
||||
### Added
|
||||
|
||||
- Added `qwenTemplateReasoningEffort` to the `models.yml` `compat` schema, so the auto-enabled Qwen 3.8+ template effort dialect (`chat_template_kwargs.reasoning_effort`) can be switched off per provider/model for strict local servers that reject unknown `chat_template_kwargs`.
|
||||
- Added `tokenizer` to custom model and `modelOverrides` configuration. It overrides the catalog-resolved local tokenizer family for a model when a proxy serves a known model id with a different tokenizer.
|
||||
|
||||
### Changed
|
||||
|
||||
- `/settings` rows can now carry a risk note: a warning glyph on the row plus a warning-colored line above the description. `External Thinking` (`externalThinking`, `--external-thinking`) is the first user — providers have flagged the request shape it produces as abuse, up to account-level enforcement, so both the settings entry and `--help` now say so.
|
||||
- Token counting is now scoped to the model being billed rather than to a process-global tokenizer: session maintenance, stats, advisors, `/context`, snapcompact inline imaging, and `compress` each count through the owning agent's `Tokenizer` (`agent.tokenizer`). Message counting is `Tokenizer.countMessage`/`countMessages` (replacing the free `estimateTokens(message, tokenizer)` helper; the legacy shim keeps a compat `estimateTokens` export for legacy pi extensions). `estimateToolSchemaTokens`, `estimateSkillsTokens`, `computeNonMessageTokens`, and `computeNonMessageBreakdown` take an explicit tokenizer; `scripts/measure-prompt-tokens.ts` accepts an optional model id (argv) so its numbers match what that model is charged.
|
||||
- Token counting is now scoped to the model being billed rather than to a process-global tokenizer: session maintenance, stats, advisors, `/context`, snapcompact inline imaging, and `compress` each count through the owning agent's `Tokenizer` (`agent.tokenizer`). Message counting is `Tokenizer.countMessage`/`countMessages` (replacing the free `estimateTokens(message, tokenizer)` helper; the legacy shim keeps a compat `estimateTokens` export for legacy pi extensions). `estimateToolSchemaTokens`, `estimateSkillsTokens`, `computeNonMessageTokens`, and `computeNonMessageBreakdown` take an explicit tokenizer; standalone prompt inspection intentionally keeps the default estimate because it has no resolved catalog model.
|
||||
- The advisor runtime's `maintainContext` hook now receives the pending update as a message instead of a pre-computed token count — sizing it needs the advisor model's tokenizer, which the host owns.
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -14,10 +14,9 @@ function est(s: string): number {
|
||||
await Settings.init({ inMemory: true, cwd: process.cwd() });
|
||||
const settings = Settings.isolated({});
|
||||
|
||||
// Optional model id (argv[2]) scopes the counter to that model's tokenizer, so
|
||||
// the numbers match what the agent will actually be charged. Without it the
|
||||
// counts are the fast byte estimate the runtime uses for non-Claude models.
|
||||
const tokenizer = new Tokenizer(process.argv[2]);
|
||||
// This standalone inspection script has no resolved catalog Model; its counts
|
||||
// therefore intentionally use the runtime's default estimate policy.
|
||||
const tokenizer = new Tokenizer();
|
||||
|
||||
const session: ToolSession = {
|
||||
cwd: process.cwd(),
|
||||
|
||||
@@ -75,13 +75,12 @@ export class CompressProtocol {
|
||||
#verdict: string | undefined;
|
||||
|
||||
/**
|
||||
* `modelId` scopes the token counter to the compressing model. Metrics are
|
||||
* source-vs-draft ratios measured with one counter, so they stay coherent
|
||||
* even when the model is unknown at construction time (the session that
|
||||
* resolves it is built from this protocol).
|
||||
* Metrics measure source-vs-draft ratios with the default estimate. The
|
||||
* compress session resolves its model after this ledger is constructed, so
|
||||
* no catalog model is available here.
|
||||
*/
|
||||
constructor(source: string, modelId?: string | null) {
|
||||
this.#tokenizer = new Tokenizer(modelId);
|
||||
constructor(source: string) {
|
||||
this.#tokenizer = new Tokenizer();
|
||||
this.#sourceWords = words(source);
|
||||
this.#sourceTokens = this.#tokenizer.countTokens(source);
|
||||
}
|
||||
|
||||
@@ -86,6 +86,7 @@ export function buildCustomModelOverlay(
|
||||
thinking: modelDef.thinking,
|
||||
input: modelDef.input,
|
||||
imageInputDecoder: modelDef.imageInputDecoder,
|
||||
tokenizer: modelDef.tokenizer,
|
||||
supportsTools: modelDef.supportsTools,
|
||||
cost: modelDef.cost,
|
||||
contextWindow: modelDef.contextWindow,
|
||||
@@ -136,6 +137,7 @@ export function finalizeCustomModel(model: CustomModelOverlay, options: CustomMo
|
||||
headers: resolvedModel.headers,
|
||||
omitMaxOutputTokens: resolvedModel.omitMaxOutputTokens ?? reference?.omitMaxOutputTokens,
|
||||
compat: mergeCompat(reference?.compatConfig, resolvedModel.compat),
|
||||
tokenizer: resolvedModel.tokenizer,
|
||||
contextPromotionTarget: resolvedModel.contextPromotionTarget,
|
||||
compactionModel: resolvedModel.compactionModel,
|
||||
remoteCompaction: resolvedModel.remoteCompaction,
|
||||
|
||||
@@ -184,6 +184,7 @@ export interface ModelPatch {
|
||||
thinking?: ThinkingConfig;
|
||||
input?: ("text" | "image")[];
|
||||
imageInputDecoder?: Model<Api>["imageInputDecoder"];
|
||||
tokenizer?: Model<Api>["tokenizer"];
|
||||
supportsTools?: boolean;
|
||||
cost?: Partial<Model<Api>["cost"]>;
|
||||
contextWindow?: number;
|
||||
@@ -211,6 +212,7 @@ export function applyModelPatch(base: Model<Api>, patch: ModelPatch, transport:
|
||||
if (patch.reasoning !== undefined) result.reasoning = patch.reasoning;
|
||||
if (patch.thinking !== undefined) result.thinking = patch.thinking;
|
||||
if (patch.input !== undefined) result.input = patch.input;
|
||||
if (patch.tokenizer !== undefined) result.tokenizer = patch.tokenizer;
|
||||
if (patch.imageInputDecoder !== undefined) result.imageInputDecoder = patch.imageInputDecoder;
|
||||
if (patch.supportsTools !== undefined) result.supportsTools = patch.supportsTools;
|
||||
if (patch.contextWindow !== undefined) result.contextWindow = patch.contextWindow;
|
||||
|
||||
@@ -132,6 +132,10 @@ export const getModelsConfigSchemaBundle = once(() => {
|
||||
};
|
||||
});
|
||||
|
||||
const ModelTokenizerSchema = type(
|
||||
'"claude-v3" | "claude-v47" | "claude-v5" | "claude-v5-sonnet" | "qwen3" | "deepseek-v3" | "kimi-k2" | "glm5"',
|
||||
);
|
||||
|
||||
const RemoteCompactionSchema = type({
|
||||
"enabled?": "boolean",
|
||||
"api?": ApiSchema,
|
||||
@@ -169,6 +173,7 @@ export const getModelsConfigSchemaBundle = once(() => {
|
||||
"thinking?": ModelThinkingSchema,
|
||||
"input?": '("text" | "image")[]',
|
||||
"imageInputDecoder?": '"stb"',
|
||||
"tokenizer?": ModelTokenizerSchema,
|
||||
"supportsTools?": "boolean",
|
||||
"cost?": {
|
||||
input: "number",
|
||||
@@ -219,6 +224,7 @@ export const getModelsConfigSchemaBundle = once(() => {
|
||||
"thinking?": ModelThinkingSchema,
|
||||
"input?": '("text" | "image")[]',
|
||||
"imageInputDecoder?": '"stb"',
|
||||
"tokenizer?": ModelTokenizerSchema,
|
||||
"supportsTools?": "boolean",
|
||||
"cost?": {
|
||||
"input?": "number",
|
||||
|
||||
@@ -282,7 +282,7 @@ export function estimateInlineSavings(input: {
|
||||
}
|
||||
|
||||
const shape = snapcompact.resolveShape(model, options.shape);
|
||||
const tokenizer = new Tokenizer(model.id);
|
||||
const tokenizer = new Tokenizer(model);
|
||||
let existingImages = 0;
|
||||
for (const message of input.messages) {
|
||||
if (!Array.isArray(message.content)) continue;
|
||||
@@ -421,7 +421,7 @@ export class SnapcompactInlineTransformer {
|
||||
if (!model.input.includes("image")) return context;
|
||||
|
||||
const shape = snapcompact.resolveShape(model, this.options.shape);
|
||||
const tokenizer = new Tokenizer(model.id);
|
||||
const tokenizer = new Tokenizer(model);
|
||||
const budget = snapcompact.providerImageBudget(model.provider) - countContextImages(context);
|
||||
if (budget <= 0) return context;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user