fix umans max token cap

This commit is contained in:
oldschoola
2026-06-15 23:12:27 -07:00
parent e24f789de4
commit 6382896115
6 changed files with 166 additions and 54 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed Kimi output caps for Umans AI Coding Plan and Venice so discovery metadata cannot use context-sized token ceilings as request caps.
## [16.0.1] - 2026-06-15
### Added
+46 -34
View File
@@ -30,7 +30,9 @@ import {
ANTHROPIC_CURATED_FALLBACK_MODELS,
buildXaiOAuthStaticSeed,
clampFireworksKimiMaxTokens,
clampKimiK27CodeMaxTokens,
isFireworksKimiK2ModelId,
isKimiK27CodeModelId,
MODELS_DEV_PROVIDER_DESCRIPTORS,
mapModelsDevToModels,
stripFireworksDeepSeekThinkingToggle,
@@ -245,21 +247,22 @@ function applyCodexPricingFallback(models: readonly ModelSpec[]): ModelSpec[] {
}
/**
* Fireworks-backed Kimi K2.x deployments report `max_completion_tokens: 65536`
* over `/v1/models`, but Kimi's documented output budget on Fireworks is
* lower (#1849). Cap them here so the post-processing pass — which also folds
* in the `prevModelsJson` static fallback used by `firepass` — never lets a
* stale or inflated upstream value through. The resolver applies the same
* cap when discovery runs at runtime; this is the bundle-time safety net.
* Provider discovery sometimes reports context-sized Kimi output ceilings. Keep
* the bundled catalog at the documented/provider-safe caps so request builders
* that always send `max_tokens` do not over-allocate.
*/
function applyFireworksKimiMaxTokensCap(models: readonly ModelSpec[]): ModelSpec[] {
function applyKimiMaxTokensCap(models: readonly ModelSpec[]): ModelSpec[] {
const FIREWORKS_KIMI_PROVIDERS = new Set(["fireworks", "firepass"]);
return models.map(model => {
if (!FIREWORKS_KIMI_PROVIDERS.has(model.provider)) return model;
if (!isFireworksKimiK2ModelId(model.id)) return model;
const capped = clampFireworksKimiMaxTokens(model.id, model.maxTokens);
if (capped === model.maxTokens) return model;
return { ...model, maxTokens: capped };
if (FIREWORKS_KIMI_PROVIDERS.has(model.provider) && isFireworksKimiK2ModelId(model.id)) {
const capped = clampFireworksKimiMaxTokens(model.id, model.maxTokens);
return capped === model.maxTokens ? model : { ...model, maxTokens: capped };
}
if (model.provider === "venice" && isKimiK27CodeModelId(model.id)) {
const capped = clampKimiK27CodeMaxTokens(model.id, model.maxTokens);
return capped === model.maxTokens ? model : { ...model, maxTokens: capped };
}
return model;
});
}
@@ -418,20 +421,31 @@ async function fetchCodexDiscoveryModels(): Promise<ModelSpec<"openai-codex-resp
}
async function generateModels() {
// Fetch models from dynamic sources
// Fetch models from dynamic sources.
const modelsDevModels = await loadModelsDevData();
const catalogProviderModels = (
await Promise.all(
PROVIDER_DESCRIPTORS.filter(
descriptor => isCatalogDescriptor(descriptor) && !DISCOVERY_ONLY_PROVIDERS.has(descriptor.providerId),
).map(descriptor => fetchProviderModelsFromCatalog(descriptor as CatalogProviderDescriptor)),
)
).flat();
const catalogProviderDescriptors = PROVIDER_DESCRIPTORS.filter(
(descriptor): descriptor is CatalogProviderDescriptor =>
isCatalogDescriptor(descriptor) && !DISCOVERY_ONLY_PROVIDERS.has(descriptor.providerId),
);
const catalogProviderModelBatches = await Promise.all(
catalogProviderDescriptors.map(async descriptor => ({
descriptor,
models: await fetchProviderModelsFromCatalog(descriptor),
})),
);
const authoritativeCatalogProviders = new Set(
catalogProviderModelBatches
.filter(batch => batch.descriptor.dynamicModelsAuthoritative === true && batch.models.length > 0)
.map(batch => batch.descriptor.providerId),
);
const catalogProviderModels = catalogProviderModelBatches.flatMap(batch => batch.models);
const bundledModelsDevModels = modelsDevModels.filter(model => !authoritativeCatalogProviders.has(model.provider));
// getGitLabDuoModels returns built models; project back to spec stage for the bundle.
const gitLabDuoModels = getGitLabDuoModels().map(model => toModelSpec(model));
// Combine models (models.dev has priority)
// Combine models. models.dev has priority unless a provider's successful endpoint
// discovery is authoritative; those endpoint snapshots replace models.dev rows.
let allModels = applyGlobalModelsDevFallback(
[...modelsDevModels, ...catalogProviderModels, ...gitLabDuoModels],
[...bundledModelsDevModels, ...catalogProviderModels, ...gitLabDuoModels],
modelsDevModels,
);
@@ -471,19 +485,16 @@ async function generateModels() {
}
}
const modelsDevAuthoritativeProviders = new Set<string>();
const modelsDevSnapshotExcludedProviders = new Set<string>();
for (const model of modelsDevModels) {
if (model.provider === "google-vertex") {
modelsDevAuthoritativeProviders.add(model.provider);
modelsDevSnapshotExcludedProviders.add(model.provider);
}
}
if (catalogProviderModels.some(model => model.provider === "aimlapi")) {
modelsDevAuthoritativeProviders.add("aimlapi");
}
// Merge previous models.json entries as fallback for provider/model pairs not
// fetched dynamically. Providers that models.dev covers authoritatively keep
// the upstream list exactly, so retired entries from the previous snapshot do
// not reappear during regeneration.
// fetched dynamically. Providers covered by authoritative endpoint discovery
// or authoritative models.dev sources keep that upstream list exactly, so
// retired entries from the previous snapshot do not reappear during regeneration.
// Discovery-only providers (local inference servers) — never bundle static models.
const fetchedKeys = new Set(allModels.map(model => `${model.provider}/${model.id}`));
@@ -495,7 +506,8 @@ async function generateModels() {
if (
!fetchedKeys.has(`${model.provider}/${model.id}`) &&
!DISCOVERY_ONLY_PROVIDERS.has(model.provider) &&
!modelsDevAuthoritativeProviders.has(model.provider)
!authoritativeCatalogProviders.has(model.provider) &&
!modelsDevSnapshotExcludedProviders.has(model.provider)
) {
allModels.push(model);
}
@@ -505,7 +517,7 @@ async function generateModels() {
allModels = applyGlobalModelsDevFallback(allModels, modelsDevModels);
allModels = applyPremiumMultiplierOverrides(allModels);
allModels = applyCodexPricingFallback(allModels);
allModels = applyFireworksKimiMaxTokensCap(allModels);
allModels = applyKimiMaxTokensCap(allModels);
allModels = applyFireworksDeepSeekReasoningShape(allModels);
allModels = dropFireworksWireIds(allModels);
allModels = dropUnusableZaiContextTierIds(allModels);
@@ -534,8 +546,8 @@ async function generateModels() {
if (!providers[model.provider]) {
providers[model.provider] = {};
}
// Use model ID as key to automatically deduplicate
// Only add if not already present (models.dev takes priority over endpoint discovery)
// Use model ID as key to deduplicate the ordered sources assembled above.
// Earlier sources win.
if (!providers[model.provider][model.id]) {
providers[model.provider][model.id] = model;
}
+38 -16
View File
@@ -60012,8 +60012,8 @@
"text"
],
"cost": {
"input": 0.09,
"output": 0.18,
"input": 0.098,
"output": 0.196,
"cacheRead": 0.02,
"cacheWrite": 0
},
@@ -65288,7 +65288,7 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144,
"maxTokens": 81920,
"thinking": {
"mode": "effort",
"efforts": [
@@ -65311,12 +65311,12 @@
"image"
],
"cost": {
"input": 0.39,
"output": 2.34,
"input": 0.385,
"output": 2.4499999999999997,
"cacheRead": 0.195,
"cacheWrite": 0
},
"contextWindow": 262144,
"contextWindow": 256000,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
@@ -68906,7 +68906,7 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144,
"maxTokens": 32768,
"thinking": {
"mode": "budget",
"efforts": [
@@ -68936,7 +68936,7 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144,
"maxTokens": 32768,
"thinking": {
"mode": "budget",
"efforts": [
@@ -68950,7 +68950,7 @@
},
"umans-glm-5.1": {
"id": "umans-glm-5.1",
"name": "GLM 5.1",
"name": "Umans GLM 5.1",
"api": "anthropic-messages",
"provider": "umans",
"baseUrl": "https://api.code.umans.ai",
@@ -68965,7 +68965,7 @@
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 204800,
"contextWindow": 202752,
"maxTokens": 131072,
"thinking": {
"mode": "budget",
@@ -68980,7 +68980,7 @@
},
"umans-kimi-k2.6": {
"id": "umans-kimi-k2.6",
"name": "Kimi K2.6",
"name": "Umans Kimi K2.6",
"api": "anthropic-messages",
"provider": "umans",
"baseUrl": "https://api.code.umans.ai",
@@ -68996,7 +68996,7 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144,
"maxTokens": 32768,
"thinking": {
"mode": "budget",
"efforts": [
@@ -69010,7 +69010,7 @@
},
"umans-kimi-k2.7": {
"id": "umans-kimi-k2.7",
"name": "Kimi K2.7 Code",
"name": "Umans Kimi K2.7 Code",
"api": "anthropic-messages",
"provider": "umans",
"baseUrl": "https://api.code.umans.ai",
@@ -69026,7 +69026,7 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144,
"maxTokens": 32768,
"thinking": {
"mode": "budget",
"efforts": [
@@ -69041,7 +69041,7 @@
},
"umans-qwen3.6-35b-a3b": {
"id": "umans-qwen3.6-35b-a3b",
"name": "Qwen3.6 35B A3B",
"name": "Umans Qwen3.6 35B A3B",
"api": "anthropic-messages",
"provider": "umans",
"baseUrl": "https://api.code.umans.ai",
@@ -69057,7 +69057,7 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144,
"maxTokens": 32768,
"thinking": {
"mode": "budget",
"efforts": [
@@ -70539,6 +70539,28 @@
]
}
},
"kimi-k2-7-code": {
"id": "kimi-k2-7-code",
"name": "kimi-k2-7-code",
"api": "openai-completions",
"provider": "venice",
"baseUrl": "https://api.venice.ai/api/v1",
"reasoning": false,
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 256000,
"maxTokens": 32768,
"compat": {
"supportsUsageInStreaming": false
}
},
"kimi-k2-thinking": {
"id": "kimi-k2-thinking",
"name": "Kimi K2 Thinking",
@@ -664,7 +664,10 @@ function mapUmansModelInfo(
...(supportsTools === false ? { supportsTools: false } : {}),
cost: reference?.cost ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: toPositiveNumber(capabilities.context_window, reference?.contextWindow ?? null),
maxTokens: toPositiveNumber(capabilities.max_completion_tokens, reference?.maxTokens ?? null),
maxTokens: toPositiveNumber(
capabilities.recommended_max_tokens,
toPositiveNumber(capabilities.max_completion_tokens, reference?.maxTokens ?? null),
),
};
}
@@ -1232,6 +1235,23 @@ export function clampFireworksKimiMaxTokens(modelId: string, candidate: number |
return isFireworksKimiK2ModelId(modelId) ? Math.min(candidate, FIREWORKS_KIMI_MAX_TOKENS) : candidate;
}
/**
* Kimi K2.7 Code's documented recommended output budget. Some provider
* discovery rows report the context-sized `max_completion_tokens` instead.
*/
export const KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS = 32_768;
export function isKimiK27CodeModelId(modelId: string): boolean {
return /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code$/i.test(modelId);
}
export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number): number;
export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number | null): number | null;
export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number | null): number | null {
if (candidate === null) return null;
return isKimiK27CodeModelId(modelId) ? Math.min(candidate, KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS) : candidate;
}
/**
* Fireworks DeepSeek V4 accepts effort via `reasoning_effort` but rejects the
* DeepSeek-native binary `thinking` toggle when both are present.
@@ -2276,6 +2296,7 @@ export function veniceModelManagerOptions(
const model = mapWithBundledReference(entry, defaults, reference);
return {
...model,
maxTokens: clampKimiK27CodeMaxTokens(defaults.id, model.maxTokens),
compat: { ...model.compat, supportsUsageInStreaming: false },
};
},
@@ -3544,7 +3565,12 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_SPECIALIZED: readonly ModelsDevProviderDes
// --- Synthetic ---
openAiCompletionsDescriptor("synthetic", "synthetic", "https://api.synthetic.new/openai/v1"),
// --- Venice AI ---
openAiCompletionsDescriptor("venice", "venice", "https://api.venice.ai/api/v1"),
openAiCompletionsDescriptor("venice", "venice", "https://api.venice.ai/api/v1", {
transformModel: model => {
const maxTokens = clampKimiK27CodeMaxTokens(model.id, model.maxTokens);
return maxTokens === model.maxTokens ? model : { ...model, maxTokens };
},
}),
// --- Ollama Cloud ---
simpleModelsDevDescriptor("ollama-cloud", "ollama-cloud", "ollama-chat", "https://ollama.com"),
// --- Xiaomi Token Plan ---
+6 -2
View File
@@ -33,6 +33,7 @@ describe("umans provider catalog", () => {
capabilities: {
context_window: 262_144,
max_completion_tokens: 262_144,
recommended_max_tokens: 32_768,
supports_vision: true,
supports_tools: true,
reasoning: { supported: true, can_disable: true, default_level: "medium" },
@@ -43,6 +44,7 @@ describe("umans provider catalog", () => {
capabilities: {
context_window: 262_144,
max_completion_tokens: 262_144,
recommended_max_tokens: 32_768,
supports_vision: true,
supports_tools: true,
reasoning: { supported: true, can_disable: false, default_level: "medium" },
@@ -71,13 +73,14 @@ describe("umans provider catalog", () => {
reasoning: true,
input: ["text", "image"],
contextWindow: 262_144,
maxTokens: 262_144,
maxTokens: 32_768,
thinking: { defaultLevel: "medium" },
});
const mandatoryReasoningModel = models?.find(item => item.id === "umans-kimi-k2.7");
expect(mandatoryReasoningModel).toMatchObject({
id: "umans-kimi-k2.7",
reasoning: true,
maxTokens: 32_768,
thinking: { defaultLevel: "medium", requiresEffort: true },
});
});
@@ -137,7 +140,7 @@ describe("umans provider catalog", () => {
reasoning: true,
input: ["text", "image"],
contextWindow: 262_144,
maxTokens: 262_144,
maxTokens: 32_768,
});
});
@@ -146,6 +149,7 @@ describe("umans provider catalog", () => {
const model = providers.umans?.["umans-kimi-k2.7"];
expect(model).toBeDefined();
expect(model.maxTokens).toBe(32_768);
expect(model.thinking).toMatchObject({
requiresEffort: true,
});
@@ -0,0 +1,44 @@
import { describe, expect, it } from "bun:test";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import {
KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS,
veniceModelManagerOptions,
} from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
describe("Venice provider catalog", () => {
it("bundles Kimi K2.7 Code with its recommended output cap", () => {
const model = getBundledModel("venice", "kimi-k2-7-code");
expect(model).toBeDefined();
expect(model.maxTokens).toBe(KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS);
});
it("caps Kimi K2.7 Code during runtime discovery", async () => {
const requestedUrls: string[] = [];
const fetchImpl: FetchImpl = async input => {
requestedUrls.push(input instanceof Request ? input.url : String(input));
return new Response(
JSON.stringify({
data: [
{
id: "kimi-k2-7-code",
name: "kimi-k2-7-code",
context_length: 256_000,
max_completion_tokens: 262_144,
},
],
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
);
};
const options = veniceModelManagerOptions({ apiKey: "venice-test-key", fetch: fetchImpl });
const models = await options.fetchDynamicModels?.();
const model = models?.find(candidate => candidate.id === "kimi-k2-7-code");
expect(requestedUrls).toEqual(["https://api.venice.ai/api/v1/models"]);
expect(model).toBeDefined();
expect(model?.maxTokens).toBe(KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS);
});
});