fix umans max token cap
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Kimi output caps for Umans AI Coding Plan and Venice so discovery metadata cannot use context-sized token ceilings as request caps.
|
||||
|
||||
## [16.0.1] - 2026-06-15
|
||||
|
||||
### Added
|
||||
|
||||
@@ -30,7 +30,9 @@ import {
|
||||
ANTHROPIC_CURATED_FALLBACK_MODELS,
|
||||
buildXaiOAuthStaticSeed,
|
||||
clampFireworksKimiMaxTokens,
|
||||
clampKimiK27CodeMaxTokens,
|
||||
isFireworksKimiK2ModelId,
|
||||
isKimiK27CodeModelId,
|
||||
MODELS_DEV_PROVIDER_DESCRIPTORS,
|
||||
mapModelsDevToModels,
|
||||
stripFireworksDeepSeekThinkingToggle,
|
||||
@@ -245,21 +247,22 @@ function applyCodexPricingFallback(models: readonly ModelSpec[]): ModelSpec[] {
|
||||
}
|
||||
|
||||
/**
|
||||
* Fireworks-backed Kimi K2.x deployments report `max_completion_tokens: 65536`
|
||||
* over `/v1/models`, but Kimi's documented output budget on Fireworks is
|
||||
* lower (#1849). Cap them here so the post-processing pass — which also folds
|
||||
* in the `prevModelsJson` static fallback used by `firepass` — never lets a
|
||||
* stale or inflated upstream value through. The resolver applies the same
|
||||
* cap when discovery runs at runtime; this is the bundle-time safety net.
|
||||
* Provider discovery sometimes reports context-sized Kimi output ceilings. Keep
|
||||
* the bundled catalog at the documented/provider-safe caps so request builders
|
||||
* that always send `max_tokens` do not over-allocate.
|
||||
*/
|
||||
function applyFireworksKimiMaxTokensCap(models: readonly ModelSpec[]): ModelSpec[] {
|
||||
function applyKimiMaxTokensCap(models: readonly ModelSpec[]): ModelSpec[] {
|
||||
const FIREWORKS_KIMI_PROVIDERS = new Set(["fireworks", "firepass"]);
|
||||
return models.map(model => {
|
||||
if (!FIREWORKS_KIMI_PROVIDERS.has(model.provider)) return model;
|
||||
if (!isFireworksKimiK2ModelId(model.id)) return model;
|
||||
const capped = clampFireworksKimiMaxTokens(model.id, model.maxTokens);
|
||||
if (capped === model.maxTokens) return model;
|
||||
return { ...model, maxTokens: capped };
|
||||
if (FIREWORKS_KIMI_PROVIDERS.has(model.provider) && isFireworksKimiK2ModelId(model.id)) {
|
||||
const capped = clampFireworksKimiMaxTokens(model.id, model.maxTokens);
|
||||
return capped === model.maxTokens ? model : { ...model, maxTokens: capped };
|
||||
}
|
||||
if (model.provider === "venice" && isKimiK27CodeModelId(model.id)) {
|
||||
const capped = clampKimiK27CodeMaxTokens(model.id, model.maxTokens);
|
||||
return capped === model.maxTokens ? model : { ...model, maxTokens: capped };
|
||||
}
|
||||
return model;
|
||||
});
|
||||
}
|
||||
|
||||
@@ -418,20 +421,31 @@ async function fetchCodexDiscoveryModels(): Promise<ModelSpec<"openai-codex-resp
|
||||
}
|
||||
|
||||
async function generateModels() {
|
||||
// Fetch models from dynamic sources
|
||||
// Fetch models from dynamic sources.
|
||||
const modelsDevModels = await loadModelsDevData();
|
||||
const catalogProviderModels = (
|
||||
await Promise.all(
|
||||
PROVIDER_DESCRIPTORS.filter(
|
||||
descriptor => isCatalogDescriptor(descriptor) && !DISCOVERY_ONLY_PROVIDERS.has(descriptor.providerId),
|
||||
).map(descriptor => fetchProviderModelsFromCatalog(descriptor as CatalogProviderDescriptor)),
|
||||
)
|
||||
).flat();
|
||||
const catalogProviderDescriptors = PROVIDER_DESCRIPTORS.filter(
|
||||
(descriptor): descriptor is CatalogProviderDescriptor =>
|
||||
isCatalogDescriptor(descriptor) && !DISCOVERY_ONLY_PROVIDERS.has(descriptor.providerId),
|
||||
);
|
||||
const catalogProviderModelBatches = await Promise.all(
|
||||
catalogProviderDescriptors.map(async descriptor => ({
|
||||
descriptor,
|
||||
models: await fetchProviderModelsFromCatalog(descriptor),
|
||||
})),
|
||||
);
|
||||
const authoritativeCatalogProviders = new Set(
|
||||
catalogProviderModelBatches
|
||||
.filter(batch => batch.descriptor.dynamicModelsAuthoritative === true && batch.models.length > 0)
|
||||
.map(batch => batch.descriptor.providerId),
|
||||
);
|
||||
const catalogProviderModels = catalogProviderModelBatches.flatMap(batch => batch.models);
|
||||
const bundledModelsDevModels = modelsDevModels.filter(model => !authoritativeCatalogProviders.has(model.provider));
|
||||
// getGitLabDuoModels returns built models; project back to spec stage for the bundle.
|
||||
const gitLabDuoModels = getGitLabDuoModels().map(model => toModelSpec(model));
|
||||
// Combine models (models.dev has priority)
|
||||
// Combine models. models.dev has priority unless a provider's successful endpoint
|
||||
// discovery is authoritative; those endpoint snapshots replace models.dev rows.
|
||||
let allModels = applyGlobalModelsDevFallback(
|
||||
[...modelsDevModels, ...catalogProviderModels, ...gitLabDuoModels],
|
||||
[...bundledModelsDevModels, ...catalogProviderModels, ...gitLabDuoModels],
|
||||
modelsDevModels,
|
||||
);
|
||||
|
||||
@@ -471,19 +485,16 @@ async function generateModels() {
|
||||
}
|
||||
}
|
||||
|
||||
const modelsDevAuthoritativeProviders = new Set<string>();
|
||||
const modelsDevSnapshotExcludedProviders = new Set<string>();
|
||||
for (const model of modelsDevModels) {
|
||||
if (model.provider === "google-vertex") {
|
||||
modelsDevAuthoritativeProviders.add(model.provider);
|
||||
modelsDevSnapshotExcludedProviders.add(model.provider);
|
||||
}
|
||||
}
|
||||
if (catalogProviderModels.some(model => model.provider === "aimlapi")) {
|
||||
modelsDevAuthoritativeProviders.add("aimlapi");
|
||||
}
|
||||
// Merge previous models.json entries as fallback for provider/model pairs not
|
||||
// fetched dynamically. Providers that models.dev covers authoritatively keep
|
||||
// the upstream list exactly, so retired entries from the previous snapshot do
|
||||
// not reappear during regeneration.
|
||||
// fetched dynamically. Providers covered by authoritative endpoint discovery
|
||||
// or authoritative models.dev sources keep that upstream list exactly, so
|
||||
// retired entries from the previous snapshot do not reappear during regeneration.
|
||||
// Discovery-only providers (local inference servers) — never bundle static models.
|
||||
const fetchedKeys = new Set(allModels.map(model => `${model.provider}/${model.id}`));
|
||||
|
||||
@@ -495,7 +506,8 @@ async function generateModels() {
|
||||
if (
|
||||
!fetchedKeys.has(`${model.provider}/${model.id}`) &&
|
||||
!DISCOVERY_ONLY_PROVIDERS.has(model.provider) &&
|
||||
!modelsDevAuthoritativeProviders.has(model.provider)
|
||||
!authoritativeCatalogProviders.has(model.provider) &&
|
||||
!modelsDevSnapshotExcludedProviders.has(model.provider)
|
||||
) {
|
||||
allModels.push(model);
|
||||
}
|
||||
@@ -505,7 +517,7 @@ async function generateModels() {
|
||||
allModels = applyGlobalModelsDevFallback(allModels, modelsDevModels);
|
||||
allModels = applyPremiumMultiplierOverrides(allModels);
|
||||
allModels = applyCodexPricingFallback(allModels);
|
||||
allModels = applyFireworksKimiMaxTokensCap(allModels);
|
||||
allModels = applyKimiMaxTokensCap(allModels);
|
||||
allModels = applyFireworksDeepSeekReasoningShape(allModels);
|
||||
allModels = dropFireworksWireIds(allModels);
|
||||
allModels = dropUnusableZaiContextTierIds(allModels);
|
||||
@@ -534,8 +546,8 @@ async function generateModels() {
|
||||
if (!providers[model.provider]) {
|
||||
providers[model.provider] = {};
|
||||
}
|
||||
// Use model ID as key to automatically deduplicate
|
||||
// Only add if not already present (models.dev takes priority over endpoint discovery)
|
||||
// Use model ID as key to deduplicate the ordered sources assembled above.
|
||||
// Earlier sources win.
|
||||
if (!providers[model.provider][model.id]) {
|
||||
providers[model.provider][model.id] = model;
|
||||
}
|
||||
|
||||
@@ -60012,8 +60012,8 @@
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.09,
|
||||
"output": 0.18,
|
||||
"input": 0.098,
|
||||
"output": 0.196,
|
||||
"cacheRead": 0.02,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
@@ -65288,7 +65288,7 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 262144,
|
||||
"maxTokens": 81920,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
@@ -65311,12 +65311,12 @@
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.39,
|
||||
"output": 2.34,
|
||||
"input": 0.385,
|
||||
"output": 2.4499999999999997,
|
||||
"cacheRead": 0.195,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"contextWindow": 256000,
|
||||
"maxTokens": 65536,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
@@ -68906,7 +68906,7 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 262144,
|
||||
"maxTokens": 32768,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
"efforts": [
|
||||
@@ -68936,7 +68936,7 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 262144,
|
||||
"maxTokens": 32768,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
"efforts": [
|
||||
@@ -68950,7 +68950,7 @@
|
||||
},
|
||||
"umans-glm-5.1": {
|
||||
"id": "umans-glm-5.1",
|
||||
"name": "GLM 5.1",
|
||||
"name": "Umans GLM 5.1",
|
||||
"api": "anthropic-messages",
|
||||
"provider": "umans",
|
||||
"baseUrl": "https://api.code.umans.ai",
|
||||
@@ -68965,7 +68965,7 @@
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 204800,
|
||||
"contextWindow": 202752,
|
||||
"maxTokens": 131072,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
@@ -68980,7 +68980,7 @@
|
||||
},
|
||||
"umans-kimi-k2.6": {
|
||||
"id": "umans-kimi-k2.6",
|
||||
"name": "Kimi K2.6",
|
||||
"name": "Umans Kimi K2.6",
|
||||
"api": "anthropic-messages",
|
||||
"provider": "umans",
|
||||
"baseUrl": "https://api.code.umans.ai",
|
||||
@@ -68996,7 +68996,7 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 262144,
|
||||
"maxTokens": 32768,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
"efforts": [
|
||||
@@ -69010,7 +69010,7 @@
|
||||
},
|
||||
"umans-kimi-k2.7": {
|
||||
"id": "umans-kimi-k2.7",
|
||||
"name": "Kimi K2.7 Code",
|
||||
"name": "Umans Kimi K2.7 Code",
|
||||
"api": "anthropic-messages",
|
||||
"provider": "umans",
|
||||
"baseUrl": "https://api.code.umans.ai",
|
||||
@@ -69026,7 +69026,7 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 262144,
|
||||
"maxTokens": 32768,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
"efforts": [
|
||||
@@ -69041,7 +69041,7 @@
|
||||
},
|
||||
"umans-qwen3.6-35b-a3b": {
|
||||
"id": "umans-qwen3.6-35b-a3b",
|
||||
"name": "Qwen3.6 35B A3B",
|
||||
"name": "Umans Qwen3.6 35B A3B",
|
||||
"api": "anthropic-messages",
|
||||
"provider": "umans",
|
||||
"baseUrl": "https://api.code.umans.ai",
|
||||
@@ -69057,7 +69057,7 @@
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 262144,
|
||||
"maxTokens": 32768,
|
||||
"thinking": {
|
||||
"mode": "budget",
|
||||
"efforts": [
|
||||
@@ -70539,6 +70539,28 @@
|
||||
]
|
||||
}
|
||||
},
|
||||
"kimi-k2-7-code": {
|
||||
"id": "kimi-k2-7-code",
|
||||
"name": "kimi-k2-7-code",
|
||||
"api": "openai-completions",
|
||||
"provider": "venice",
|
||||
"baseUrl": "https://api.venice.ai/api/v1",
|
||||
"reasoning": false,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 256000,
|
||||
"maxTokens": 32768,
|
||||
"compat": {
|
||||
"supportsUsageInStreaming": false
|
||||
}
|
||||
},
|
||||
"kimi-k2-thinking": {
|
||||
"id": "kimi-k2-thinking",
|
||||
"name": "Kimi K2 Thinking",
|
||||
|
||||
@@ -664,7 +664,10 @@ function mapUmansModelInfo(
|
||||
...(supportsTools === false ? { supportsTools: false } : {}),
|
||||
cost: reference?.cost ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: toPositiveNumber(capabilities.context_window, reference?.contextWindow ?? null),
|
||||
maxTokens: toPositiveNumber(capabilities.max_completion_tokens, reference?.maxTokens ?? null),
|
||||
maxTokens: toPositiveNumber(
|
||||
capabilities.recommended_max_tokens,
|
||||
toPositiveNumber(capabilities.max_completion_tokens, reference?.maxTokens ?? null),
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1232,6 +1235,23 @@ export function clampFireworksKimiMaxTokens(modelId: string, candidate: number |
|
||||
return isFireworksKimiK2ModelId(modelId) ? Math.min(candidate, FIREWORKS_KIMI_MAX_TOKENS) : candidate;
|
||||
}
|
||||
|
||||
/**
|
||||
* Kimi K2.7 Code's documented recommended output budget. Some provider
|
||||
* discovery rows report the context-sized `max_completion_tokens` instead.
|
||||
*/
|
||||
export const KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS = 32_768;
|
||||
|
||||
export function isKimiK27CodeModelId(modelId: string): boolean {
|
||||
return /(?:^|\/)kimi[-._]?k2(?:[._-]?|p)7[-._]?code$/i.test(modelId);
|
||||
}
|
||||
|
||||
export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number): number;
|
||||
export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number | null): number | null;
|
||||
export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number | null): number | null {
|
||||
if (candidate === null) return null;
|
||||
return isKimiK27CodeModelId(modelId) ? Math.min(candidate, KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS) : candidate;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fireworks DeepSeek V4 accepts effort via `reasoning_effort` but rejects the
|
||||
* DeepSeek-native binary `thinking` toggle when both are present.
|
||||
@@ -2276,6 +2296,7 @@ export function veniceModelManagerOptions(
|
||||
const model = mapWithBundledReference(entry, defaults, reference);
|
||||
return {
|
||||
...model,
|
||||
maxTokens: clampKimiK27CodeMaxTokens(defaults.id, model.maxTokens),
|
||||
compat: { ...model.compat, supportsUsageInStreaming: false },
|
||||
};
|
||||
},
|
||||
@@ -3544,7 +3565,12 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_SPECIALIZED: readonly ModelsDevProviderDes
|
||||
// --- Synthetic ---
|
||||
openAiCompletionsDescriptor("synthetic", "synthetic", "https://api.synthetic.new/openai/v1"),
|
||||
// --- Venice AI ---
|
||||
openAiCompletionsDescriptor("venice", "venice", "https://api.venice.ai/api/v1"),
|
||||
openAiCompletionsDescriptor("venice", "venice", "https://api.venice.ai/api/v1", {
|
||||
transformModel: model => {
|
||||
const maxTokens = clampKimiK27CodeMaxTokens(model.id, model.maxTokens);
|
||||
return maxTokens === model.maxTokens ? model : { ...model, maxTokens };
|
||||
},
|
||||
}),
|
||||
// --- Ollama Cloud ---
|
||||
simpleModelsDevDescriptor("ollama-cloud", "ollama-cloud", "ollama-chat", "https://ollama.com"),
|
||||
// --- Xiaomi Token Plan ---
|
||||
|
||||
@@ -33,6 +33,7 @@ describe("umans provider catalog", () => {
|
||||
capabilities: {
|
||||
context_window: 262_144,
|
||||
max_completion_tokens: 262_144,
|
||||
recommended_max_tokens: 32_768,
|
||||
supports_vision: true,
|
||||
supports_tools: true,
|
||||
reasoning: { supported: true, can_disable: true, default_level: "medium" },
|
||||
@@ -43,6 +44,7 @@ describe("umans provider catalog", () => {
|
||||
capabilities: {
|
||||
context_window: 262_144,
|
||||
max_completion_tokens: 262_144,
|
||||
recommended_max_tokens: 32_768,
|
||||
supports_vision: true,
|
||||
supports_tools: true,
|
||||
reasoning: { supported: true, can_disable: false, default_level: "medium" },
|
||||
@@ -71,13 +73,14 @@ describe("umans provider catalog", () => {
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: 262_144,
|
||||
maxTokens: 262_144,
|
||||
maxTokens: 32_768,
|
||||
thinking: { defaultLevel: "medium" },
|
||||
});
|
||||
const mandatoryReasoningModel = models?.find(item => item.id === "umans-kimi-k2.7");
|
||||
expect(mandatoryReasoningModel).toMatchObject({
|
||||
id: "umans-kimi-k2.7",
|
||||
reasoning: true,
|
||||
maxTokens: 32_768,
|
||||
thinking: { defaultLevel: "medium", requiresEffort: true },
|
||||
});
|
||||
});
|
||||
@@ -137,7 +140,7 @@ describe("umans provider catalog", () => {
|
||||
reasoning: true,
|
||||
input: ["text", "image"],
|
||||
contextWindow: 262_144,
|
||||
maxTokens: 262_144,
|
||||
maxTokens: 32_768,
|
||||
});
|
||||
});
|
||||
|
||||
@@ -146,6 +149,7 @@ describe("umans provider catalog", () => {
|
||||
const model = providers.umans?.["umans-kimi-k2.7"];
|
||||
|
||||
expect(model).toBeDefined();
|
||||
expect(model.maxTokens).toBe(32_768);
|
||||
expect(model.thinking).toMatchObject({
|
||||
requiresEffort: true,
|
||||
});
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import {
|
||||
KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS,
|
||||
veniceModelManagerOptions,
|
||||
} from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
|
||||
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
describe("Venice provider catalog", () => {
|
||||
it("bundles Kimi K2.7 Code with its recommended output cap", () => {
|
||||
const model = getBundledModel("venice", "kimi-k2-7-code");
|
||||
|
||||
expect(model).toBeDefined();
|
||||
expect(model.maxTokens).toBe(KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS);
|
||||
});
|
||||
|
||||
it("caps Kimi K2.7 Code during runtime discovery", async () => {
|
||||
const requestedUrls: string[] = [];
|
||||
const fetchImpl: FetchImpl = async input => {
|
||||
requestedUrls.push(input instanceof Request ? input.url : String(input));
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
data: [
|
||||
{
|
||||
id: "kimi-k2-7-code",
|
||||
name: "kimi-k2-7-code",
|
||||
context_length: 256_000,
|
||||
max_completion_tokens: 262_144,
|
||||
},
|
||||
],
|
||||
}),
|
||||
{ status: 200, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
};
|
||||
|
||||
const options = veniceModelManagerOptions({ apiKey: "venice-test-key", fetch: fetchImpl });
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
const model = models?.find(candidate => candidate.id === "kimi-k2-7-code");
|
||||
|
||||
expect(requestedUrls).toEqual(["https://api.venice.ai/api/v1/models"]);
|
||||
expect(model).toBeDefined();
|
||||
expect(model?.maxTokens).toBe(KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user