fix(catalog): expose Baseten Kimi K3 thinking levels
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Baseten's `moonshotai/Kimi-K3` catalog metadata so its `low`/`high`/`max` thinking levels are available.
|
||||
|
||||
## [17.2.15] - 2026-08-12
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -13622,7 +13622,7 @@
|
||||
"api": "openai-completions",
|
||||
"provider": "baseten",
|
||||
"baseUrl": "https://inference.baseten.co/v1",
|
||||
"reasoning": false,
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
@@ -13636,7 +13636,20 @@
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 262144,
|
||||
"supportsComputerUse": false,
|
||||
"supportsComputerUseConfig": false
|
||||
"supportsComputerUseConfig": false,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"low",
|
||||
"high",
|
||||
"max"
|
||||
],
|
||||
"defaultLevel": "max",
|
||||
"effortMap": {
|
||||
"max": "max"
|
||||
},
|
||||
"requiresEffort": true
|
||||
}
|
||||
},
|
||||
"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": {
|
||||
"id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B",
|
||||
|
||||
@@ -3550,15 +3550,18 @@ export function basetenModelManagerOptions(
|
||||
const features = Array.isArray(raw.supported_features) ? raw.supported_features : [];
|
||||
const modalities = Array.isArray(raw.input_modalities) ? raw.input_modalities : [];
|
||||
|
||||
// Baseten's reasoning router accepts only the high/max
|
||||
// effort tiers for its GLM-5.2 and gpt-oss routes.
|
||||
const isEffortReasoning =
|
||||
// Keep dynamic discovery conservative: Baseten's generic feature
|
||||
// list can mention reasoning for routes whose wire dialect OMP does
|
||||
// not know. The shared model-thinking policy derives the exact
|
||||
// effort ladder after this spec is built.
|
||||
const isKnownBasetenReasoningRoute =
|
||||
isKimiK3ModelId(defaults.id) ||
|
||||
defaults.id === "openai/gpt-oss-120b" ||
|
||||
defaults.id === "deepseek-ai/DeepSeek-V4-Pro" ||
|
||||
defaults.id === "zai-org/GLM-5.2" ||
|
||||
defaults.id === "zai-org/GLM-5.2-Fast";
|
||||
const isBasetenNativeReasoning = isEffortReasoning || defaults.id === "deepseek-ai/DeepSeek-V4-Pro";
|
||||
const reasoning =
|
||||
isBasetenNativeReasoning && (features.includes("reasoning") || features.includes("reasoning_effort"));
|
||||
isKnownBasetenReasoningRoute && (features.includes("reasoning") || features.includes("reasoning_effort"));
|
||||
const supportsTools = features.includes("tools") ? undefined : false;
|
||||
const vision = modalities.includes("image") || (reference?.input.includes("image") ?? false);
|
||||
|
||||
@@ -3572,14 +3575,7 @@ export function basetenModelManagerOptions(
|
||||
|
||||
const contextWindow = toPositiveNumber(raw.context_length, reference?.contextWindow ?? defaults.contextWindow);
|
||||
const maxTokens = toPositiveNumber(raw.max_completion_tokens, reference?.maxTokens ?? defaults.maxTokens);
|
||||
|
||||
const baseModel = mapWithBundledReference(entry, defaults, reference);
|
||||
const thinking = isEffortReasoning
|
||||
? {
|
||||
mode: "effort" as const,
|
||||
efforts: [Effort.High, Effort.Max],
|
||||
}
|
||||
: undefined;
|
||||
|
||||
return {
|
||||
...baseModel,
|
||||
@@ -3588,7 +3584,6 @@ export function basetenModelManagerOptions(
|
||||
cost,
|
||||
contextWindow,
|
||||
maxTokens,
|
||||
...(thinking ? { thinking } : {}),
|
||||
...(supportsTools === false ? { supportsTools } : {}),
|
||||
};
|
||||
},
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { basetenModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
|
||||
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
@@ -28,6 +29,20 @@ describe("Baseten provider discovery", () => {
|
||||
input_cache_read: "0.00000016",
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "moonshotai/Kimi-K3",
|
||||
object: "model",
|
||||
name: "Kimi K3",
|
||||
context_length: 1048576,
|
||||
max_completion_tokens: 262144,
|
||||
supported_features: ["tools", "json_mode", "structured_outputs", "reasoning_effort"],
|
||||
input_modalities: ["text", "image"],
|
||||
pricing: {
|
||||
prompt: "0.000003",
|
||||
completion: "0.000015",
|
||||
input_cache_read: "0.0000003",
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "deepseek-ai/DeepSeek-V4-Pro",
|
||||
object: "model",
|
||||
@@ -90,6 +105,15 @@ describe("Baseten provider discovery", () => {
|
||||
},
|
||||
});
|
||||
|
||||
const kimiK3 = models?.find(model => model.id === "moonshotai/Kimi-K3");
|
||||
if (!kimiK3) throw new Error("Baseten Kimi K3 was not discovered");
|
||||
expect(kimiK3.reasoning).toBe(true);
|
||||
expect(buildModel(kimiK3).thinking).toMatchObject({
|
||||
mode: "effort",
|
||||
efforts: ["low", "high", "max"],
|
||||
defaultLevel: "max",
|
||||
});
|
||||
|
||||
const deepseek = models?.find(model => model.id === "deepseek-ai/DeepSeek-V4-Pro");
|
||||
expect(deepseek).toBeDefined();
|
||||
expect(deepseek).toMatchObject({
|
||||
@@ -110,14 +134,10 @@ describe("Baseten provider discovery", () => {
|
||||
|
||||
const glmFast = models?.find(model => model.id === "zai-org/GLM-5.2-Fast");
|
||||
expect(glmFast).toBeDefined();
|
||||
expect(glmFast).toMatchObject({
|
||||
provider: "baseten",
|
||||
api: "openai-completions",
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: ["high", "max"],
|
||||
},
|
||||
if (!glmFast) throw new Error("Baseten GLM-5.2 Fast was not discovered");
|
||||
expect(buildModel(glmFast).thinking).toMatchObject({
|
||||
mode: "effort",
|
||||
efforts: ["high", "max"],
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user