fix(catalog): corrected moonshot kimi request shaping

Moonshot-native OpenAI-compatible requests now omit OpenAI-only store metadata and use max_tokens for Kimi output budgets.\n\nFixes #2289
This commit is contained in:
roboomp
2026-06-11 05:28:17 +00:00
parent e6ee124d7c
commit 8a36b8feca
3 changed files with 23 additions and 4 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased] ## [Unreleased]
### Fixed
- Fixed Moonshot/Kimi native OpenAI-compatible request metadata so Kimi K2 uses `max_tokens` and omits OpenAI-only `store`, restoring first-turn output with `MOONSHOT_API_KEY` ([#2289](https://github.com/can1357/oh-my-pi/issues/2289)).
## [15.11.0] - 2026-06-10 ## [15.11.0] - 2026-06-10
### Fixed ### Fixed
+8 -2
View File
@@ -102,7 +102,8 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
const isZhipu = modelMatchesHost(hostModel, "zhipu"); const isZhipu = modelMatchesHost(hostModel, "zhipu");
const isKilo = modelMatchesHost(hostModel, "kilo"); const isKilo = modelMatchesHost(hostModel, "kilo");
const isKimiModel = isKimiModelId(spec.id); const isKimiModel = isKimiModelId(spec.id);
const isMoonshotKimi = isKimiModel && modelMatchesHost(hostModel, "moonshotNative"); const isMoonshotNative = modelMatchesHost(hostModel, "moonshotNative");
const isMoonshotKimi = isKimiModel && isMoonshotNative;
const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(spec.id); const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(spec.id);
const isAnthropicModel = const isAnthropicModel =
modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(spec.id) || isAnthropicNamespacedModelId(spec.id); modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(spec.id) || isAnthropicNamespacedModelId(spec.id);
@@ -145,11 +146,16 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
isKilo || isKilo ||
isQwen || isQwen ||
isXiaomiHost || isXiaomiHost ||
isMoonshotNative ||
isOpenCodeHost; isOpenCodeHost;
const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen"; const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen";
const useMaxTokens = const useMaxTokens =
isMistral || hostMatchesUrl(baseUrl, "chutes") || hostMatchesUrl(baseUrl, "fireworks") || isDirectDeepseekApi; isMistral ||
isMoonshotNative ||
hostMatchesUrl(baseUrl, "chutes") ||
hostMatchesUrl(baseUrl, "fireworks") ||
isDirectDeepseekApi;
// Hosts whose chat-completions endpoints are known to accept multiple // Hosts whose chat-completions endpoints are known to accept multiple
// leading `system`/`developer` messages (preferred for KV-cache reuse). // leading `system`/`developer` messages (preferred for KV-cache reuse).
+11 -2
View File
@@ -17,7 +17,7 @@
* moonshot discovery mapper and stamps default thinking metadata. * moonshot discovery mapper and stamps default thinking metadata.
*/ */
import { describe, expect, it } from "bun:test"; import { describe, expect, it } from "bun:test";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { type OpenAICompletionsOptions, streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import type { AssistantMessage, Context } from "@oh-my-pi/pi-ai/types"; import type { AssistantMessage, Context } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { Effort } from "@oh-my-pi/pi-catalog/effort";
@@ -82,6 +82,7 @@ interface CapturedRequest {
async function runHiTurn( async function runHiTurn(
model: Model<"openai-completions">, model: Model<"openai-completions">,
options?: Pick<OpenAICompletionsOptions, "reasoning">,
): Promise<{ captured: CapturedRequest; assistant: AssistantMessage }> { ): Promise<{ captured: CapturedRequest; assistant: AssistantMessage }> {
const captured: CapturedRequest = { url: "", body: {} }; const captured: CapturedRequest = { url: "", body: {} };
const fetchMock = (async (input: string | URL | Request, init?: RequestInit): Promise<Response> => { const fetchMock = (async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
@@ -92,7 +93,7 @@ async function runHiTurn(
return buildMockMoonshotResponse(); return buildMockMoonshotResponse();
}) as typeof fetch; }) as typeof fetch;
const stream = streamOpenAICompletions(model, basicContext(), { apiKey: "test-key", fetch: fetchMock }); const stream = streamOpenAICompletions(model, basicContext(), { apiKey: "test-key", fetch: fetchMock, ...options });
for await (const _ of stream) { for await (const _ of stream) {
// drain until terminal event // drain until terminal event
} }
@@ -155,6 +156,14 @@ describe("issue #2113 — moonshot kimi-k2.6 discovery and wire format", () => {
expect(textBlock).toBeDefined(); expect(textBlock).toBeDefined();
}); });
it("uses Moonshot-native max_tokens and omits OpenAI store control", async () => {
const model = moonshotKimiModel("kimi-k2.5", true);
const { captured } = await runHiTurn(model, { reasoning: "high" });
expect(captured.body.max_tokens).toBeDefined();
expect(captured.body.max_completion_tokens).toBeUndefined();
expect(captured.body.store).toBeUndefined();
});
it("wire body includes thinking.keep='all' when reasoning is explicitly requested", async () => { it("wire body includes thinking.keep='all' when reasoning is explicitly requested", async () => {
const model = moonshotKimiModel("kimi-k2.6", true); const model = moonshotKimiModel("kimi-k2.6", true);
const captured: CapturedRequest = { url: "", body: {} }; const captured: CapturedRequest = { url: "", body: {} };