Merge PR #6721: fix(ai): preserve Kimi Code cache affinity (@usr-bin-roygbiv)
This commit is contained in:
@@ -6,6 +6,7 @@
|
||||
|
||||
- Fixed OpenAI Responses replay treating a tool output as paired with a matching call that appeared later in the input, or a tool call as paired with an earlier output. Pair repair now respects wire order before preserving or synthesizing each side.
|
||||
- Fixed adaptive-thinking Anthropic models omitting the interleaved-thinking beta on signature-enforcing proxies, which caused persisted interleaved assistant turns to fail on replay ([#6717](https://github.com/can1357/oh-my-pi/issues/6717)).
|
||||
- Kimi Code now sends its session-stable prompt cache key on both supported transports: `prompt_cache_key` for OpenAI-compatible requests and `metadata.user_id` for Anthropic-compatible requests. Explicit keys survive side-channel session IDs, while `cacheRetention: "none"` still disables automatic affinity ([#6049](https://github.com/can1357/oh-my-pi/issues/6049)).
|
||||
|
||||
## [17.1.4] - 2026-07-26
|
||||
|
||||
|
||||
@@ -7,7 +7,7 @@ import * as kimiOauth from "../../registry/oauth/kimi";
|
||||
import { streamSimple } from "../../stream";
|
||||
import type { Context, Model } from "../../types";
|
||||
import type { MessageCreateParamsStreaming } from "../anthropic-wire";
|
||||
import { type KimiApiFormat, streamKimi } from "../kimi";
|
||||
import { type KimiApiFormat, type KimiOptions, streamKimi } from "../kimi";
|
||||
import { streamOpenAIAnthropicShim } from "../openai-anthropic-shim";
|
||||
import {
|
||||
applyChatCompletionsCompatPolicy,
|
||||
@@ -95,6 +95,35 @@ async function captureKimiPayload(
|
||||
return payload;
|
||||
}
|
||||
|
||||
async function captureKimiCachePayload(
|
||||
format: KimiApiFormat,
|
||||
options: Omit<KimiOptions, "apiKey" | "format" | "onPayload">,
|
||||
): Promise<Record<string, unknown>> {
|
||||
let payload: unknown;
|
||||
const stream = streamKimi(
|
||||
K3_MODEL,
|
||||
{
|
||||
systemPrompt: [],
|
||||
messages: [{ role: "user", content: "Reply OK", timestamp: 0 }],
|
||||
tools: [],
|
||||
},
|
||||
{
|
||||
...options,
|
||||
apiKey: "test-key",
|
||||
format,
|
||||
onPayload: body => {
|
||||
payload = body;
|
||||
throw new Error("stop after payload capture");
|
||||
},
|
||||
},
|
||||
);
|
||||
await stream.result();
|
||||
if (payload === undefined || typeof payload !== "object" || payload === null) {
|
||||
throw new Error("Kimi cache-affinity payload was not captured");
|
||||
}
|
||||
return payload as Record<string, unknown>;
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
@@ -122,6 +151,7 @@ describe("OpenAI/Anthropic compatibility shim cache affinity", () => {
|
||||
{
|
||||
apiKey: "test-key",
|
||||
format: "openai",
|
||||
cacheRetention: "none",
|
||||
promptCacheKey: cacheKey,
|
||||
fetch: async (_input, init) => {
|
||||
requestHeaders = new Headers(init?.headers);
|
||||
@@ -143,6 +173,71 @@ describe("OpenAI/Anthropic compatibility shim cache affinity", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("Kimi Code prompt cache affinity", () => {
|
||||
it("sends the explicit cache key on both supported transports", async () => {
|
||||
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
|
||||
|
||||
const openaiPayload = await captureKimiCachePayload("openai", {
|
||||
promptCacheKey: "stable-cache-key",
|
||||
sessionId: "side-channel-session",
|
||||
});
|
||||
const anthropicPayload = await captureKimiCachePayload("anthropic", {
|
||||
promptCacheKey: "stable-cache-key",
|
||||
sessionId: "side-channel-session",
|
||||
});
|
||||
|
||||
expect(openaiPayload.prompt_cache_key).toBe("stable-cache-key");
|
||||
expect(anthropicPayload.metadata).toEqual({ user_id: "stable-cache-key" });
|
||||
});
|
||||
|
||||
it("falls back to the provider session on both supported transports", async () => {
|
||||
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
|
||||
|
||||
const openaiPayload = await captureKimiCachePayload("openai", { sessionId: "stable-session" });
|
||||
const anthropicPayload = await captureKimiCachePayload("anthropic", { sessionId: "stable-session" });
|
||||
|
||||
expect(openaiPayload.prompt_cache_key).toBe("stable-session");
|
||||
expect(anthropicPayload.metadata).toEqual({ user_id: "stable-session" });
|
||||
});
|
||||
|
||||
it("preserves an explicit Anthropic metadata user id", async () => {
|
||||
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
|
||||
|
||||
const payload = await captureKimiCachePayload("anthropic", {
|
||||
metadata: { user_id: "caller-user-id" },
|
||||
promptCacheKey: "automatic-cache-key",
|
||||
});
|
||||
|
||||
expect(payload.metadata).toEqual({ user_id: "caller-user-id" });
|
||||
});
|
||||
|
||||
it("falls back from an invalid Anthropic metadata user id", async () => {
|
||||
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
|
||||
|
||||
const payload = await captureKimiCachePayload("anthropic", {
|
||||
metadata: { user_id: 0 },
|
||||
promptCacheKey: "automatic-cache-key",
|
||||
});
|
||||
|
||||
expect(payload.metadata).toEqual({ user_id: "automatic-cache-key" });
|
||||
});
|
||||
|
||||
it("omits automatic affinity when prompt caching is disabled", async () => {
|
||||
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
|
||||
const options = {
|
||||
cacheRetention: "none",
|
||||
promptCacheKey: "disabled-cache-key",
|
||||
sessionId: "disabled-session",
|
||||
} as const;
|
||||
|
||||
const openaiPayload = await captureKimiCachePayload("openai", options);
|
||||
const anthropicPayload = await captureKimiCachePayload("anthropic", options);
|
||||
|
||||
expect(openaiPayload).not.toHaveProperty("prompt_cache_key");
|
||||
expect(anthropicPayload).not.toHaveProperty("metadata");
|
||||
});
|
||||
});
|
||||
|
||||
describe("Kimi K3 thinking transport", () => {
|
||||
it("sends every live named effort through Kimi's native thinking object by default", async () => {
|
||||
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
|
||||
|
||||
@@ -99,6 +99,7 @@ import {
|
||||
hasCopilotVisionInput,
|
||||
resolveGitHubCopilotBaseUrl,
|
||||
} from "./github-copilot-headers";
|
||||
import { getOpenAIPromptCacheKey } from "./openai-shared";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
import { NON_VISION_IMAGE_PLACEHOLDER } from "./vision-guard";
|
||||
|
||||
@@ -3407,7 +3408,9 @@ function buildParams(
|
||||
// Pre-compute metadata.
|
||||
const metadataAccountId = readAnthropicMetadataAccountId(options?.metadata);
|
||||
const metadataUserId = resolveAnthropicMetadataUserId(
|
||||
options?.metadata?.user_id,
|
||||
readMetadataString(options?.metadata, "user_id") ??
|
||||
// Deliberately share the normalized affinity identity across Kimi's two transports.
|
||||
(model.provider === "kimi-code" ? getOpenAIPromptCacheKey(options) : undefined),
|
||||
isOAuthToken,
|
||||
options?.sessionId,
|
||||
metadataAccountId,
|
||||
|
||||
@@ -38,6 +38,7 @@ export function streamKimi(
|
||||
anthropicBaseUrl: model.baseUrl.replace(/\/v1\/?$/, ""),
|
||||
defaultFormat: model.compat.kimiApiFormat ?? "anthropic",
|
||||
anthropicThinkingMode: model.compat.thinkingFormat === "kimi" ? "anthropic-adaptive" : undefined,
|
||||
forwardCacheOptions: true,
|
||||
extraHeaders: getKimiCommonHeaders,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -31,6 +31,8 @@ export interface OpenAIAnthropicShimConfig {
|
||||
defaultFormat: OpenAIAnthropicApiFormat;
|
||||
/** Thinking transport used when this provider's Anthropic endpoint differs from generic budget semantics. */
|
||||
anthropicThinkingMode?: ThinkingControlMode;
|
||||
/** Forward cache-retention and request-metadata options to the selected transport. Default: false. */
|
||||
forwardCacheOptions?: boolean;
|
||||
/** Provider-specific headers (e.g. auth/session) merged ahead of user-supplied headers. */
|
||||
extraHeaders?: () => Record<string, string>;
|
||||
}
|
||||
@@ -93,6 +95,8 @@ export function streamOpenAIAnthropicShim(
|
||||
maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
|
||||
signal: options?.signal,
|
||||
headers: mergedHeaders,
|
||||
cacheRetention: config.forwardCacheOptions ? options?.cacheRetention : undefined,
|
||||
metadata: config.forwardCacheOptions ? options?.metadata : undefined,
|
||||
sessionId: options?.sessionId,
|
||||
promptCacheKey: options?.promptCacheKey,
|
||||
onPayload: options?.onPayload,
|
||||
@@ -131,6 +135,8 @@ export function streamOpenAIAnthropicShim(
|
||||
maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
|
||||
signal: options?.signal,
|
||||
headers: mergedHeaders,
|
||||
cacheRetention: config.forwardCacheOptions ? options?.cacheRetention : undefined,
|
||||
metadata: config.forwardCacheOptions ? options?.metadata : undefined,
|
||||
sessionId: options?.sessionId,
|
||||
promptCacheKey: options?.promptCacheKey,
|
||||
onPayload: options?.onPayload,
|
||||
|
||||
@@ -1498,6 +1498,11 @@ function applyOpenAIChatCompletionsPromptCachePolicy(
|
||||
model: Model<"openai-completions">,
|
||||
options: OpenAICompletionsOptions | undefined,
|
||||
): void {
|
||||
const promptCacheKey = getOpenAIPromptCacheKey(options);
|
||||
if (model.provider === "kimi-code" && promptCacheKey !== undefined) {
|
||||
params.prompt_cache_key = promptCacheKey;
|
||||
}
|
||||
|
||||
const promptCache = options?.promptCache;
|
||||
if (!promptCache || resolveCacheRetention(options?.cacheRetention) === "none") return;
|
||||
if (!model.compat.supportsPromptCacheBreakpoints) {
|
||||
@@ -1509,7 +1514,7 @@ function applyOpenAIChatCompletionsPromptCachePolicy(
|
||||
return;
|
||||
}
|
||||
|
||||
params.prompt_cache_key = getOpenAIPromptCacheKey(options);
|
||||
params.prompt_cache_key = promptCacheKey;
|
||||
params.prompt_cache_options = {
|
||||
mode: promptCache.mode,
|
||||
ttl: promptCache.ttl ?? model.compat.promptCacheBreakpointTtl,
|
||||
|
||||
Reference in New Issue
Block a user