Merge PR #6721: fix(ai): preserve Kimi Code cache affinity (@usr-bin-roygbiv)

This commit is contained in:
can1357
2026-07-27 04:58:24 +02:00
6 changed files with 114 additions and 3 deletions
+1
View File
@@ -6,6 +6,7 @@
- Fixed OpenAI Responses replay treating a tool output as paired with a matching call that appeared later in the input, or a tool call as paired with an earlier output. Pair repair now respects wire order before preserving or synthesizing each side.
- Fixed adaptive-thinking Anthropic models omitting the interleaved-thinking beta on signature-enforcing proxies, which caused persisted interleaved assistant turns to fail on replay ([#6717](https://github.com/can1357/oh-my-pi/issues/6717)).
- Kimi Code now sends its session-stable prompt cache key on both supported transports: `prompt_cache_key` for OpenAI-compatible requests and `metadata.user_id` for Anthropic-compatible requests. Explicit keys survive side-channel session IDs, while `cacheRetention: "none"` still disables automatic affinity ([#6049](https://github.com/can1357/oh-my-pi/issues/6049)).
## [17.1.4] - 2026-07-26
@@ -7,7 +7,7 @@ import * as kimiOauth from "../../registry/oauth/kimi";
import { streamSimple } from "../../stream";
import type { Context, Model } from "../../types";
import type { MessageCreateParamsStreaming } from "../anthropic-wire";
import { type KimiApiFormat, streamKimi } from "../kimi";
import { type KimiApiFormat, type KimiOptions, streamKimi } from "../kimi";
import { streamOpenAIAnthropicShim } from "../openai-anthropic-shim";
import {
applyChatCompletionsCompatPolicy,
@@ -95,6 +95,35 @@ async function captureKimiPayload(
return payload;
}
async function captureKimiCachePayload(
format: KimiApiFormat,
options: Omit<KimiOptions, "apiKey" | "format" | "onPayload">,
): Promise<Record<string, unknown>> {
let payload: unknown;
const stream = streamKimi(
K3_MODEL,
{
systemPrompt: [],
messages: [{ role: "user", content: "Reply OK", timestamp: 0 }],
tools: [],
},
{
...options,
apiKey: "test-key",
format,
onPayload: body => {
payload = body;
throw new Error("stop after payload capture");
},
},
);
await stream.result();
if (payload === undefined || typeof payload !== "object" || payload === null) {
throw new Error("Kimi cache-affinity payload was not captured");
}
return payload as Record<string, unknown>;
}
afterEach(() => {
vi.restoreAllMocks();
});
@@ -122,6 +151,7 @@ describe("OpenAI/Anthropic compatibility shim cache affinity", () => {
{
apiKey: "test-key",
format: "openai",
cacheRetention: "none",
promptCacheKey: cacheKey,
fetch: async (_input, init) => {
requestHeaders = new Headers(init?.headers);
@@ -143,6 +173,71 @@ describe("OpenAI/Anthropic compatibility shim cache affinity", () => {
});
});
describe("Kimi Code prompt cache affinity", () => {
it("sends the explicit cache key on both supported transports", async () => {
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
const openaiPayload = await captureKimiCachePayload("openai", {
promptCacheKey: "stable-cache-key",
sessionId: "side-channel-session",
});
const anthropicPayload = await captureKimiCachePayload("anthropic", {
promptCacheKey: "stable-cache-key",
sessionId: "side-channel-session",
});
expect(openaiPayload.prompt_cache_key).toBe("stable-cache-key");
expect(anthropicPayload.metadata).toEqual({ user_id: "stable-cache-key" });
});
it("falls back to the provider session on both supported transports", async () => {
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
const openaiPayload = await captureKimiCachePayload("openai", { sessionId: "stable-session" });
const anthropicPayload = await captureKimiCachePayload("anthropic", { sessionId: "stable-session" });
expect(openaiPayload.prompt_cache_key).toBe("stable-session");
expect(anthropicPayload.metadata).toEqual({ user_id: "stable-session" });
});
it("preserves an explicit Anthropic metadata user id", async () => {
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
const payload = await captureKimiCachePayload("anthropic", {
metadata: { user_id: "caller-user-id" },
promptCacheKey: "automatic-cache-key",
});
expect(payload.metadata).toEqual({ user_id: "caller-user-id" });
});
it("falls back from an invalid Anthropic metadata user id", async () => {
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
const payload = await captureKimiCachePayload("anthropic", {
metadata: { user_id: 0 },
promptCacheKey: "automatic-cache-key",
});
expect(payload.metadata).toEqual({ user_id: "automatic-cache-key" });
});
it("omits automatic affinity when prompt caching is disabled", async () => {
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
const options = {
cacheRetention: "none",
promptCacheKey: "disabled-cache-key",
sessionId: "disabled-session",
} as const;
const openaiPayload = await captureKimiCachePayload("openai", options);
const anthropicPayload = await captureKimiCachePayload("anthropic", options);
expect(openaiPayload).not.toHaveProperty("prompt_cache_key");
expect(anthropicPayload).not.toHaveProperty("metadata");
});
});
describe("Kimi K3 thinking transport", () => {
it("sends every live named effort through Kimi's native thinking object by default", async () => {
vi.spyOn(kimiOauth, "getKimiCommonHeaders").mockReturnValue(KIMI_HEADERS);
+4 -1
View File
@@ -99,6 +99,7 @@ import {
hasCopilotVisionInput,
resolveGitHubCopilotBaseUrl,
} from "./github-copilot-headers";
import { getOpenAIPromptCacheKey } from "./openai-shared";
import { transformMessages } from "./transform-messages";
import { NON_VISION_IMAGE_PLACEHOLDER } from "./vision-guard";
@@ -3407,7 +3408,9 @@ function buildParams(
// Pre-compute metadata.
const metadataAccountId = readAnthropicMetadataAccountId(options?.metadata);
const metadataUserId = resolveAnthropicMetadataUserId(
options?.metadata?.user_id,
readMetadataString(options?.metadata, "user_id") ??
// Deliberately share the normalized affinity identity across Kimi's two transports.
(model.provider === "kimi-code" ? getOpenAIPromptCacheKey(options) : undefined),
isOAuthToken,
options?.sessionId,
metadataAccountId,
+1
View File
@@ -38,6 +38,7 @@ export function streamKimi(
anthropicBaseUrl: model.baseUrl.replace(/\/v1\/?$/, ""),
defaultFormat: model.compat.kimiApiFormat ?? "anthropic",
anthropicThinkingMode: model.compat.thinkingFormat === "kimi" ? "anthropic-adaptive" : undefined,
forwardCacheOptions: true,
extraHeaders: getKimiCommonHeaders,
});
}
@@ -31,6 +31,8 @@ export interface OpenAIAnthropicShimConfig {
defaultFormat: OpenAIAnthropicApiFormat;
/** Thinking transport used when this provider's Anthropic endpoint differs from generic budget semantics. */
anthropicThinkingMode?: ThinkingControlMode;
/** Forward cache-retention and request-metadata options to the selected transport. Default: false. */
forwardCacheOptions?: boolean;
/** Provider-specific headers (e.g. auth/session) merged ahead of user-supplied headers. */
extraHeaders?: () => Record<string, string>;
}
@@ -93,6 +95,8 @@ export function streamOpenAIAnthropicShim(
maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
signal: options?.signal,
headers: mergedHeaders,
cacheRetention: config.forwardCacheOptions ? options?.cacheRetention : undefined,
metadata: config.forwardCacheOptions ? options?.metadata : undefined,
sessionId: options?.sessionId,
promptCacheKey: options?.promptCacheKey,
onPayload: options?.onPayload,
@@ -131,6 +135,8 @@ export function streamOpenAIAnthropicShim(
maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
signal: options?.signal,
headers: mergedHeaders,
cacheRetention: config.forwardCacheOptions ? options?.cacheRetention : undefined,
metadata: config.forwardCacheOptions ? options?.metadata : undefined,
sessionId: options?.sessionId,
promptCacheKey: options?.promptCacheKey,
onPayload: options?.onPayload,
@@ -1498,6 +1498,11 @@ function applyOpenAIChatCompletionsPromptCachePolicy(
model: Model<"openai-completions">,
options: OpenAICompletionsOptions | undefined,
): void {
const promptCacheKey = getOpenAIPromptCacheKey(options);
if (model.provider === "kimi-code" && promptCacheKey !== undefined) {
params.prompt_cache_key = promptCacheKey;
}
const promptCache = options?.promptCache;
if (!promptCache || resolveCacheRetention(options?.cacheRetention) === "none") return;
if (!model.compat.supportsPromptCacheBreakpoints) {
@@ -1509,7 +1514,7 @@ function applyOpenAIChatCompletionsPromptCachePolicy(
return;
}
params.prompt_cache_key = getOpenAIPromptCacheKey(options);
params.prompt_cache_key = promptCacheKey;
params.prompt_cache_options = {
mode: promptCache.mode,
ttl: promptCache.ttl ?? model.compat.promptCacheBreakpointTtl,