feat(openai): add explicit prompt cache policy

This commit is contained in:
Alexander Kirilin
2026-07-23 14:01:24 -04:00
parent 7ae6ebfa8b
commit 61c24d6adf
17 changed files with 797 additions and 14 deletions
+1
View File
@@ -5,6 +5,7 @@
### Added
- Added Anthropic extra-usage reporting across `omp usage`, interactive `/usage`, and ACP `/usage`: the OAuth usage endpoint's authoritative `spend` payload (or legacy `extra_usage` fallback when absent) is normalized into a `Claude Extra Usage` USD row; capped accounts show limit/remaining/fractions and status, while uncapped spend exposes only its absolute used amount—rendered as `$… used` in CLI/TUI and `123.45 usd used` in ACP—without a fabricated cap, percentage, or status. ([#5575](https://github.com/can1357/oh-my-pi/issues/5575))
- Added opt-in OpenAI GPT-5.6 explicit prompt-cache controls for Responses and Chat Completions. Existing requests remain implicit; the policy marks at most one existing stable-history block and is rejected locally on unsupported explicit routes.
### Fixed
@@ -49,6 +49,35 @@ function isServiceTier(value: unknown): value is ServiceTier {
return value === "auto" || value === "default" || value === "flex" || value === "scale" || value === "priority";
}
const UNSUPPORTED_EXPLICIT_PROMPT_CACHE_MESSAGE =
"openai-chat: prompt_cache_options and prompt_cache_breakpoint are unsupported by this auth-gateway route; use /v1/pi/stream with options.promptCache instead";
function hasUnsupportedExplicitPromptCacheFields(body: unknown): boolean {
if (typeof body !== "object" || body === null || Array.isArray(body)) return false;
const request = body as Record<string, unknown>;
if ("prompt_cache_options" in request || "prompt_cache_breakpoint" in request) return true;
if (!Array.isArray(request.messages)) return false;
return request.messages.some(message => {
if (typeof message !== "object" || message === null || Array.isArray(message)) return false;
const wireMessage = message as Record<string, unknown>;
if ("prompt_cache_breakpoint" in wireMessage) return true;
return (
Array.isArray(wireMessage.content) &&
wireMessage.content.some(
part =>
typeof part === "object" && part !== null && !Array.isArray(part) && "prompt_cache_breakpoint" in part,
)
);
});
}
function rejectUnsupportedExplicitPromptCacheFields(body: unknown): void {
if (hasUnsupportedExplicitPromptCacheFields(body)) {
throw new AIError.ValidationError(UNSUPPORTED_EXPLICIT_PROMPT_CACHE_MESSAGE);
}
}
// ---------------------------------------------------------------------------
// parseRequest
// ---------------------------------------------------------------------------
@@ -59,6 +88,7 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
// land on `options.headers` automatically). We consult `headers` here too
// for `resolvePromptCacheKey` to pull a cache identity out of inbound
// vendor-neutral headers when the body doesn't carry one.
rejectUnsupportedExplicitPromptCacheFields(body);
const parsed = openaiChatRequestSchema(body);
if (parsed instanceof type.errors) {
throw new AIError.ValidationError(`openai-chat: ${parsed.summary}`);
@@ -196,6 +196,8 @@ export interface ChatCompletionContentPartText {
text: string;
/** Always `text`. */
type: "text";
/** Explicit OpenAI prompt-cache breakpoint. */
prompt_cache_breakpoint?: { mode: "explicit" };
}
/** Image content part. */
@@ -203,6 +205,8 @@ export interface ChatCompletionContentPartImage {
image_url: ChatCompletionContentPartImageImageURL;
/** Always `image_url`. */
type: "image_url";
/** Explicit OpenAI prompt-cache breakpoint. */
prompt_cache_breakpoint?: { mode: "explicit" };
}
/** SDK `ChatCompletionContentPartImage.ImageURL`. */
@@ -218,6 +222,8 @@ export interface ChatCompletionContentPartInputAudio {
input_audio: ChatCompletionContentPartInputAudioInputAudio;
/** Always `input_audio`. */
type: "input_audio";
/** Explicit OpenAI prompt-cache breakpoint. */
prompt_cache_breakpoint?: { mode: "explicit" };
}
/** SDK `ChatCompletionContentPartInputAudio.InputAudio`. */
@@ -233,6 +239,8 @@ export interface ChatCompletionContentPartFile {
file: ChatCompletionContentPartFileFile;
/** Always `file`. */
type: "file";
/** Explicit OpenAI prompt-cache breakpoint. */
prompt_cache_breakpoint?: { mode: "explicit" };
}
/** SDK `ChatCompletionContentPart.File.File`. */
@@ -251,6 +259,8 @@ export interface ChatCompletionContentPartRefusal {
refusal: string;
/** Always `refusal`. */
type: "refusal";
/** Explicit OpenAI prompt-cache breakpoint. */
prompt_cache_breakpoint?: { mode: "explicit" };
}
/** User-message content part union. */
@@ -798,6 +808,8 @@ export interface ChatCompletionCreateParamsBase {
prompt_cache_key?: string;
/** Retention policy for the prompt cache; `24h` enables extended caching. */
prompt_cache_retention?: "in_memory" | "24h" | null;
/** Explicit prompt-cache mode and minimum lifetime for GPT-5.6+ models. */
prompt_cache_options?: { mode: "implicit" | "explicit"; ttl?: "30m" };
/** Constrains effort on reasoning for reasoning models. */
reasoning_effort?: ReasoningEffort | null;
/** Output format: text, JSON mode, or Structured Outputs JSON schema. */
@@ -27,7 +27,7 @@ import type {
ToolChoice,
ToolResultMessage,
} from "../types";
import { normalizeSystemPrompts } from "../utils";
import { normalizeSystemPrompts, resolveCacheRetention } from "../utils";
import { createAbortSourceTracker } from "../utils/abort";
import { isDemotedThinking, kStreamingLastParseLen } from "../utils/block-symbols";
import { hasVisibleAssistantContent, withEmptyCompletionRetry } from "../utils/empty-completion-retry";
@@ -96,6 +96,7 @@ import {
isStrictToolsDisabledForScope,
type OpenAICompatPolicy,
type OpenAICompletionsParams,
type OpenAIPromptCacheOptions,
type OpenAIRequestSetup,
type OpenAIStrictToolsState,
parseAzureDeploymentNameMap,
@@ -481,6 +482,8 @@ export interface OpenAICompletionsOptions extends StreamOptions {
* with the variant baked in).
*/
openrouterVariant?: string;
/** Opt-in GPT-5.6+ prompt-cache policy. Unsupported explicit mode fails locally. */
promptCache?: OpenAIPromptCacheOptions;
}
type AppliedToolStrictMode = "mixed" | "all_strict" | "none";
@@ -1451,6 +1454,68 @@ function hasActiveNativeKimiK3Reasoning(
}
}
function isChatCompletionsPromptCacheableContentBlock(
block: unknown,
): block is { type: "text" | "image_url" | "input_audio" | "file"; prompt_cache_breakpoint?: { mode: "explicit" } } {
if (typeof block !== "object" || block === null || !("type" in block)) return false;
return block.type === "text" || block.type === "image_url" || block.type === "input_audio" || block.type === "file";
}
function markLatestStableChatCompletionsCacheBreakpoint(messages: ChatCompletionMessageParam[]): boolean {
let latestInputMessage = -1;
for (let i = messages.length - 1; i >= 0; i--) {
const message = messages[i];
if (message.role === "user" || message.role === "developer") {
latestInputMessage = i;
break;
}
}
if (latestInputMessage <= 0) return false;
for (let i = latestInputMessage - 1; i >= 0; i--) {
const message = messages[i];
if (message.role !== "user" && message.role !== "developer" && message.role !== "system") continue;
if (typeof message.content === "string") {
messages[i] = {
...message,
content: [{ type: "text", text: message.content, prompt_cache_breakpoint: { mode: "explicit" } }],
};
return true;
}
for (let j = message.content.length - 1; j >= 0; j--) {
const block = message.content[j];
if (!isChatCompletionsPromptCacheableContentBlock(block)) continue;
Object.assign(block, { prompt_cache_breakpoint: { mode: "explicit" } });
return true;
}
}
return false;
}
function applyOpenAIChatCompletionsPromptCachePolicy(
params: OpenAICompletionsParams,
model: Model<"openai-completions">,
options: OpenAICompletionsOptions | undefined,
): void {
const promptCache = options?.promptCache;
if (!promptCache || resolveCacheRetention(options?.cacheRetention) === "none") return;
if (!model.compat.supportsPromptCacheBreakpoints) {
if (promptCache.mode === "explicit") {
throw new AIError.ConfigurationError(
`OpenAI explicit prompt caching is unsupported for ${model.provider}/${model.id}; enable compat.supportsPromptCacheBreakpoints only for a compatible endpoint.`,
);
}
return;
}
params.prompt_cache_key = getOpenAIPromptCacheKey(options);
params.prompt_cache_options = {
mode: promptCache.mode,
ttl: promptCache.ttl ?? model.compat.promptCacheBreakpointTtl,
};
if (promptCache.breakpoint !== "none") markLatestStableChatCompletionsCacheBreakpoint(params.messages);
}
function buildParams(
model: Model<"openai-completions">,
context: Context,
@@ -1620,6 +1685,7 @@ function buildParams(
applyOpenAIExtraBody(params, compat.extraBody, {
dropThinkingWhenReasoningEffort: compat.dropThinkingWhenReasoningEffort,
});
applyOpenAIChatCompletionsPromptCachePolicy(params, model, options);
return { params, toolStrictMode, strictToolsApplied };
}
@@ -58,6 +58,27 @@ function isObj(v: unknown): v is Record<string, unknown> {
return typeof v === "object" && v !== null && !Array.isArray(v);
}
const UNSUPPORTED_EXPLICIT_PROMPT_CACHE_MESSAGE =
"openai-responses: prompt_cache_options and prompt_cache_breakpoint are unsupported by this auth-gateway route; use /v1/pi/stream with options.promptCache instead";
function hasUnsupportedExplicitPromptCacheFields(body: unknown): boolean {
if (!isObj(body)) return false;
if ("prompt_cache_options" in body || "prompt_cache_breakpoint" in body) return true;
if (!Array.isArray(body.input)) return false;
return body.input.some(item => {
if (!isObj(item)) return false;
if ("prompt_cache_breakpoint" in item) return true;
return Array.isArray(item.content) && item.content.some(part => isObj(part) && "prompt_cache_breakpoint" in part);
});
}
function rejectUnsupportedExplicitPromptCacheFields(body: unknown): void {
if (hasUnsupportedExplicitPromptCacheFields(body)) {
throw new AIError.ValidationError(UNSUPPORTED_EXPLICIT_PROMPT_CACHE_MESSAGE);
}
}
function asString(v: unknown): string | undefined {
return typeof v === "string" ? v : undefined;
}
@@ -294,6 +315,7 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
// client signals a cache identity outside the body — see the
// `resolvePromptCacheKey` call further down.
rejectUnsupportedExplicitPromptCacheFields(body);
const data = openaiResponsesRequestSchema(body);
if (data instanceof type.errors) {
throw new AIError.ValidationError(`openai-responses: ${data.summary}`);
@@ -2884,6 +2884,8 @@ export interface ResponseInputFile {
* The name of the file to be sent to the model.
*/
filename?: string;
/** Explicit OpenAI prompt-cache breakpoint. */
prompt_cache_breakpoint?: { mode: "explicit" };
}
/**
* A file input to the model.
@@ -2939,6 +2941,8 @@ export interface ResponseInputImage {
* encoded image in a data URL.
*/
image_url?: string | null;
/** Explicit OpenAI prompt-cache breakpoint. */
prompt_cache_breakpoint?: { mode: "explicit" };
}
/**
* An image input to the model. Learn about
@@ -3642,6 +3646,8 @@ export interface ResponseInputText {
* The type of the input item. Always `input_text`.
*/
type: "input_text";
/** Explicit OpenAI prompt-cache breakpoint. */
prompt_cache_breakpoint?: { mode: "explicit" };
}
/**
* A text input to the model.
@@ -5922,6 +5928,8 @@ export interface ResponseCreateParamsBase {
* `prompt_cache_retention` is not specified.
*/
prompt_cache_retention?: "in_memory" | "24h" | null;
/** Explicit prompt-cache mode and minimum lifetime for GPT-5.6+ models. */
prompt_cache_options?: { mode: "implicit" | "explicit"; ttl?: "30m" } | null;
/**
* **gpt-5 and o-series models only**
*
@@ -64,6 +64,7 @@ import type {
Tool as OpenAITool,
ResponseCreateParamsStreaming,
ResponseInput,
ResponseInputContent,
ResponseStreamEvent,
} from "./openai-responses-wire";
import {
@@ -86,6 +87,7 @@ import {
isOpenAIResponsesProgressEvent,
isOpenRouterAnthropicModel,
isStrictToolsDisabledForScope,
type OpenAIPromptCacheOptions,
type OpenAIStrictToolsScope,
type OpenAIStrictToolsState,
processResponsesStream,
@@ -150,6 +152,8 @@ export interface OpenAIResponsesOptions extends StreamOptions {
* prompt_cache_key for prompt-cache routing).
*/
extraBody?: Record<string, unknown>;
/** Opt-in GPT-5.6+ prompt-cache policy. Unsupported explicit mode fails locally. */
promptCache?: OpenAIPromptCacheOptions;
}
const OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX = "openai-responses:";
@@ -871,6 +875,97 @@ function isOfficialOpenAIResponsesEndpoint(model: Model<"openai-responses">): bo
}
}
function isResponsesPromptCacheableContentBlock(block: unknown): block is ResponseInputContent {
if (typeof block !== "object" || block === null || !("type" in block)) return false;
return block.type === "input_text" || block.type === "input_image" || block.type === "input_file";
}
type ResponsesPromptCacheableMessage = {
role: "assistant" | "developer" | "system" | "user";
content: ResponseInputContent[];
};
function isResponsesPromptCacheableMessage(item: unknown): item is ResponsesPromptCacheableMessage {
if (typeof item !== "object" || item === null || !("role" in item) || !("content" in item)) return false;
if (item.role !== "assistant" && item.role !== "developer" && item.role !== "system" && item.role !== "user")
return false;
return Array.isArray(item.content) && item.content.every(isResponsesPromptCacheableContentBlock);
}
type ResponsesStringInstruction = {
role: "developer" | "system";
content: string | ResponseInputContent[];
};
function isStableStringResponsesInstruction(item: unknown): item is ResponsesStringInstruction {
if (typeof item !== "object" || item === null || !("role" in item) || !("content" in item)) return false;
return (
(item.role === "developer" || item.role === "system") &&
typeof item.content === "string" &&
item.content.length > 0
);
}
function markLatestStableResponsesCacheBreakpoint(input: ResponseInput | undefined): boolean {
if (!input) return false;
let latestInputMessage = -1;
for (let i = input.length - 1; i >= 0; i--) {
const message = input[i];
if (!("role" in message)) continue;
if (message.role === "user" || message.role === "developer") {
latestInputMessage = i;
break;
}
}
if (latestInputMessage <= 0) return false;
for (let i = latestInputMessage - 1; i >= 0; i--) {
const message = input[i];
if (isResponsesPromptCacheableMessage(message)) {
const block = message.content[message.content.length - 1];
if (block) {
Object.assign(block, { prompt_cache_breakpoint: { mode: "explicit" } });
return true;
}
}
if (isStableStringResponsesInstruction(message) && typeof message.content === "string") {
const text = message.content;
message.content = [
{
type: "input_text",
text,
prompt_cache_breakpoint: { mode: "explicit" },
},
];
return true;
}
}
return false;
}
function applyOpenAIResponsesPromptCachePolicy(
params: OpenAIResponsesSamplingParams,
model: Model<"openai-responses">,
options: OpenAIResponsesOptions | undefined,
): void {
const promptCache = options?.promptCache;
if (!promptCache || resolveCacheRetention(options?.cacheRetention) === "none") return;
if (!model.compat.supportsPromptCacheBreakpoints) {
if (promptCache.mode === "explicit") {
throw new AIError.ConfigurationError(
`OpenAI explicit prompt caching is unsupported for ${model.provider}/${model.id}; enable compat.supportsPromptCacheBreakpoints only for a compatible endpoint.`,
);
}
return;
}
params.prompt_cache_options = {
mode: promptCache.mode,
ttl: promptCache.ttl ?? model.compat.promptCacheBreakpointTtl,
};
if (promptCache.breakpoint !== "none") markLatestStableResponsesCacheBreakpoint(params.input);
}
export function buildParams(
model: Model<"openai-responses">,
context: Context,
@@ -1027,6 +1122,7 @@ export function buildParams(
applyOpenAIGatewayRouting(params, model.compat);
applyOpenAIExtraBody(params, options?.extraBody);
applyOpenAIResponsesPromptCachePolicy(params, model, options);
return { params, strictToolsApplied };
}
@@ -54,6 +54,9 @@ import {
type ToolResultMessage,
type Usage,
} from "../types";
export type { OpenAIPromptCacheOptions } from "../types";
import {
getOpenAIResponsesHistoryItems,
getOpenAIResponsesHistoryPayload,
@@ -62,6 +62,7 @@ const ALLOWED_OPTION_KEYS: ReadonlySet<keyof SimpleStreamOptions> = new Set([
"metadata",
"sessionId",
"promptCacheKey",
"promptCache",
"streamFirstEventTimeoutMs",
"streamIdleTimeoutMs",
"reasoning",
+41
View File
@@ -1012,6 +1012,8 @@ export function streamSimple<TApi extends Api>(
...debugOptions,
fetch: wrapFetchForProxy(debugOptions.fetch ?? (globalThis.fetch as FetchImpl), model.provider),
} as SimpleStreamOptions;
assertExplicitOpenAIResponsesPromptCacheSupport(model, requestOptions);
const apiKeyResolver = isApiKeyResolver(requestOptions?.apiKey) ? requestOptions.apiKey : undefined;
if (apiKeyResolver) {
const outer = new AssistantMessageEventStream();
@@ -1401,6 +1403,40 @@ function normalizeMandatoryReasoningOptions<TApi extends Api>(
return { ...options, reasoning: floor, disableReasoning: undefined };
}
function supportsExplicitOpenAIResponsesPromptCache(compat: unknown): boolean {
return (
typeof compat === "object" &&
compat !== null &&
"supportsPromptCacheBreakpoints" in compat &&
compat.supportsPromptCacheBreakpoints === true
);
}
function isOpenAIResponsesPromptCacheSurface<TApi extends Api>(model: Model<TApi>): boolean {
return (
model.api === "openai-responses" ||
model.api === "azure-openai-responses" ||
(model.api === "openrouter" && $env.PI_OPENROUTER_RESPONSES !== "0")
);
}
function assertExplicitOpenAIResponsesPromptCacheSupport<TApi extends Api>(
model: Model<TApi>,
options?: SimpleStreamOptions,
): void {
if (
options?.cacheRetention === "none" ||
options?.promptCache?.mode !== "explicit" ||
!isOpenAIResponsesPromptCacheSurface(model) ||
supportsExplicitOpenAIResponsesPromptCache(model.compat)
) {
return;
}
throw new AIError.ConfigurationError(
`OpenAI explicit prompt caching is unsupported for ${model.provider}/${model.id}; enable compat.supportsPromptCacheBreakpoints only for a compatible endpoint.`,
);
}
function mapOptionsForApi<TApi extends Api>(
model: Model<TApi>,
rawOptions?: SimpleStreamOptions,
@@ -1575,6 +1611,7 @@ function mapOptionsForApi<TApi extends Api>(
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
disableReasoning: options?.disableReasoning,
textVerbosity: options?.textVerbosity,
promptCache: options?.promptCache,
});
}
return castApi<"openai-completions">({
@@ -1585,6 +1622,7 @@ function mapOptionsForApi<TApi extends Api>(
serviceTier: options?.serviceTier,
openrouterVariant: options?.openrouterVariant,
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
promptCache: options?.promptCache,
});
}
@@ -1597,6 +1635,7 @@ function mapOptionsForApi<TApi extends Api>(
serviceTier: options?.serviceTier,
openrouterVariant: options?.openrouterVariant,
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
promptCache: options?.promptCache,
});
case "openai-responses":
@@ -1610,6 +1649,7 @@ function mapOptionsForApi<TApi extends Api>(
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
disableReasoning: options?.disableReasoning,
textVerbosity: options?.textVerbosity,
promptCache: options?.promptCache,
});
case "azure-openai-responses":
@@ -1619,6 +1659,7 @@ function mapOptionsForApi<TApi extends Api>(
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
serviceTier: options?.serviceTier,
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
promptCache: options?.promptCache,
});
case "openai-codex-responses":
+16
View File
@@ -352,6 +352,16 @@ export interface CodexCompactionRequestContext extends CodexCompactionMetadata {
operationId: string;
}
/** OpenAI's GPT-5.6+ explicit prompt-cache controls. */
export interface OpenAIPromptCacheOptions {
/** `explicit` disables OpenAI's automatic latest-message breakpoint. */
mode: "implicit" | "explicit";
/** The only currently supported minimum breakpoint lifetime. */
ttl?: "30m";
/** By default, mark one existing block from stable history; `none` suppresses that marker. */
breakpoint?: "latest-stable-message" | "none";
}
export interface StreamOptions {
temperature?: number;
topP?: number;
@@ -422,6 +432,12 @@ export interface StreamOptions {
* `x-grok-conv-id`; when omitted, they fall back to `sessionId`.
*/
promptCacheKey?: string;
/**
* OpenAI GPT-5.6+ prompt-cache policy. Ignored by providers that do not
* support explicit OpenAI cache breakpoints; explicit mode fails locally on
* incompatible OpenAI-compatible endpoints.
*/
promptCache?: OpenAIPromptCacheOptions;
/**
* Provider-scoped mutable state store for this agent session.
* Providers can use this to persist transport/session state between turns.
@@ -142,6 +142,27 @@ describe("auth-gateway openai-chat: parseRequest", () => {
expect(parsed.options.extra).toEqual({ includeStreamingUsage: true });
});
it("rejects raw explicit prompt-cache controls instead of silently dropping them", () => {
expect(() =>
parseRequest({
model: "gpt-5.6",
messages: [{ role: "user", content: "hi" }],
prompt_cache_options: { mode: "explicit", ttl: "30m" },
}),
).toThrow("prompt_cache_options and prompt_cache_breakpoint are unsupported");
expect(() =>
parseRequest({
model: "gpt-5.6",
messages: [
{
role: "user",
content: [{ type: "text", text: "hi", prompt_cache_breakpoint: { mode: "explicit" } }],
},
],
}),
).toThrow("prompt_cache_options and prompt_cache_breakpoint are unsupported");
});
it("rejects missing required fields", () => {
expect(() => parseRequest({ messages: [] })).toThrow(/model/);
expect(() => parseRequest({ model: "x" })).toThrow(/messages/);
@@ -0,0 +1,88 @@
import { describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { clearCustomApis } from "@oh-my-pi/pi-ai/api-registry";
import { startAuthGateway } from "@oh-my-pi/pi-ai/auth-gateway";
import { AuthStorage } from "@oh-my-pi/pi-ai/auth-storage";
import { createMockModel, registerMockApi } from "@oh-my-pi/pi-ai/providers/mock";
describe("auth-gateway explicit OpenAI prompt cache controls", () => {
it("rejects raw controls clearly and forwards the pi-native policy", async () => {
registerMockApi();
const dir = await fs.mkdtemp(path.join(os.tmpdir(), "gw-openai-prompt-cache-"));
const storage = await AuthStorage.create(path.join(dir, "auth.db"));
storage.setRuntimeApiKey("mock", "test-key");
const mock = createMockModel({
provider: "mock",
id: "gateway-prompt-cache",
handler: () => ({ content: ["ok"] }),
});
const handle = startAuthGateway({
bind: "127.0.0.1:0",
bearerTokens: ["t"],
storage,
resolveModel: () => mock.model,
version: "test",
});
try {
const chatResponse = await fetch(`${handle.url}/v1/chat/completions`, {
method: "POST",
headers: { "Content-Type": "application/json", Authorization: "Bearer t" },
body: JSON.stringify({
model: "gateway-prompt-cache",
messages: [{ role: "user", content: "hi" }],
prompt_cache_options: { mode: "explicit", ttl: "30m" },
}),
});
const chatBody = (await chatResponse.json()) as { error?: { type?: string; message?: string } };
expect(chatResponse.status).toBe(400);
expect(chatBody.error).toEqual({
type: "invalid_request_error",
message:
"openai-chat: prompt_cache_options and prompt_cache_breakpoint are unsupported by this auth-gateway route; use /v1/pi/stream with options.promptCache instead",
});
const responsesResponse = await fetch(`${handle.url}/v1/responses`, {
method: "POST",
headers: { "Content-Type": "application/json", Authorization: "Bearer t" },
body: JSON.stringify({
model: "gateway-prompt-cache",
input: [
{
role: "user",
content: [{ type: "input_text", text: "hi", prompt_cache_breakpoint: { mode: "explicit" } }],
},
],
}),
});
const responsesBody = (await responsesResponse.json()) as { error?: { type?: string; message?: string } };
expect(responsesResponse.status).toBe(400);
expect(responsesBody.error).toEqual({
type: "invalid_request_error",
message:
"openai-responses: prompt_cache_options and prompt_cache_breakpoint are unsupported by this auth-gateway route; use /v1/pi/stream with options.promptCache instead",
});
const piResponse = await fetch(`${handle.url}/v1/pi/stream`, {
method: "POST",
headers: { "Content-Type": "application/json", Authorization: "Bearer t" },
body: JSON.stringify({
modelId: "gateway-prompt-cache",
context: { messages: [{ role: "user", content: "hi", timestamp: 0 }] },
options: { promptCache: { mode: "explicit", ttl: "30m", breakpoint: "none" } },
stream: false,
}),
});
expect(piResponse.status).toBe(200);
expect(mock.calls).toHaveLength(1);
expect(mock.calls[0]?.options?.promptCache).toEqual({ mode: "explicit", ttl: "30m", breakpoint: "none" });
} finally {
await handle.close();
storage.close();
clearCustomApis();
await fs.rm(dir, { recursive: true, force: true });
}
});
});
@@ -168,6 +168,27 @@ describe("openai-responses parseRequest", () => {
expect(parsed.options.extra).toBeUndefined();
});
it("rejects raw explicit prompt-cache controls instead of silently dropping them", () => {
expect(() =>
parseRequest({
model: "gpt-5.6",
input: "hi",
prompt_cache_options: { mode: "explicit", ttl: "30m" },
}),
).toThrow("prompt_cache_options and prompt_cache_breakpoint are unsupported");
expect(() =>
parseRequest({
model: "gpt-5.6",
input: [
{
role: "user",
content: [{ type: "input_text", text: "hi", prompt_cache_breakpoint: { mode: "explicit" } }],
},
],
}),
).toThrow("prompt_cache_options and prompt_cache_breakpoint are unsupported");
});
it("accepts a bare string input and rejects a missing model", () => {
const parsed = parseRequest({ model: "m", input: "hi" });
expect(parsed.context.messages).toHaveLength(1);
@@ -160,6 +160,16 @@ describe("pi-native parseRequest", () => {
expect(parsed.options.cacheRetention).toBe("long");
});
it("forwards the explicit prompt-cache policy through the canonical options bag", () => {
const parsed = parseRequest({
modelId: "gpt-5.6",
context: baseContext,
options: { promptCache: { mode: "explicit", ttl: "30m", breakpoint: "none" } },
});
expect(parsed.options.promptCache).toEqual({ mode: "explicit", ttl: "30m", breakpoint: "none" });
});
it("rejects missing required fields", () => {
expect(() => parseRequest({ context: baseContext })).toThrow(/modelId/);
expect(() => parseRequest({ modelId: "x" })).toThrow(/context/);
@@ -1,6 +1,8 @@
import { describe, expect, it } from "bun:test";
import { type OpenAICompletionsOptions, streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import type { Context, FetchImpl } from "@oh-my-pi/pi-ai/types";
import { streamSimple } from "@oh-my-pi/pi-ai/stream";
import type { AssistantMessage, Context, FetchImpl, Model, SimpleStreamOptions, Usage } from "@oh-my-pi/pi-ai/types";
import { buildOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
const model = getBundledModel<"openai-completions">("xai", "grok-code-fast-1");
@@ -8,6 +10,34 @@ if (!model) throw new Error("Expected bundled xAI Grok model");
if (model.api !== "openai-completions") throw new Error(`Expected Chat Completions model, received ${model.api}`);
const context: Context = { messages: [{ role: "user", content: "hello", timestamp: 0 }] };
const openAI56ResponsesModel = getBundledModel<"openai-responses">("openai", "gpt-5.6");
if (!openAI56ResponsesModel) throw new Error("Expected bundled OpenAI GPT-5.6 model");
if (openAI56ResponsesModel.api !== "openai-responses") {
throw new Error(`Expected OpenAI Responses model, received ${openAI56ResponsesModel.api}`);
}
const {
compat: _responsesCompat,
remoteCompaction: _responsesRemoteCompaction,
...openAI56CompletionsSpec
} = openAI56ResponsesModel;
const openAI56CompletionsModel: Model<"openai-completions"> = {
...openAI56CompletionsSpec,
api: "openai-completions",
compat: buildOpenAICompat({
...openAI56CompletionsSpec,
api: "openai-completions",
}),
};
const emptyUsage: Usage = {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
function chatCompletionsSse(): Response {
const chunk = (delta: unknown, finishReason: string | null) =>
JSON.stringify({
@@ -24,25 +54,54 @@ function chatCompletionsSse(): Response {
);
}
async function captureRequestHeaders(options: OpenAICompletionsOptions): Promise<Headers> {
async function captureRequest(
options: OpenAICompletionsOptions,
requestModel: Model<"openai-completions"> = model,
requestContext: Context = context,
): Promise<{ headers: Headers; body: Record<string, unknown> }> {
let requestHeaders: Headers | undefined;
let body: Record<string, unknown> | undefined;
const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => {
const request =
input instanceof Request
? new Request(input, init)
: new Request(input instanceof URL ? input.href : input, init);
requestHeaders = request.headers;
body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : undefined;
return chatCompletionsSse();
};
await streamOpenAICompletions(model, context, {
await streamOpenAICompletions(requestModel, requestContext, {
apiKey: "test-key",
...options,
fetch: fetchMock,
}).result();
if (!requestHeaders) throw new Error("Expected a serialized Chat Completions request");
return requestHeaders;
if (!requestHeaders || !body) throw new Error("Expected a serialized Chat Completions request");
return { headers: requestHeaders, body };
}
async function captureSimpleRequest(
options: SimpleStreamOptions,
requestModel: Model<"openai-completions"> = model,
requestContext: Context = context,
): Promise<{ headers: Headers; body: Record<string, unknown> }> {
let requestHeaders: Headers | undefined;
let body: Record<string, unknown> | undefined;
const fetchMock: FetchImpl = async (input: string | URL | Request, init?: RequestInit) => {
const request =
input instanceof Request
? new Request(input, init)
: new Request(input instanceof URL ? input.href : input, init);
requestHeaders = request.headers;
body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : undefined;
return chatCompletionsSse();
};
await streamSimple(requestModel, requestContext, { apiKey: "test-key", ...options, fetch: fetchMock }).result();
if (!requestHeaders || !body) throw new Error("Expected a serialized Chat Completions request");
return { headers: requestHeaders, body };
}
describe("openai-completions xAI cache affinity", () => {
@@ -83,9 +142,81 @@ describe("openai-completions xAI cache affinity", () => {
for (const { name, options, expectedHeader } of cases) {
it(name, async () => {
const headers = await captureRequestHeaders(options);
const { headers } = await captureRequest(options);
expect(headers.get("x-grok-conv-id")).toBe(expectedHeader);
});
}
});
describe("OpenAI Chat Completions explicit prompt cache policy", () => {
const historicalContext: Context = {
messages: [
{ role: "user", content: [{ type: "text", text: "stable history" }], timestamp: 0 },
{ role: "user", content: [{ type: "text", text: "current prompt" }], timestamp: 1 },
],
};
it("leaves the wire shape unchanged when the policy is unset", async () => {
const { body } = await captureRequest({ sessionId: "cache-key" }, openAI56CompletionsModel, historicalContext);
expect(body).not.toHaveProperty("prompt_cache_options");
expect(body).not.toHaveProperty("prompt_cache_key");
});
it("routes explicit policy through streamSimple and marks existing text-only history", async () => {
const previousAssistant: AssistantMessage = {
role: "assistant",
content: [{ type: "text", text: "previous answer" }],
api: "openai-completions",
provider: "openai",
model: "gpt-5.6",
usage: emptyUsage,
stopReason: "stop",
timestamp: 1,
};
const textOnlyHistory: Context = {
messages: [
{ role: "user", content: "stable history", timestamp: 0 },
previousAssistant,
{ role: "user", content: "current prompt", timestamp: 2 },
],
};
const { body } = await captureSimpleRequest(
{ sessionId: "cache-key", promptCache: { mode: "explicit" } },
openAI56CompletionsModel,
textOnlyHistory,
);
expect(body.prompt_cache_key).toBe("cache-key");
expect(body.prompt_cache_options).toEqual({ mode: "explicit", ttl: "30m" });
const messages = body.messages;
if (!Array.isArray(messages)) throw new Error("Expected Chat Completions messages");
expect(messages).toHaveLength(3);
expect(messages[0]).toMatchObject({
content: [{ type: "text", text: "stable history", prompt_cache_breakpoint: { mode: "explicit" } }],
});
expect(messages[1]).toMatchObject({ content: "previous answer" });
expect(messages[2]).toMatchObject({ content: "current prompt" });
});
it("does not synthesize first-turn content or a caller-disabled breakpoint", async () => {
const firstTurn: Context = {
messages: [{ role: "user", content: "only prompt", timestamp: 0 }],
};
const first = await captureRequest({ promptCache: { mode: "explicit" } }, openAI56CompletionsModel, firstTurn);
const none = await captureRequest(
{ promptCache: { mode: "explicit", breakpoint: "none" } },
openAI56CompletionsModel,
historicalContext,
);
for (const body of [first.body, none.body]) {
const messages = body.messages;
if (!Array.isArray(messages)) throw new Error("Expected Chat Completions messages");
for (const message of messages) {
expect(message).not.toMatchObject({ content: [{ prompt_cache_breakpoint: { mode: "explicit" } }] });
}
}
});
});
@@ -1,8 +1,14 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { type OpenAIResponsesOptions, streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
import {
buildParams,
type OpenAIResponsesOptions,
streamOpenAIResponses,
} from "@oh-my-pi/pi-ai/providers/openai-responses";
import { stream as streamModel, streamSimple } from "@oh-my-pi/pi-ai/stream";
import type { Context, FetchImpl, Model, ProviderSessionState, SimpleStreamOptions } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
const model = getBundledModel("openai", "gpt-5-mini") as Model<"openai-responses">;
@@ -47,6 +53,31 @@ const xaiOAuthResponsesModel: Model<"openai-responses"> = {
}),
};
const openAI56ResponsesModel: Model<"openai-responses"> = {
...model,
id: "gpt-5.6",
name: "GPT-5.6",
compat: buildOpenAIResponsesCompat({
id: "gpt-5.6",
name: "GPT-5.6",
provider: "openai",
baseUrl: "https://api.openai.com/v1",
}),
};
const azureOpenAI56ResponsesModel: Model<"azure-openai-responses"> = buildModel({
id: "gpt-5.6",
name: "GPT-5.6",
api: "azure-openai-responses",
provider: "azure",
baseUrl: "https://example.openai.azure.com/openai/v1",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 400_000,
maxTokens: 128_000,
});
function createSseResponse(events: unknown[]): Response {
const payload = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`;
return new Response(payload, {
@@ -192,6 +223,10 @@ async function captureDispatchedOpenAIResponseHeaders(
async function captureSimpleOpenAIResponseBody(
options: SimpleStreamOptions,
requestModel: Model<"openai-responses"> = model,
requestContext: Context = {
systemPrompt: ["stable system", "stable durable context"],
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
},
): Promise<Record<string, unknown> | null> {
let body: Record<string, unknown> | null = null;
const fetchMock: FetchImpl = vi.fn(async (_input: string | URL | Request, init?: RequestInit) => {
@@ -228,12 +263,7 @@ async function captureSimpleOpenAIResponseBody(
]);
});
const context: Context = {
systemPrompt: ["stable system", "stable durable context"],
messages: [{ role: "user", content: "hi", timestamp: Date.now() }],
};
const stream = streamSimple(requestModel, context, { apiKey: "test-key", ...options, fetch: fetchMock });
const stream = streamSimple(requestModel, requestContext, { apiKey: "test-key", ...options, fetch: fetchMock });
for await (const event of stream) {
if (event.type === "done" || event.type === "error") break;
}
@@ -245,6 +275,192 @@ afterEach(() => {
vi.restoreAllMocks();
});
describe("OpenAI Responses explicit prompt cache policy", () => {
const historicalContext: Context = {
messages: [
{ role: "user", content: [{ type: "text", text: "stable history" }], timestamp: 0 },
{ role: "user", content: [{ type: "text", text: "current prompt" }], timestamp: 1 },
],
};
it("leaves the existing request shape unchanged when the policy is unset", () => {
const params = buildParams(
openAI56ResponsesModel,
historicalContext,
{ sessionId: "cache-key" },
undefined,
).params;
expect(params.prompt_cache_key).toBe("cache-key");
expect(params).not.toHaveProperty("prompt_cache_options");
const [firstMessage] = params.input ?? [];
if (!firstMessage || !("content" in firstMessage) || !Array.isArray(firstMessage.content)) {
throw new Error("Expected Responses input message content");
}
expect(firstMessage.content[0]).not.toHaveProperty("prompt_cache_breakpoint");
});
it("marks one existing stable history block and leaves the current prompt unmodified", () => {
const params = buildParams(
openAI56ResponsesModel,
historicalContext,
{ sessionId: "cache-key", promptCache: { mode: "explicit" } },
undefined,
).params;
expect(params.prompt_cache_options).toEqual({ mode: "explicit", ttl: "30m" });
const [historical, current] = params.input ?? [];
if (
!historical ||
!current ||
!("content" in historical) ||
!Array.isArray(historical.content) ||
!("content" in current) ||
!Array.isArray(current.content)
) {
throw new Error("Expected Responses input message content");
}
expect(historical.content[0]).toMatchObject({ prompt_cache_breakpoint: { mode: "explicit" } });
expect(current.content[0]).not.toHaveProperty("prompt_cache_breakpoint");
expect(historicalContext.messages[0].content).toEqual([{ type: "text", text: "stable history" }]);
});
it("marks an existing first-turn developer string without adding a message or changing its text", () => {
const firstTurnWithSystem: Context = {
systemPrompt: ["stable developer instruction"],
messages: [{ role: "user", content: [{ type: "text", text: "only prompt" }], timestamp: 0 }],
};
const params = buildParams(
openAI56ResponsesModel,
firstTurnWithSystem,
{ promptCache: { mode: "explicit" } },
undefined,
).params;
expect(params.input).toEqual([
{
role: "developer",
content: [
{
type: "input_text",
text: "stable developer instruction",
prompt_cache_breakpoint: { mode: "explicit" },
},
],
},
{ role: "user", content: [{ type: "input_text", text: "only prompt" }] },
]);
});
it("routes explicit policy through streamSimple", async () => {
const body = await captureSimpleOpenAIResponseBody(
{ sessionId: "cache-key", promptCache: { mode: "explicit" } },
openAI56ResponsesModel,
historicalContext,
);
expect(body?.prompt_cache_key).toBe("cache-key");
expect(body?.prompt_cache_options).toEqual({ mode: "explicit", ttl: "30m" });
const input = body?.input;
if (!Array.isArray(input)) throw new Error("Expected Responses input");
expect(input[0]).toMatchObject({
content: [{ type: "input_text", text: "stable history", prompt_cache_breakpoint: { mode: "explicit" } }],
});
});
it("does not manufacture a breakpoint on a first-turn prompt or when the caller opts out", () => {
const firstTurn: Context = {
messages: [{ role: "user", content: [{ type: "text", text: "only prompt" }], timestamp: 0 }],
};
const firstTurnParams = buildParams(
openAI56ResponsesModel,
firstTurn,
{ promptCache: { mode: "explicit" } },
undefined,
).params;
const noBreakpointParams = buildParams(
openAI56ResponsesModel,
historicalContext,
{ promptCache: { mode: "explicit", breakpoint: "none" } },
undefined,
).params;
for (const params of [firstTurnParams, noBreakpointParams]) {
for (const item of params.input ?? []) {
if (!("content" in item) || !Array.isArray(item.content)) continue;
for (const block of item.content) {
expect(block).not.toHaveProperty("prompt_cache_breakpoint");
}
}
}
});
it("rejects explicit policy through streamSimple before sending unsupported Responses requests", () => {
const unsupportedModel: Model<"openai-responses"> = {
...openAI56ResponsesModel,
id: "gpt-5.5",
compat: buildOpenAIResponsesCompat({
id: "gpt-5.5",
name: "GPT-5.5",
provider: "openai",
baseUrl: "https://api.openai.com/v1",
}),
};
const fetchMock: FetchImpl = vi.fn(async () => {
throw new Error("Unsupported Responses requests must not reach fetch");
});
const context: Context = {
messages: [{ role: "user", content: [{ type: "text", text: "prompt" }], timestamp: 0 }],
};
const options: SimpleStreamOptions = {
apiKey: "test-key",
promptCache: { mode: "explicit" },
fetch: fetchMock,
};
expect(() => streamSimple(unsupportedModel, context, options)).toThrow(
"OpenAI explicit prompt caching is unsupported",
);
expect(() => streamSimple(azureOpenAI56ResponsesModel, context, options)).toThrow(
"OpenAI explicit prompt caching is unsupported",
);
expect(fetchMock).not.toHaveBeenCalled();
});
it("treats cacheRetention none as a disabled no-op before public policy validation", async () => {
const unsupportedModel: Model<"openai-responses"> = {
...openAI56ResponsesModel,
id: "gpt-5.5",
compat: buildOpenAIResponsesCompat({
id: "gpt-5.5",
name: "GPT-5.5",
provider: "openai",
baseUrl: "https://api.openai.com/v1",
}),
};
const body = await captureSimpleOpenAIResponseBody(
{ cacheRetention: "none", promptCache: { mode: "explicit" } },
unsupportedModel,
);
if (body === null) throw new Error("Expected disabled prompt-cache request to reach the provider");
expect(body).not.toHaveProperty("prompt_cache_options");
const input = body.input;
if (!Array.isArray(input)) throw new Error("Expected Responses input");
const contentBlocks = input.flatMap(item => {
if (typeof item !== "object" || item === null || !("content" in item) || !Array.isArray(item.content)) {
return [];
}
return item.content;
});
expect(contentBlocks.length).toBeGreaterThan(0);
for (const block of contentBlocks) {
expect(block).not.toHaveProperty("prompt_cache_breakpoint");
}
});
});
describe("openai-responses cache affinity", () => {
it("sets session routing headers for official OpenAI Responses requests with a sessionId", async () => {
const captured = await captureOpenAIResponseHeaders({ sessionId: "session-123" });