feat(gateway): map Vercel Responses cache anchors

This commit is contained in:
Alexander Kirilin
2026-07-23 13:43:46 -04:00
parent 6ed55a2c12
commit ed0f5d6e29
9 changed files with 266 additions and 6 deletions
+1
View File
@@ -6,6 +6,7 @@
- Added Anthropic extra-usage reporting across `omp usage`, interactive `/usage`, and ACP `/usage`: the OAuth usage endpoint's authoritative `spend` payload (or legacy `extra_usage` fallback when absent) is normalized into a `Claude Extra Usage` USD row; capped accounts show limit/remaining/fractions and status, while uncapped spend exposes only its absolute used amount—rendered as `$… used` in CLI/TUI and `123.45 usd used` in ACP—without a fabricated cap, percentage, or status. ([#5575](https://github.com/can1357/oh-my-pi/issues/5575)) - Added Anthropic extra-usage reporting across `omp usage`, interactive `/usage`, and ACP `/usage`: the OAuth usage endpoint's authoritative `spend` payload (or legacy `extra_usage` fallback when absent) is normalized into a `Claude Extra Usage` USD row; capped accounts show limit/remaining/fractions and status, while uncapped spend exposes only its absolute used amount—rendered as `$… used` in CLI/TUI and `123.45 usd used` in ACP—without a fabricated cap, percentage, or status. ([#5575](https://github.com/can1357/oh-my-pi/issues/5575))
- Added opt-in Vercel AI Gateway automatic prompt caching for OpenAI Chat Completions while preserving `only` and `order` routing preferences. - Added opt-in Vercel AI Gateway automatic prompt caching for OpenAI Chat Completions while preserving `only` and `order` routing preferences.
- Added Vercel AI Gateway Responses cache anchors and cache lifetimes, emitted only with automatic caching.
### Fixed ### Fixed
@@ -27,7 +27,7 @@ import type {
ToolChoice, ToolChoice,
ToolResultMessage, ToolResultMessage,
} from "../types"; } from "../types";
import { normalizeSystemPrompts } from "../utils"; import { normalizeSystemPrompts, resolveCacheRetention } from "../utils";
import { createAbortSourceTracker } from "../utils/abort"; import { createAbortSourceTracker } from "../utils/abort";
import { isDemotedThinking, kStreamingLastParseLen } from "../utils/block-symbols"; import { isDemotedThinking, kStreamingLastParseLen } from "../utils/block-symbols";
import { hasVisibleAssistantContent, withEmptyCompletionRetry } from "../utils/empty-completion-retry"; import { hasVisibleAssistantContent, withEmptyCompletionRetry } from "../utils/empty-completion-retry";
@@ -1463,6 +1463,7 @@ function buildParams(
} { } {
const initialPolicy = resolveOpenAICompatForRequest(model, options); const initialPolicy = resolveOpenAICompatForRequest(model, options);
const initialCompat = initialPolicy.compat as ResolvedOpenAICompat; const initialCompat = initialPolicy.compat as ResolvedOpenAICompat;
const cacheRetention = resolveCacheRetention(options?.cacheRetention);
const requestModelId = resolveOpenAICompletionsModelId(model, options); const requestModelId = resolveOpenAICompletionsModelId(model, options);
const params: OpenAICompletionsParams = { const params: OpenAICompletionsParams = {
@@ -1615,7 +1616,7 @@ function buildParams(
applyChatCompletionsCompatPolicy(params, finalPolicy); applyChatCompletionsCompatPolicy(params, finalPolicy);
dropOpenRouterKimiForcedToolReasoning(params, model, finalPolicy); dropOpenRouterKimiForcedToolReasoning(params, model, finalPolicy);
applyOpenAIGatewayRouting(params, compat); applyOpenAIGatewayRouting(params, compat, cacheRetention !== "none");
applyOpenAIExtraBody(params, compat.extraBody, { applyOpenAIExtraBody(params, compat.extraBody, {
dropThinkingWhenReasoningEffort: compat.dropThinkingWhenReasoningEffort, dropThinkingWhenReasoningEffort: compat.dropThinkingWhenReasoningEffort,
@@ -71,6 +71,7 @@ import {
applyOpenAIExtraBody, applyOpenAIExtraBody,
applyOpenAIGatewayRouting, applyOpenAIGatewayRouting,
applyResponsesCompatPolicy, applyResponsesCompatPolicy,
applyVercelResponsesCacheControls,
applyWireModelIdTransform, applyWireModelIdTransform,
buildResponsesDeltaInput, buildResponsesDeltaInput,
buildResponsesInput, buildResponsesInput,
@@ -359,6 +360,9 @@ type OpenAIResponsesSamplingParams = ResponseCreateParamsStreaming & {
provider?: OpenAICompat["openRouterRouting"]; provider?: OpenAICompat["openRouterRouting"];
reasoning?: { effort?: string } | { enabled: false }; reasoning?: { effort?: string } | { enabled: false };
cache_control?: OpenRouterAnthropicCacheControl; cache_control?: OpenRouterAnthropicCacheControl;
caching?: "auto";
cache_anchor_items?: number;
cache_ttl?: "5m" | "1h";
}; };
function maybeAddOpenRouterAnthropicCacheControl( function maybeAddOpenRouterAnthropicCacheControl(
@@ -1024,7 +1028,11 @@ export function buildParams(
params.reasoning = { ...params.reasoning, mode: model.reasoningMode }; params.reasoning = { ...params.reasoning, mode: model.reasoningMode };
} }
applyOpenAIGatewayRouting(params, model.compat); if (model.compat.isVercelGatewayHost) {
applyVercelResponsesCacheControls(params, model.compat, cacheRetention !== "none");
} else {
applyOpenAIGatewayRouting(params, model.compat);
}
applyOpenAIExtraBody(params, options?.extraBody); applyOpenAIExtraBody(params, options?.extraBody);
+31 -2
View File
@@ -616,22 +616,51 @@ export interface OpenAIGatewayRoutingCompat {
export function applyOpenAIGatewayRouting( export function applyOpenAIGatewayRouting(
params: OpenAIGatewayRoutingParams, params: OpenAIGatewayRoutingParams,
compat: OpenAIGatewayRoutingCompat, compat: OpenAIGatewayRoutingCompat,
cacheEnabled = true,
): void { ): void {
if (compat.isOpenRouterHost && compat.openRouterRouting) { if (compat.isOpenRouterHost && compat.openRouterRouting) {
params.provider = compat.openRouterRouting; params.provider = compat.openRouterRouting;
} }
if (compat.isVercelGatewayHost && compat.vercelGatewayRouting) { if (compat.isVercelGatewayHost && compat.vercelGatewayRouting) {
const routing = compat.vercelGatewayRouting; const routing = compat.vercelGatewayRouting;
if (routing.only || routing.order || routing.caching) { if (routing.only || routing.order || (cacheEnabled && routing.caching)) {
const gatewayOptions: Pick<VercelGatewayRouting, "only" | "order" | "caching"> = {}; const gatewayOptions: Pick<VercelGatewayRouting, "only" | "order" | "caching"> = {};
if (routing.only) gatewayOptions.only = routing.only; if (routing.only) gatewayOptions.only = routing.only;
if (routing.order) gatewayOptions.order = routing.order; if (routing.order) gatewayOptions.order = routing.order;
if (routing.caching) gatewayOptions.caching = routing.caching; if (cacheEnabled && routing.caching) gatewayOptions.caching = routing.caching;
params.providerOptions = { gateway: gatewayOptions }; params.providerOptions = { gateway: gatewayOptions };
} }
} }
} }
export interface VercelResponsesCacheParams {
caching?: "auto";
cache_anchor_items?: number;
cache_ttl?: "5m" | "1h";
}
export interface VercelResponsesCacheCompat {
isVercelGatewayHost: boolean;
vercelGatewayRouting?: VercelGatewayRouting;
}
/**
* Apply Vercel AI Gateway's Responses-only automatic cache controls. Chat
* Completions uses the distinct `providerOptions.gateway` shape above.
*/
export function applyVercelResponsesCacheControls(
params: VercelResponsesCacheParams,
compat: VercelResponsesCacheCompat,
cacheEnabled = true,
): void {
const routing = compat.vercelGatewayRouting;
if (!cacheEnabled || !compat.isVercelGatewayHost || routing?.caching !== "auto") return;
params.caching = "auto";
if (routing.cacheAnchorItems !== undefined) params.cache_anchor_items = routing.cacheAnchorItems;
if (routing.cacheTtl !== undefined) params.cache_ttl = routing.cacheTtl;
}
export interface OpenAIExtraBodyOptions { export interface OpenAIExtraBodyOptions {
/** /**
* Fireworks rejects DeepSeek-style `thinking` toggles alongside OpenAI-style * Fireworks rejects DeepSeek-style `thinking` toggles alongside OpenAI-style
@@ -0,0 +1,175 @@
import { describe, expect, it } from "bun:test";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
import type { Context, Model, ModelSpec, VercelGatewayRouting } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { withEnv } from "./helpers";
const context: Context = {
messages: [{ role: "user", content: "Hello", timestamp: 0 }],
};
type Payload = Record<string, unknown>;
function abortedSignal(): AbortSignal {
const controller = new AbortController();
controller.abort();
return controller.signal;
}
function vercelChatModel(routing?: VercelGatewayRouting): Model<"openai-completions"> {
return buildModel({
id: "anthropic/claude-sonnet-4.6",
name: "Claude Sonnet 4.6",
api: "openai-completions",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 16_384,
...(routing ? { compat: { vercelGatewayRouting: routing } } : {}),
} satisfies ModelSpec<"openai-completions">);
}
function responsesModel(provider: string, baseUrl: string, routing?: VercelGatewayRouting): Model<"openai-responses"> {
return buildModel({
id: "anthropic/claude-sonnet-4.6",
name: "Claude Sonnet 4.6",
api: "openai-responses",
provider,
baseUrl,
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 16_384,
...(routing ? { compat: { vercelGatewayRouting: routing } } : {}),
} satisfies ModelSpec<"openai-responses">);
}
function captureChatPayload(
model: Model<"openai-completions">,
options: { cacheRetention?: "none" } = {},
): Promise<Payload> {
const { promise, resolve } = Promise.withResolvers<Payload>();
streamOpenAICompletions(model, context, {
apiKey: "test-key",
signal: abortedSignal(),
...options,
onPayload: payload => resolve(payload as Payload),
});
return promise;
}
function captureResponsesPayload(
model: Model<"openai-responses">,
options: { cacheRetention?: "none" } = {},
): Promise<Payload> {
const { promise, resolve } = Promise.withResolvers<Payload>();
streamOpenAIResponses(model, context, {
apiKey: "test-key",
signal: abortedSignal(),
...options,
onPayload: payload => resolve(payload as Payload),
});
return promise;
}
describe("Vercel AI Gateway automatic cache controls", () => {
it("maps cache fields to their documented Chat and Responses request shapes", async () => {
const routing: VercelGatewayRouting = {
only: ["anthropic"],
order: ["anthropic", "bedrock"],
caching: "auto",
cacheAnchorItems: 1,
cacheTtl: "1h",
};
const [chat, responses] = await Promise.all([
captureChatPayload(vercelChatModel(routing)),
captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1", routing)),
]);
expect(chat.providerOptions).toEqual({
gateway: { only: ["anthropic"], order: ["anthropic", "bedrock"], caching: "auto" },
});
expect(chat.caching).toBeUndefined();
expect(chat.cache_anchor_items).toBeUndefined();
expect(chat.cache_ttl).toBeUndefined();
expect(responses.caching).toBe("auto");
expect(responses.cache_anchor_items).toBe(1);
expect(responses.cache_ttl).toBe("1h");
expect(responses.providerOptions).toBeUndefined();
});
it("omits Chat and Responses automatic cache controls when cache retention is none", async () => {
const routing: VercelGatewayRouting = {
only: ["anthropic"],
order: ["anthropic", "bedrock"],
caching: "auto",
cacheAnchorItems: 1,
cacheTtl: "1h",
};
const [chat, responses] = await Promise.all([
captureChatPayload(vercelChatModel(routing), { cacheRetention: "none" }),
captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1", routing), {
cacheRetention: "none",
}),
]);
expect(chat.providerOptions).toEqual({ gateway: { only: ["anthropic"], order: ["anthropic", "bedrock"] } });
expect(chat.caching).toBeUndefined();
expect(chat.cache_anchor_items).toBeUndefined();
expect(chat.cache_ttl).toBeUndefined();
expect(responses.caching).toBeUndefined();
expect(responses.cache_anchor_items).toBeUndefined();
expect(responses.cache_ttl).toBeUndefined();
});
it("omits Chat and Responses automatic cache controls when PI_CACHE_RETENTION is none", async () => {
const routing: VercelGatewayRouting = {
only: ["anthropic"],
order: ["anthropic", "bedrock"],
caching: "auto",
cacheAnchorItems: 1,
cacheTtl: "1h",
};
await withEnv({ PI_CACHE_RETENTION: "none" }, async () => {
const [chat, responses] = await Promise.all([
captureChatPayload(vercelChatModel(routing)),
captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1", routing)),
]);
expect(chat.providerOptions).toEqual({ gateway: { only: ["anthropic"], order: ["anthropic", "bedrock"] } });
expect(chat.caching).toBeUndefined();
expect(chat.cache_anchor_items).toBeUndefined();
expect(chat.cache_ttl).toBeUndefined();
expect(responses.caching).toBeUndefined();
expect(responses.cache_anchor_items).toBeUndefined();
expect(responses.cache_ttl).toBeUndefined();
});
});
it("leaves unconfigured and non-Vercel Responses requests unchanged", async () => {
const routing: VercelGatewayRouting = {
caching: "auto",
cacheAnchorItems: 1,
cacheTtl: "1h",
};
const [unconfiguredVercel, nonVercel] = await Promise.all([
captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1")),
captureResponsesPayload(responsesModel("custom", "https://api.example.com/v1", routing)),
]);
for (const payload of [unconfiguredVercel, nonVercel]) {
expect(payload.caching).toBeUndefined();
expect(payload.cache_anchor_items).toBeUndefined();
expect(payload.cache_ttl).toBeUndefined();
expect(payload.providerOptions).toBeUndefined();
}
});
});
+1
View File
@@ -6,6 +6,7 @@
- Added the native Meta Model API provider and Muse Spark 1.1 with Responses API reasoning replay, image input, and the full supported reasoning-effort ladder ([#4941](https://github.com/can1357/oh-my-pi/issues/4941)). - Added the native Meta Model API provider and Muse Spark 1.1 with Responses API reasoning replay, image input, and the full supported reasoning-effort ladder ([#4941](https://github.com/can1357/oh-my-pi/issues/4941)).
- Added an opt-in Vercel AI Gateway automatic prompt-cache compatibility option alongside provider routing preferences. - Added an opt-in Vercel AI Gateway automatic prompt-cache compatibility option alongside provider routing preferences.
- Added Vercel AI Gateway Responses cache-anchor and cache-lifetime compatibility controls.
## [17.0.9] - 2026-07-23 ## [17.0.9] - 2026-07-23
+4
View File
@@ -623,6 +623,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
const isAzure = modelMatchesHost({ provider: spec.provider, baseUrl }, "azureOpenAI"); const isAzure = modelMatchesHost({ provider: spec.provider, baseUrl }, "azureOpenAI");
const isOpenRouter = modelMatchesHost({ provider: spec.provider, baseUrl }, "openrouter"); const isOpenRouter = modelMatchesHost({ provider: spec.provider, baseUrl }, "openrouter");
const isOpenAIUrl = hostMatchesUrl(baseUrl, "openai"); const isOpenAIUrl = hostMatchesUrl(baseUrl, "openai");
const isVercelGateway = modelMatchesHost({ provider: spec.provider, baseUrl }, "vercelAIGateway");
const id = spec.id ?? ""; const id = spec.id ?? "";
const thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"] = isOpenRouter ? "openrouter" : "openai"; const thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"] = isOpenRouter ? "openrouter" : "openai";
const isKimiModel = id ? isKimiModelId(id) : false; const isKimiModel = id ? isKimiModelId(id) : false;
@@ -686,7 +687,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
requiresAssistantAfterToolResult: false, requiresAssistantAfterToolResult: false,
requiresAssistantContentForToolCalls: isKimiModel, requiresAssistantContentForToolCalls: isKimiModel,
openRouterRouting: undefined, openRouterRouting: undefined,
vercelGatewayRouting: undefined,
isOpenRouterHost: isOpenRouter, isOpenRouterHost: isOpenRouter,
isVercelGatewayHost: isVercelGateway,
wireModelIdMode: isOpenRouter ? "openrouter" : "raw", wireModelIdMode: isOpenRouter ? "openrouter" : "raw",
// Mirrors buildOpenAICompat: Kimi behind a Responses-capable proxy still // Mirrors buildOpenAICompat: Kimi behind a Responses-capable proxy still
// lands on Moonshot's MFJS validator. // lands on Moonshot's MFJS validator.
@@ -724,6 +727,7 @@ function pickResponsesOnly(compat: ResolvedOpenAIResponsesCompat): ResponsesOnly
strictResponsesPairing: compat.strictResponsesPairing, strictResponsesPairing: compat.strictResponsesPairing,
supportsImageDetailOriginal: compat.supportsImageDetailOriginal, supportsImageDetailOriginal: compat.supportsImageDetailOriginal,
supportsObfuscationOptOut: compat.supportsObfuscationOptOut, supportsObfuscationOptOut: compat.supportsObfuscationOptOut,
isVercelGatewayHost: compat.isVercelGatewayHost,
} satisfies ResponsesOnlyCompat; } satisfies ResponsesOnlyCompat;
} }
+7
View File
@@ -483,6 +483,10 @@ export interface VercelGatewayRouting {
order?: string[]; order?: string[];
/** Enables Vercel AI Gateway's provider-aware automatic prompt caching. */ /** Enables Vercel AI Gateway's provider-aware automatic prompt caching. */
caching?: "auto"; caching?: "auto";
/** Stable Responses input-item prefix to anchor for automatic caching. */
cacheAnchorItems?: number;
/** Requested automatic-cache lifetime for the Responses API. */
cacheTtl?: "5m" | "1h";
} }
type ResolvedToolStrictMode = NonNullable<OpenAICompat["toolStrictMode"]> | "mixed"; type ResolvedToolStrictMode = NonNullable<OpenAICompat["toolStrictMode"]> | "mixed";
@@ -622,6 +626,9 @@ export interface ResolvedOpenAIResponsesCompat extends ResolvedOpenAISharedCompa
supportsImageDetailOriginal: boolean; supportsImageDetailOriginal: boolean;
supportsObfuscationOptOut: boolean; supportsObfuscationOptOut: boolean;
streamIdleTimeoutMs?: number; streamIdleTimeoutMs?: number;
vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
/** The model sits behind Vercel AI Gateway's Responses endpoint. */
isVercelGatewayHost: boolean;
} }
/** /**
@@ -1,7 +1,7 @@
import { describe, expect, test } from "bun:test"; import { describe, expect, test } from "bun:test";
import { buildModel } from "../src/build";
import { getBundledModelReferenceIndex } from "../src/identity/bundled"; import { getBundledModelReferenceIndex } from "../src/identity/bundled";
import { inheritReferenceThinking, resolveModelReference } from "../src/identity/reference"; import { inheritReferenceThinking, resolveModelReference } from "../src/identity/reference";
import { buildModel } from "../src/build";
import type { ModelSpec } from "../src/types"; import type { ModelSpec } from "../src/types";
describe("Portkey gateway model references", () => { describe("Portkey gateway model references", () => {
@@ -49,3 +49,37 @@ describe("Vercel AI Gateway cache compat", () => {
}); });
}); });
}); });
test("resolves Responses cache controls only for the Vercel endpoint", () => {
const routing = { caching: "auto" as const, cacheAnchorItems: 1, cacheTtl: "1h" as const };
const vercel = buildModel({
id: "anthropic/claude-sonnet-4.6",
name: "Claude Sonnet 4.6",
api: "openai-responses",
provider: "vercel-ai-gateway",
baseUrl: "https://ai-gateway.vercel.sh/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 16_384,
compat: { vercelGatewayRouting: routing },
} satisfies ModelSpec<"openai-responses">);
const direct = buildModel({
id: "anthropic/claude-sonnet-4.6",
name: "Claude Sonnet 4.6",
api: "openai-responses",
provider: "custom",
baseUrl: "https://api.example.com/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 16_384,
compat: { vercelGatewayRouting: routing },
} satisfies ModelSpec<"openai-responses">);
expect(vercel.compat.isVercelGatewayHost).toBe(true);
expect(vercel.compat.vercelGatewayRouting).toEqual(routing);
expect(direct.compat.isVercelGatewayHost).toBe(false);
});