feat(gateway): map Vercel Responses cache anchors
This commit is contained in:
@@ -6,6 +6,7 @@
|
||||
|
||||
- Added Anthropic extra-usage reporting across `omp usage`, interactive `/usage`, and ACP `/usage`: the OAuth usage endpoint's authoritative `spend` payload (or legacy `extra_usage` fallback when absent) is normalized into a `Claude Extra Usage` USD row; capped accounts show limit/remaining/fractions and status, while uncapped spend exposes only its absolute used amount—rendered as `$… used` in CLI/TUI and `123.45 usd used` in ACP—without a fabricated cap, percentage, or status. ([#5575](https://github.com/can1357/oh-my-pi/issues/5575))
|
||||
- Added opt-in Vercel AI Gateway automatic prompt caching for OpenAI Chat Completions while preserving `only` and `order` routing preferences.
|
||||
- Added Vercel AI Gateway Responses cache anchors and cache lifetimes, emitted only with automatic caching.
|
||||
|
||||
### Fixed
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ import type {
|
||||
ToolChoice,
|
||||
ToolResultMessage,
|
||||
} from "../types";
|
||||
import { normalizeSystemPrompts } from "../utils";
|
||||
import { normalizeSystemPrompts, resolveCacheRetention } from "../utils";
|
||||
import { createAbortSourceTracker } from "../utils/abort";
|
||||
import { isDemotedThinking, kStreamingLastParseLen } from "../utils/block-symbols";
|
||||
import { hasVisibleAssistantContent, withEmptyCompletionRetry } from "../utils/empty-completion-retry";
|
||||
@@ -1463,6 +1463,7 @@ function buildParams(
|
||||
} {
|
||||
const initialPolicy = resolveOpenAICompatForRequest(model, options);
|
||||
const initialCompat = initialPolicy.compat as ResolvedOpenAICompat;
|
||||
const cacheRetention = resolveCacheRetention(options?.cacheRetention);
|
||||
|
||||
const requestModelId = resolveOpenAICompletionsModelId(model, options);
|
||||
const params: OpenAICompletionsParams = {
|
||||
@@ -1615,7 +1616,7 @@ function buildParams(
|
||||
applyChatCompletionsCompatPolicy(params, finalPolicy);
|
||||
dropOpenRouterKimiForcedToolReasoning(params, model, finalPolicy);
|
||||
|
||||
applyOpenAIGatewayRouting(params, compat);
|
||||
applyOpenAIGatewayRouting(params, compat, cacheRetention !== "none");
|
||||
|
||||
applyOpenAIExtraBody(params, compat.extraBody, {
|
||||
dropThinkingWhenReasoningEffort: compat.dropThinkingWhenReasoningEffort,
|
||||
|
||||
@@ -71,6 +71,7 @@ import {
|
||||
applyOpenAIExtraBody,
|
||||
applyOpenAIGatewayRouting,
|
||||
applyResponsesCompatPolicy,
|
||||
applyVercelResponsesCacheControls,
|
||||
applyWireModelIdTransform,
|
||||
buildResponsesDeltaInput,
|
||||
buildResponsesInput,
|
||||
@@ -359,6 +360,9 @@ type OpenAIResponsesSamplingParams = ResponseCreateParamsStreaming & {
|
||||
provider?: OpenAICompat["openRouterRouting"];
|
||||
reasoning?: { effort?: string } | { enabled: false };
|
||||
cache_control?: OpenRouterAnthropicCacheControl;
|
||||
caching?: "auto";
|
||||
cache_anchor_items?: number;
|
||||
cache_ttl?: "5m" | "1h";
|
||||
};
|
||||
|
||||
function maybeAddOpenRouterAnthropicCacheControl(
|
||||
@@ -1024,7 +1028,11 @@ export function buildParams(
|
||||
params.reasoning = { ...params.reasoning, mode: model.reasoningMode };
|
||||
}
|
||||
|
||||
applyOpenAIGatewayRouting(params, model.compat);
|
||||
if (model.compat.isVercelGatewayHost) {
|
||||
applyVercelResponsesCacheControls(params, model.compat, cacheRetention !== "none");
|
||||
} else {
|
||||
applyOpenAIGatewayRouting(params, model.compat);
|
||||
}
|
||||
|
||||
applyOpenAIExtraBody(params, options?.extraBody);
|
||||
|
||||
|
||||
@@ -616,22 +616,51 @@ export interface OpenAIGatewayRoutingCompat {
|
||||
export function applyOpenAIGatewayRouting(
|
||||
params: OpenAIGatewayRoutingParams,
|
||||
compat: OpenAIGatewayRoutingCompat,
|
||||
cacheEnabled = true,
|
||||
): void {
|
||||
if (compat.isOpenRouterHost && compat.openRouterRouting) {
|
||||
params.provider = compat.openRouterRouting;
|
||||
}
|
||||
if (compat.isVercelGatewayHost && compat.vercelGatewayRouting) {
|
||||
const routing = compat.vercelGatewayRouting;
|
||||
if (routing.only || routing.order || routing.caching) {
|
||||
if (routing.only || routing.order || (cacheEnabled && routing.caching)) {
|
||||
const gatewayOptions: Pick<VercelGatewayRouting, "only" | "order" | "caching"> = {};
|
||||
if (routing.only) gatewayOptions.only = routing.only;
|
||||
if (routing.order) gatewayOptions.order = routing.order;
|
||||
if (routing.caching) gatewayOptions.caching = routing.caching;
|
||||
if (cacheEnabled && routing.caching) gatewayOptions.caching = routing.caching;
|
||||
params.providerOptions = { gateway: gatewayOptions };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export interface VercelResponsesCacheParams {
|
||||
caching?: "auto";
|
||||
cache_anchor_items?: number;
|
||||
cache_ttl?: "5m" | "1h";
|
||||
}
|
||||
|
||||
export interface VercelResponsesCacheCompat {
|
||||
isVercelGatewayHost: boolean;
|
||||
vercelGatewayRouting?: VercelGatewayRouting;
|
||||
}
|
||||
|
||||
/**
|
||||
* Apply Vercel AI Gateway's Responses-only automatic cache controls. Chat
|
||||
* Completions uses the distinct `providerOptions.gateway` shape above.
|
||||
*/
|
||||
export function applyVercelResponsesCacheControls(
|
||||
params: VercelResponsesCacheParams,
|
||||
compat: VercelResponsesCacheCompat,
|
||||
cacheEnabled = true,
|
||||
): void {
|
||||
const routing = compat.vercelGatewayRouting;
|
||||
if (!cacheEnabled || !compat.isVercelGatewayHost || routing?.caching !== "auto") return;
|
||||
|
||||
params.caching = "auto";
|
||||
if (routing.cacheAnchorItems !== undefined) params.cache_anchor_items = routing.cacheAnchorItems;
|
||||
if (routing.cacheTtl !== undefined) params.cache_ttl = routing.cacheTtl;
|
||||
}
|
||||
|
||||
export interface OpenAIExtraBodyOptions {
|
||||
/**
|
||||
* Fireworks rejects DeepSeek-style `thinking` toggles alongside OpenAI-style
|
||||
|
||||
@@ -0,0 +1,175 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
||||
import type { Context, Model, ModelSpec, VercelGatewayRouting } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { withEnv } from "./helpers";
|
||||
|
||||
const context: Context = {
|
||||
messages: [{ role: "user", content: "Hello", timestamp: 0 }],
|
||||
};
|
||||
|
||||
type Payload = Record<string, unknown>;
|
||||
|
||||
function abortedSignal(): AbortSignal {
|
||||
const controller = new AbortController();
|
||||
controller.abort();
|
||||
return controller.signal;
|
||||
}
|
||||
|
||||
function vercelChatModel(routing?: VercelGatewayRouting): Model<"openai-completions"> {
|
||||
return buildModel({
|
||||
id: "anthropic/claude-sonnet-4.6",
|
||||
name: "Claude Sonnet 4.6",
|
||||
api: "openai-completions",
|
||||
provider: "vercel-ai-gateway",
|
||||
baseUrl: "https://ai-gateway.vercel.sh/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 16_384,
|
||||
...(routing ? { compat: { vercelGatewayRouting: routing } } : {}),
|
||||
} satisfies ModelSpec<"openai-completions">);
|
||||
}
|
||||
|
||||
function responsesModel(provider: string, baseUrl: string, routing?: VercelGatewayRouting): Model<"openai-responses"> {
|
||||
return buildModel({
|
||||
id: "anthropic/claude-sonnet-4.6",
|
||||
name: "Claude Sonnet 4.6",
|
||||
api: "openai-responses",
|
||||
provider,
|
||||
baseUrl,
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 16_384,
|
||||
...(routing ? { compat: { vercelGatewayRouting: routing } } : {}),
|
||||
} satisfies ModelSpec<"openai-responses">);
|
||||
}
|
||||
|
||||
function captureChatPayload(
|
||||
model: Model<"openai-completions">,
|
||||
options: { cacheRetention?: "none" } = {},
|
||||
): Promise<Payload> {
|
||||
const { promise, resolve } = Promise.withResolvers<Payload>();
|
||||
streamOpenAICompletions(model, context, {
|
||||
apiKey: "test-key",
|
||||
signal: abortedSignal(),
|
||||
...options,
|
||||
onPayload: payload => resolve(payload as Payload),
|
||||
});
|
||||
return promise;
|
||||
}
|
||||
|
||||
function captureResponsesPayload(
|
||||
model: Model<"openai-responses">,
|
||||
options: { cacheRetention?: "none" } = {},
|
||||
): Promise<Payload> {
|
||||
const { promise, resolve } = Promise.withResolvers<Payload>();
|
||||
streamOpenAIResponses(model, context, {
|
||||
apiKey: "test-key",
|
||||
signal: abortedSignal(),
|
||||
...options,
|
||||
onPayload: payload => resolve(payload as Payload),
|
||||
});
|
||||
return promise;
|
||||
}
|
||||
|
||||
describe("Vercel AI Gateway automatic cache controls", () => {
|
||||
it("maps cache fields to their documented Chat and Responses request shapes", async () => {
|
||||
const routing: VercelGatewayRouting = {
|
||||
only: ["anthropic"],
|
||||
order: ["anthropic", "bedrock"],
|
||||
caching: "auto",
|
||||
cacheAnchorItems: 1,
|
||||
cacheTtl: "1h",
|
||||
};
|
||||
const [chat, responses] = await Promise.all([
|
||||
captureChatPayload(vercelChatModel(routing)),
|
||||
captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1", routing)),
|
||||
]);
|
||||
|
||||
expect(chat.providerOptions).toEqual({
|
||||
gateway: { only: ["anthropic"], order: ["anthropic", "bedrock"], caching: "auto" },
|
||||
});
|
||||
expect(chat.caching).toBeUndefined();
|
||||
expect(chat.cache_anchor_items).toBeUndefined();
|
||||
expect(chat.cache_ttl).toBeUndefined();
|
||||
|
||||
expect(responses.caching).toBe("auto");
|
||||
expect(responses.cache_anchor_items).toBe(1);
|
||||
expect(responses.cache_ttl).toBe("1h");
|
||||
expect(responses.providerOptions).toBeUndefined();
|
||||
});
|
||||
|
||||
it("omits Chat and Responses automatic cache controls when cache retention is none", async () => {
|
||||
const routing: VercelGatewayRouting = {
|
||||
only: ["anthropic"],
|
||||
order: ["anthropic", "bedrock"],
|
||||
caching: "auto",
|
||||
cacheAnchorItems: 1,
|
||||
cacheTtl: "1h",
|
||||
};
|
||||
const [chat, responses] = await Promise.all([
|
||||
captureChatPayload(vercelChatModel(routing), { cacheRetention: "none" }),
|
||||
captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1", routing), {
|
||||
cacheRetention: "none",
|
||||
}),
|
||||
]);
|
||||
|
||||
expect(chat.providerOptions).toEqual({ gateway: { only: ["anthropic"], order: ["anthropic", "bedrock"] } });
|
||||
expect(chat.caching).toBeUndefined();
|
||||
expect(chat.cache_anchor_items).toBeUndefined();
|
||||
expect(chat.cache_ttl).toBeUndefined();
|
||||
|
||||
expect(responses.caching).toBeUndefined();
|
||||
expect(responses.cache_anchor_items).toBeUndefined();
|
||||
expect(responses.cache_ttl).toBeUndefined();
|
||||
});
|
||||
|
||||
it("omits Chat and Responses automatic cache controls when PI_CACHE_RETENTION is none", async () => {
|
||||
const routing: VercelGatewayRouting = {
|
||||
only: ["anthropic"],
|
||||
order: ["anthropic", "bedrock"],
|
||||
caching: "auto",
|
||||
cacheAnchorItems: 1,
|
||||
cacheTtl: "1h",
|
||||
};
|
||||
await withEnv({ PI_CACHE_RETENTION: "none" }, async () => {
|
||||
const [chat, responses] = await Promise.all([
|
||||
captureChatPayload(vercelChatModel(routing)),
|
||||
captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1", routing)),
|
||||
]);
|
||||
|
||||
expect(chat.providerOptions).toEqual({ gateway: { only: ["anthropic"], order: ["anthropic", "bedrock"] } });
|
||||
expect(chat.caching).toBeUndefined();
|
||||
expect(chat.cache_anchor_items).toBeUndefined();
|
||||
expect(chat.cache_ttl).toBeUndefined();
|
||||
|
||||
expect(responses.caching).toBeUndefined();
|
||||
expect(responses.cache_anchor_items).toBeUndefined();
|
||||
expect(responses.cache_ttl).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
it("leaves unconfigured and non-Vercel Responses requests unchanged", async () => {
|
||||
const routing: VercelGatewayRouting = {
|
||||
caching: "auto",
|
||||
cacheAnchorItems: 1,
|
||||
cacheTtl: "1h",
|
||||
};
|
||||
const [unconfiguredVercel, nonVercel] = await Promise.all([
|
||||
captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1")),
|
||||
captureResponsesPayload(responsesModel("custom", "https://api.example.com/v1", routing)),
|
||||
]);
|
||||
|
||||
for (const payload of [unconfiguredVercel, nonVercel]) {
|
||||
expect(payload.caching).toBeUndefined();
|
||||
expect(payload.cache_anchor_items).toBeUndefined();
|
||||
expect(payload.cache_ttl).toBeUndefined();
|
||||
expect(payload.providerOptions).toBeUndefined();
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -6,6 +6,7 @@
|
||||
|
||||
- Added the native Meta Model API provider and Muse Spark 1.1 with Responses API reasoning replay, image input, and the full supported reasoning-effort ladder ([#4941](https://github.com/can1357/oh-my-pi/issues/4941)).
|
||||
- Added an opt-in Vercel AI Gateway automatic prompt-cache compatibility option alongside provider routing preferences.
|
||||
- Added Vercel AI Gateway Responses cache-anchor and cache-lifetime compatibility controls.
|
||||
|
||||
## [17.0.9] - 2026-07-23
|
||||
|
||||
|
||||
@@ -623,6 +623,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
|
||||
const isAzure = modelMatchesHost({ provider: spec.provider, baseUrl }, "azureOpenAI");
|
||||
const isOpenRouter = modelMatchesHost({ provider: spec.provider, baseUrl }, "openrouter");
|
||||
const isOpenAIUrl = hostMatchesUrl(baseUrl, "openai");
|
||||
const isVercelGateway = modelMatchesHost({ provider: spec.provider, baseUrl }, "vercelAIGateway");
|
||||
const id = spec.id ?? "";
|
||||
const thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"] = isOpenRouter ? "openrouter" : "openai";
|
||||
const isKimiModel = id ? isKimiModelId(id) : false;
|
||||
@@ -686,7 +687,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
|
||||
requiresAssistantAfterToolResult: false,
|
||||
requiresAssistantContentForToolCalls: isKimiModel,
|
||||
openRouterRouting: undefined,
|
||||
vercelGatewayRouting: undefined,
|
||||
isOpenRouterHost: isOpenRouter,
|
||||
isVercelGatewayHost: isVercelGateway,
|
||||
wireModelIdMode: isOpenRouter ? "openrouter" : "raw",
|
||||
// Mirrors buildOpenAICompat: Kimi behind a Responses-capable proxy still
|
||||
// lands on Moonshot's MFJS validator.
|
||||
@@ -724,6 +727,7 @@ function pickResponsesOnly(compat: ResolvedOpenAIResponsesCompat): ResponsesOnly
|
||||
strictResponsesPairing: compat.strictResponsesPairing,
|
||||
supportsImageDetailOriginal: compat.supportsImageDetailOriginal,
|
||||
supportsObfuscationOptOut: compat.supportsObfuscationOptOut,
|
||||
isVercelGatewayHost: compat.isVercelGatewayHost,
|
||||
} satisfies ResponsesOnlyCompat;
|
||||
}
|
||||
|
||||
|
||||
@@ -483,6 +483,10 @@ export interface VercelGatewayRouting {
|
||||
order?: string[];
|
||||
/** Enables Vercel AI Gateway's provider-aware automatic prompt caching. */
|
||||
caching?: "auto";
|
||||
/** Stable Responses input-item prefix to anchor for automatic caching. */
|
||||
cacheAnchorItems?: number;
|
||||
/** Requested automatic-cache lifetime for the Responses API. */
|
||||
cacheTtl?: "5m" | "1h";
|
||||
}
|
||||
|
||||
type ResolvedToolStrictMode = NonNullable<OpenAICompat["toolStrictMode"]> | "mixed";
|
||||
@@ -622,6 +626,9 @@ export interface ResolvedOpenAIResponsesCompat extends ResolvedOpenAISharedCompa
|
||||
supportsImageDetailOriginal: boolean;
|
||||
supportsObfuscationOptOut: boolean;
|
||||
streamIdleTimeoutMs?: number;
|
||||
vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
|
||||
/** The model sits behind Vercel AI Gateway's Responses endpoint. */
|
||||
isVercelGatewayHost: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { buildModel } from "../src/build";
|
||||
import { getBundledModelReferenceIndex } from "../src/identity/bundled";
|
||||
import { inheritReferenceThinking, resolveModelReference } from "../src/identity/reference";
|
||||
import { buildModel } from "../src/build";
|
||||
import type { ModelSpec } from "../src/types";
|
||||
|
||||
describe("Portkey gateway model references", () => {
|
||||
@@ -49,3 +49,37 @@ describe("Vercel AI Gateway cache compat", () => {
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
test("resolves Responses cache controls only for the Vercel endpoint", () => {
|
||||
const routing = { caching: "auto" as const, cacheAnchorItems: 1, cacheTtl: "1h" as const };
|
||||
const vercel = buildModel({
|
||||
id: "anthropic/claude-sonnet-4.6",
|
||||
name: "Claude Sonnet 4.6",
|
||||
api: "openai-responses",
|
||||
provider: "vercel-ai-gateway",
|
||||
baseUrl: "https://ai-gateway.vercel.sh/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 16_384,
|
||||
compat: { vercelGatewayRouting: routing },
|
||||
} satisfies ModelSpec<"openai-responses">);
|
||||
const direct = buildModel({
|
||||
id: "anthropic/claude-sonnet-4.6",
|
||||
name: "Claude Sonnet 4.6",
|
||||
api: "openai-responses",
|
||||
provider: "custom",
|
||||
baseUrl: "https://api.example.com/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 16_384,
|
||||
compat: { vercelGatewayRouting: routing },
|
||||
} satisfies ModelSpec<"openai-responses">);
|
||||
|
||||
expect(vercel.compat.isVercelGatewayHost).toBe(true);
|
||||
expect(vercel.compat.vercelGatewayRouting).toEqual(routing);
|
||||
expect(direct.compat.isVercelGatewayHost).toBe(false);
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user