diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 9cd563c65..4ce3853f1 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -6,6 +6,7 @@ - Added Anthropic extra-usage reporting across `omp usage`, interactive `/usage`, and ACP `/usage`: the OAuth usage endpoint's authoritative `spend` payload (or legacy `extra_usage` fallback when absent) is normalized into a `Claude Extra Usage` USD row; capped accounts show limit/remaining/fractions and status, while uncapped spend exposes only its absolute used amount—rendered as `$… used` in CLI/TUI and `123.45 usd used` in ACP—without a fabricated cap, percentage, or status. ([#5575](https://github.com/can1357/oh-my-pi/issues/5575)) - Added opt-in Vercel AI Gateway automatic prompt caching for OpenAI Chat Completions while preserving `only` and `order` routing preferences. +- Added Vercel AI Gateway Responses cache anchors and cache lifetimes, emitted only with automatic caching. ### Fixed diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index df46e01f2..01b172537 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -27,7 +27,7 @@ import type { ToolChoice, ToolResultMessage, } from "../types"; -import { normalizeSystemPrompts } from "../utils"; +import { normalizeSystemPrompts, resolveCacheRetention } from "../utils"; import { createAbortSourceTracker } from "../utils/abort"; import { isDemotedThinking, kStreamingLastParseLen } from "../utils/block-symbols"; import { hasVisibleAssistantContent, withEmptyCompletionRetry } from "../utils/empty-completion-retry"; @@ -1463,6 +1463,7 @@ function buildParams( } { const initialPolicy = resolveOpenAICompatForRequest(model, options); const initialCompat = initialPolicy.compat as ResolvedOpenAICompat; + const cacheRetention = resolveCacheRetention(options?.cacheRetention); const requestModelId = resolveOpenAICompletionsModelId(model, options); const params: OpenAICompletionsParams = { @@ -1615,7 +1616,7 @@ function buildParams( applyChatCompletionsCompatPolicy(params, finalPolicy); dropOpenRouterKimiForcedToolReasoning(params, model, finalPolicy); - applyOpenAIGatewayRouting(params, compat); + applyOpenAIGatewayRouting(params, compat, cacheRetention !== "none"); applyOpenAIExtraBody(params, compat.extraBody, { dropThinkingWhenReasoningEffort: compat.dropThinkingWhenReasoningEffort, diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 685eef7f3..baf6fe21a 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -71,6 +71,7 @@ import { applyOpenAIExtraBody, applyOpenAIGatewayRouting, applyResponsesCompatPolicy, + applyVercelResponsesCacheControls, applyWireModelIdTransform, buildResponsesDeltaInput, buildResponsesInput, @@ -359,6 +360,9 @@ type OpenAIResponsesSamplingParams = ResponseCreateParamsStreaming & { provider?: OpenAICompat["openRouterRouting"]; reasoning?: { effort?: string } | { enabled: false }; cache_control?: OpenRouterAnthropicCacheControl; + caching?: "auto"; + cache_anchor_items?: number; + cache_ttl?: "5m" | "1h"; }; function maybeAddOpenRouterAnthropicCacheControl( @@ -1024,7 +1028,11 @@ export function buildParams( params.reasoning = { ...params.reasoning, mode: model.reasoningMode }; } - applyOpenAIGatewayRouting(params, model.compat); + if (model.compat.isVercelGatewayHost) { + applyVercelResponsesCacheControls(params, model.compat, cacheRetention !== "none"); + } else { + applyOpenAIGatewayRouting(params, model.compat); + } applyOpenAIExtraBody(params, options?.extraBody); diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 635b4119e..9cc12e899 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -616,22 +616,51 @@ export interface OpenAIGatewayRoutingCompat { export function applyOpenAIGatewayRouting( params: OpenAIGatewayRoutingParams, compat: OpenAIGatewayRoutingCompat, + cacheEnabled = true, ): void { if (compat.isOpenRouterHost && compat.openRouterRouting) { params.provider = compat.openRouterRouting; } if (compat.isVercelGatewayHost && compat.vercelGatewayRouting) { const routing = compat.vercelGatewayRouting; - if (routing.only || routing.order || routing.caching) { + if (routing.only || routing.order || (cacheEnabled && routing.caching)) { const gatewayOptions: Pick = {}; if (routing.only) gatewayOptions.only = routing.only; if (routing.order) gatewayOptions.order = routing.order; - if (routing.caching) gatewayOptions.caching = routing.caching; + if (cacheEnabled && routing.caching) gatewayOptions.caching = routing.caching; params.providerOptions = { gateway: gatewayOptions }; } } } +export interface VercelResponsesCacheParams { + caching?: "auto"; + cache_anchor_items?: number; + cache_ttl?: "5m" | "1h"; +} + +export interface VercelResponsesCacheCompat { + isVercelGatewayHost: boolean; + vercelGatewayRouting?: VercelGatewayRouting; +} + +/** + * Apply Vercel AI Gateway's Responses-only automatic cache controls. Chat + * Completions uses the distinct `providerOptions.gateway` shape above. + */ +export function applyVercelResponsesCacheControls( + params: VercelResponsesCacheParams, + compat: VercelResponsesCacheCompat, + cacheEnabled = true, +): void { + const routing = compat.vercelGatewayRouting; + if (!cacheEnabled || !compat.isVercelGatewayHost || routing?.caching !== "auto") return; + + params.caching = "auto"; + if (routing.cacheAnchorItems !== undefined) params.cache_anchor_items = routing.cacheAnchorItems; + if (routing.cacheTtl !== undefined) params.cache_ttl = routing.cacheTtl; +} + export interface OpenAIExtraBodyOptions { /** * Fireworks rejects DeepSeek-style `thinking` toggles alongside OpenAI-style diff --git a/packages/ai/test/vercel-gateway-cache-controls.test.ts b/packages/ai/test/vercel-gateway-cache-controls.test.ts new file mode 100644 index 000000000..7126d1308 --- /dev/null +++ b/packages/ai/test/vercel-gateway-cache-controls.test.ts @@ -0,0 +1,175 @@ +import { describe, expect, it } from "bun:test"; +import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; +import type { Context, Model, ModelSpec, VercelGatewayRouting } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { withEnv } from "./helpers"; + +const context: Context = { + messages: [{ role: "user", content: "Hello", timestamp: 0 }], +}; + +type Payload = Record; + +function abortedSignal(): AbortSignal { + const controller = new AbortController(); + controller.abort(); + return controller.signal; +} + +function vercelChatModel(routing?: VercelGatewayRouting): Model<"openai-completions"> { + return buildModel({ + id: "anthropic/claude-sonnet-4.6", + name: "Claude Sonnet 4.6", + api: "openai-completions", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh/v1", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 16_384, + ...(routing ? { compat: { vercelGatewayRouting: routing } } : {}), + } satisfies ModelSpec<"openai-completions">); +} + +function responsesModel(provider: string, baseUrl: string, routing?: VercelGatewayRouting): Model<"openai-responses"> { + return buildModel({ + id: "anthropic/claude-sonnet-4.6", + name: "Claude Sonnet 4.6", + api: "openai-responses", + provider, + baseUrl, + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 16_384, + ...(routing ? { compat: { vercelGatewayRouting: routing } } : {}), + } satisfies ModelSpec<"openai-responses">); +} + +function captureChatPayload( + model: Model<"openai-completions">, + options: { cacheRetention?: "none" } = {}, +): Promise { + const { promise, resolve } = Promise.withResolvers(); + streamOpenAICompletions(model, context, { + apiKey: "test-key", + signal: abortedSignal(), + ...options, + onPayload: payload => resolve(payload as Payload), + }); + return promise; +} + +function captureResponsesPayload( + model: Model<"openai-responses">, + options: { cacheRetention?: "none" } = {}, +): Promise { + const { promise, resolve } = Promise.withResolvers(); + streamOpenAIResponses(model, context, { + apiKey: "test-key", + signal: abortedSignal(), + ...options, + onPayload: payload => resolve(payload as Payload), + }); + return promise; +} + +describe("Vercel AI Gateway automatic cache controls", () => { + it("maps cache fields to their documented Chat and Responses request shapes", async () => { + const routing: VercelGatewayRouting = { + only: ["anthropic"], + order: ["anthropic", "bedrock"], + caching: "auto", + cacheAnchorItems: 1, + cacheTtl: "1h", + }; + const [chat, responses] = await Promise.all([ + captureChatPayload(vercelChatModel(routing)), + captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1", routing)), + ]); + + expect(chat.providerOptions).toEqual({ + gateway: { only: ["anthropic"], order: ["anthropic", "bedrock"], caching: "auto" }, + }); + expect(chat.caching).toBeUndefined(); + expect(chat.cache_anchor_items).toBeUndefined(); + expect(chat.cache_ttl).toBeUndefined(); + + expect(responses.caching).toBe("auto"); + expect(responses.cache_anchor_items).toBe(1); + expect(responses.cache_ttl).toBe("1h"); + expect(responses.providerOptions).toBeUndefined(); + }); + + it("omits Chat and Responses automatic cache controls when cache retention is none", async () => { + const routing: VercelGatewayRouting = { + only: ["anthropic"], + order: ["anthropic", "bedrock"], + caching: "auto", + cacheAnchorItems: 1, + cacheTtl: "1h", + }; + const [chat, responses] = await Promise.all([ + captureChatPayload(vercelChatModel(routing), { cacheRetention: "none" }), + captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1", routing), { + cacheRetention: "none", + }), + ]); + + expect(chat.providerOptions).toEqual({ gateway: { only: ["anthropic"], order: ["anthropic", "bedrock"] } }); + expect(chat.caching).toBeUndefined(); + expect(chat.cache_anchor_items).toBeUndefined(); + expect(chat.cache_ttl).toBeUndefined(); + + expect(responses.caching).toBeUndefined(); + expect(responses.cache_anchor_items).toBeUndefined(); + expect(responses.cache_ttl).toBeUndefined(); + }); + + it("omits Chat and Responses automatic cache controls when PI_CACHE_RETENTION is none", async () => { + const routing: VercelGatewayRouting = { + only: ["anthropic"], + order: ["anthropic", "bedrock"], + caching: "auto", + cacheAnchorItems: 1, + cacheTtl: "1h", + }; + await withEnv({ PI_CACHE_RETENTION: "none" }, async () => { + const [chat, responses] = await Promise.all([ + captureChatPayload(vercelChatModel(routing)), + captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1", routing)), + ]); + + expect(chat.providerOptions).toEqual({ gateway: { only: ["anthropic"], order: ["anthropic", "bedrock"] } }); + expect(chat.caching).toBeUndefined(); + expect(chat.cache_anchor_items).toBeUndefined(); + expect(chat.cache_ttl).toBeUndefined(); + + expect(responses.caching).toBeUndefined(); + expect(responses.cache_anchor_items).toBeUndefined(); + expect(responses.cache_ttl).toBeUndefined(); + }); + }); + + it("leaves unconfigured and non-Vercel Responses requests unchanged", async () => { + const routing: VercelGatewayRouting = { + caching: "auto", + cacheAnchorItems: 1, + cacheTtl: "1h", + }; + const [unconfiguredVercel, nonVercel] = await Promise.all([ + captureResponsesPayload(responsesModel("vercel-ai-gateway", "https://ai-gateway.vercel.sh/v1")), + captureResponsesPayload(responsesModel("custom", "https://api.example.com/v1", routing)), + ]); + + for (const payload of [unconfiguredVercel, nonVercel]) { + expect(payload.caching).toBeUndefined(); + expect(payload.cache_anchor_items).toBeUndefined(); + expect(payload.cache_ttl).toBeUndefined(); + expect(payload.providerOptions).toBeUndefined(); + } + }); +}); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index aa212e397..36bf3d765 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -6,6 +6,7 @@ - Added the native Meta Model API provider and Muse Spark 1.1 with Responses API reasoning replay, image input, and the full supported reasoning-effort ladder ([#4941](https://github.com/can1357/oh-my-pi/issues/4941)). - Added an opt-in Vercel AI Gateway automatic prompt-cache compatibility option alongside provider routing preferences. +- Added Vercel AI Gateway Responses cache-anchor and cache-lifetime compatibility controls. ## [17.0.9] - 2026-07-23 diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index 7d21a3ead..c60de5662 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -623,6 +623,7 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol const isAzure = modelMatchesHost({ provider: spec.provider, baseUrl }, "azureOpenAI"); const isOpenRouter = modelMatchesHost({ provider: spec.provider, baseUrl }, "openrouter"); const isOpenAIUrl = hostMatchesUrl(baseUrl, "openai"); + const isVercelGateway = modelMatchesHost({ provider: spec.provider, baseUrl }, "vercelAIGateway"); const id = spec.id ?? ""; const thinkingFormat: ResolvedOpenAISharedCompat["thinkingFormat"] = isOpenRouter ? "openrouter" : "openai"; const isKimiModel = id ? isKimiModelId(id) : false; @@ -686,7 +687,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol requiresAssistantAfterToolResult: false, requiresAssistantContentForToolCalls: isKimiModel, openRouterRouting: undefined, + vercelGatewayRouting: undefined, isOpenRouterHost: isOpenRouter, + isVercelGatewayHost: isVercelGateway, wireModelIdMode: isOpenRouter ? "openrouter" : "raw", // Mirrors buildOpenAICompat: Kimi behind a Responses-capable proxy still // lands on Moonshot's MFJS validator. @@ -724,6 +727,7 @@ function pickResponsesOnly(compat: ResolvedOpenAIResponsesCompat): ResponsesOnly strictResponsesPairing: compat.strictResponsesPairing, supportsImageDetailOriginal: compat.supportsImageDetailOriginal, supportsObfuscationOptOut: compat.supportsObfuscationOptOut, + isVercelGatewayHost: compat.isVercelGatewayHost, } satisfies ResponsesOnlyCompat; } diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index 536138cd2..15d4c9dea 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -483,6 +483,10 @@ export interface VercelGatewayRouting { order?: string[]; /** Enables Vercel AI Gateway's provider-aware automatic prompt caching. */ caching?: "auto"; + /** Stable Responses input-item prefix to anchor for automatic caching. */ + cacheAnchorItems?: number; + /** Requested automatic-cache lifetime for the Responses API. */ + cacheTtl?: "5m" | "1h"; } type ResolvedToolStrictMode = NonNullable | "mixed"; @@ -622,6 +626,9 @@ export interface ResolvedOpenAIResponsesCompat extends ResolvedOpenAISharedCompa supportsImageDetailOriginal: boolean; supportsObfuscationOptOut: boolean; streamIdleTimeoutMs?: number; + vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"]; + /** The model sits behind Vercel AI Gateway's Responses endpoint. */ + isVercelGatewayHost: boolean; } /** diff --git a/packages/catalog/test/gateway-reference.test.ts b/packages/catalog/test/gateway-reference.test.ts index 1f3c5de5e..212c811a6 100644 --- a/packages/catalog/test/gateway-reference.test.ts +++ b/packages/catalog/test/gateway-reference.test.ts @@ -1,7 +1,7 @@ import { describe, expect, test } from "bun:test"; +import { buildModel } from "../src/build"; import { getBundledModelReferenceIndex } from "../src/identity/bundled"; import { inheritReferenceThinking, resolveModelReference } from "../src/identity/reference"; -import { buildModel } from "../src/build"; import type { ModelSpec } from "../src/types"; describe("Portkey gateway model references", () => { @@ -49,3 +49,37 @@ describe("Vercel AI Gateway cache compat", () => { }); }); }); + +test("resolves Responses cache controls only for the Vercel endpoint", () => { + const routing = { caching: "auto" as const, cacheAnchorItems: 1, cacheTtl: "1h" as const }; + const vercel = buildModel({ + id: "anthropic/claude-sonnet-4.6", + name: "Claude Sonnet 4.6", + api: "openai-responses", + provider: "vercel-ai-gateway", + baseUrl: "https://ai-gateway.vercel.sh/v1", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 16_384, + compat: { vercelGatewayRouting: routing }, + } satisfies ModelSpec<"openai-responses">); + const direct = buildModel({ + id: "anthropic/claude-sonnet-4.6", + name: "Claude Sonnet 4.6", + api: "openai-responses", + provider: "custom", + baseUrl: "https://api.example.com/v1", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 16_384, + compat: { vercelGatewayRouting: routing }, + } satisfies ModelSpec<"openai-responses">); + + expect(vercel.compat.isVercelGatewayHost).toBe(true); + expect(vercel.compat.vercelGatewayRouting).toEqual(routing); + expect(direct.compat.isVercelGatewayHost).toBe(false); +});