feat(coding-agent): added providers.cacheRetention setting for prompt caching

- Add the `providers.cacheRetention` setting to control prompt-cache retention options per request.
- Forward configured cache retention preferences through the settings-aware stream function.
- Update documentation and test coverage for long cache retention behaviors.
This commit is contained in:
can1357
2026-08-19 00:56:50 +02:00
parent bf490ae024
commit 565d53515b
6 changed files with 90 additions and 2 deletions
+2
View File
@@ -707,6 +707,7 @@ providers:
openaiWebsockets: auto
openrouterVariant: default
kimiApiFormat: auto
cacheRetention: auto
maxInFlightRequests:
anthropic: 2
@@ -738,6 +739,7 @@ searxng:
| `providers.openaiWebsockets` | enum | `auto` | `auto`, `off`, `on`. |
| `providers.openrouterVariant` | enum | `default` | `default`, `nitro`, `floor`, `online`, `exacto`. |
| `providers.kimiApiFormat` | enum | `auto` | `auto`, `openai`, `anthropic`. `auto` follows live model metadata. |
| `providers.cacheRetention` | enum | `auto` | `auto`, `short`, `long`, `none`. Prompt-cache retention forwarded to providers that support it. `auto` keeps provider defaults (Anthropic: 5m entries + idle keep-alive refreshes) and honors `PI_CACHE_RETENTION`; `short` forces 5m; `long` uses 1h TTLs where supported and disables keep-alive refreshes; `none` disables prompt caching and cache-affinity routing. |
| `provider.appendOnlyContext` | enum | `auto` | `auto`, `on`, `off`. |
| `exa.enabled` | boolean | `true` | Enable Exa integration. |
| `exa.enableSearch` | boolean | `true` | Exa search. |
@@ -1,7 +1,7 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { streamSimple } from "@oh-my-pi/pi-ai";
import type { MessageCreateParams } from "@oh-my-pi/pi-ai/providers/anthropic-wire";
import type { Context, FetchImpl, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
import type { CacheControlEphemeral, MessageCreateParams } from "@oh-my-pi/pi-ai/providers/anthropic-wire";
import type { CacheRetention, Context, FetchImpl, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
const CACHE_REFRESH_DELAY_MS = 5 * 60_000 - 15_000;
@@ -161,6 +161,7 @@ function createFetch(modes: ResponseMode[], capture: FetchCapture): FetchImpl {
interface FinishRequestOptions {
anthropicCacheRefresh?: boolean;
cacheRetention?: CacheRetention;
model?: Model<"anthropic-messages">;
sessionId?: string;
}
@@ -175,6 +176,7 @@ async function finishRequest(
fetch,
apiKey: "test-anthropic-key",
anthropicCacheRefresh: options.anthropicCacheRefresh ?? true,
cacheRetention: options.cacheRetention,
providerSessionState,
sessionId: options.sessionId ?? "cache-refresh-test-session",
});
@@ -285,4 +287,28 @@ describe("Anthropic prompt-cache refresh", () => {
expect(capture.bodies[1]?.stream).toBe(true);
expect(capture.thinkingRefreshAborted).toBe(true);
});
it("skips keep-alive refreshes and emits 1h breakpoints when retention is long", async () => {
vi.useFakeTimers();
const capture: FetchCapture = { bodies: [], thinkingRefreshAborted: false };
const fetch = createFetch(["ordinary-write"], capture);
const states = createProviderSessionState();
await finishRequest(fetch, states, { cacheRetention: "long" });
vi.advanceTimersByTime(CACHE_REFRESH_DELAY_MS * 2);
await Promise.resolve();
// No zero-output replay was scheduled for the 1h entry.
expect(capture.bodies).toHaveLength(1);
const blocks = (capture.bodies[0]?.messages ?? []).flatMap(message =>
Array.isArray(message.content) ? message.content : [],
);
const breakpoints = blocks
.map(block => ("cache_control" in block ? (block.cache_control ?? undefined) : undefined))
.filter((cc): cc is CacheControlEphemeral => cc != null);
expect(breakpoints.length).toBeGreaterThan(0);
for (const cc of breakpoints) {
expect(cc.ttl).toBe("1h");
}
});
});
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Added
- Added `providers.cacheRetention` setting (`/settings` → Providers → Protocol) to control prompt-cache retention per request: `auto` keeps the provider default (Anthropic: 5m entries with idle keep-alive refreshes), `short` forces 5m, `long` restores 1h TTLs where supported and disables the keep-alive refresh loop, `none` disables prompt caching.
## [17.3.7] - 2026-08-17
### Changed
@@ -5254,6 +5254,39 @@ export const SETTINGS_SCHEMA = {
},
},
"providers.cacheRetention": {
type: "enum",
values: ["auto", "short", "long", "none"] as const,
default: "auto",
ui: {
tab: "providers",
group: "Protocol",
label: "Prompt Cache Retention",
description:
"Prompt-cache retention forwarded to providers that support it (Anthropic, Bedrock, OpenRouter, OpenAI)",
options: [
{
value: "auto",
label: "Auto",
description:
"Provider default — Anthropic uses 5m entries kept warm by idle keep-alive refreshes; PI_CACHE_RETENTION still applies",
},
{
value: "short",
label: "Short (5m)",
description:
"Cheapest cache writes; Anthropic keeps the entry warm with bounded keep-alive refreshes while idle",
},
{
value: "long",
label: "Long (1h)",
description: "1h TTL where the provider supports it; pricier writes, no keep-alive refresh requests",
},
{ value: "none", label: "Off", description: "Disable prompt caching and cache-affinity routing" },
],
},
},
"providers.streamFirstEventTimeoutSeconds": {
type: "number",
default: -1,
@@ -41,6 +41,12 @@ export function createSettingsAwareStreamFn(settings: Settings, base: StreamFn =
: model.api === "openai-responses"
? settings.get("textVerbosity")
: undefined;
// "auto" leaves the option unset so provider defaults and the
// PI_CACHE_RETENTION env override keep working; anything else is an
// explicit per-request retention (long restores 1h Anthropic TTLs and
// implicitly disables the short-entry keep-alive refresh loop).
const cacheRetentionSetting = settings.get("providers.cacheRetention");
const cacheRetention = cacheRetentionSetting === "auto" ? undefined : cacheRetentionSetting;
const streamFirstEventTimeoutMs = timeoutSecondsToMs(settings.get("providers.streamFirstEventTimeoutSeconds"));
const streamIdleTimeoutMs = timeoutSecondsToMs(settings.get("providers.streamIdleTimeoutSeconds"));
// Server-side fallback (opt-in): when the user enables it AND the
@@ -60,6 +66,7 @@ export function createSettingsAwareStreamFn(settings: Settings, base: StreamFn =
openrouterVariant: streamOptions?.openrouterVariant ?? openrouterVariant,
antigravityEndpointMode: streamOptions?.antigravityEndpointMode ?? antigravityEndpointMode,
textVerbosity: streamOptions?.textVerbosity ?? textVerbosity,
cacheRetention: streamOptions?.cacheRetention ?? cacheRetention,
streamFirstEventTimeoutMs: streamOptions?.streamFirstEventTimeoutMs ?? streamFirstEventTimeoutMs,
streamIdleTimeoutMs: streamOptions?.streamIdleTimeoutMs ?? streamIdleTimeoutMs,
maxRetryDelayMs: streamOptions?.maxRetryDelayMs ?? settings.get("retry.maxDelayMs"),
@@ -147,6 +147,22 @@ describe("createSettingsAwareStreamFn", () => {
expect(calls[0]?.options?.openrouterVariant).toBeUndefined();
});
it("forwards configured cache retention, leaves auto unset, and lets callers override", () => {
const auto = captureBase();
createSettingsAwareStreamFn(Settings.isolated({}), auto.fn)(stubModel, stubContext, undefined);
// auto must stay unset so provider defaults and PI_CACHE_RETENTION apply
expect(auto.calls[0]?.options?.cacheRetention).toBeUndefined();
const long = captureBase();
const settings = Settings.isolated({ "providers.cacheRetention": "long" });
const wrapped = createSettingsAwareStreamFn(settings, long.fn);
wrapped(stubModel, stubContext, undefined);
expect(long.calls[0]?.options?.cacheRetention).toBe("long");
wrapped(stubModel, stubContext, { cacheRetention: "none" });
expect(long.calls[1]?.options?.cacheRetention).toBe("none");
});
it("lets caller-supplied options override the session settings", () => {
const settings = Settings.isolated({
"providers.openrouterVariant": "floor",