feat(catalog): model Bedrock prompt cache limits

This commit is contained in:
Alexander Kirilin
2026-07-23 15:04:13 -04:00
parent c64e7146e0
commit 1fa847dfdf
8 changed files with 331 additions and 10 deletions
+2
View File
@@ -4,6 +4,8 @@
### Added
- Added resolved Bedrock Converse prompt-cache compatibility limits, including explicit 5-minute checkpoint support for bundled Nova Lite, Micro, and Pro models and model-specific 1-hour Claude retention.
- Added the native Meta Model API provider and Muse Spark 1.1 with Responses API reasoning replay, image input, and the full supported reasoning-effort ladder ([#4941](https://github.com/can1357/oh-my-pi/issues/4941)).
## [17.0.9] - 2026-07-23
+4
View File
@@ -9,7 +9,9 @@
* Request handlers read fields — they never detect, parse ids, or allocate
* compat per request.
*/
import { buildAnthropicCompat } from "./compat/anthropic";
import { buildBedrockCompat } from "./compat/bedrock";
import { buildDevinCompat } from "./compat/devin";
import { buildOpenAICompat, buildOpenAIResponsesCompat, buildOpenRouterCompat } from "./compat/openai";
import { resolveModelThinking } from "./model-thinking";
@@ -39,6 +41,8 @@ export function buildCompat(spec: ModelSpec<Api>): CompatOf<Api> {
return buildOpenAIResponsesCompat(spec as ModelSpec<"openai-responses">);
case "anthropic-messages":
return buildAnthropicCompat(spec as ModelSpec<"anthropic-messages">);
case "bedrock-converse-stream":
return buildBedrockCompat(spec as ModelSpec<"bedrock-converse-stream">);
case "devin-agent":
return buildDevinCompat(spec as ModelSpec<"devin-agent">);
default:
+100
View File
@@ -0,0 +1,100 @@
import type { ModelSpec, ResolvedBedrockCompat } from "../types";
import { applyCompatOverrides } from "./apply";
const NO_EXPLICIT_CHECKPOINTS: ResolvedBedrockCompat = {
promptCacheMode: "none",
supportsLongPromptCacheRetention: false,
promptCacheMinimumTokens: 0,
promptCacheMaximumCheckpoints: 0,
};
const EXPLICIT_CHECKPOINTS_1024_5M: ResolvedBedrockCompat = {
promptCacheMode: "explicit",
supportsLongPromptCacheRetention: false,
promptCacheMinimumTokens: 1024,
promptCacheMaximumCheckpoints: 4,
};
const EXPLICIT_CHECKPOINTS_1024_1H: ResolvedBedrockCompat = {
promptCacheMode: "explicit",
supportsLongPromptCacheRetention: true,
promptCacheMinimumTokens: 1024,
promptCacheMaximumCheckpoints: 4,
};
const EXPLICIT_CHECKPOINTS_2048_5M: ResolvedBedrockCompat = {
promptCacheMode: "explicit",
supportsLongPromptCacheRetention: false,
promptCacheMinimumTokens: 2048,
promptCacheMaximumCheckpoints: 4,
};
const EXPLICIT_CHECKPOINTS_4096_5M: ResolvedBedrockCompat = {
promptCacheMode: "explicit",
supportsLongPromptCacheRetention: false,
promptCacheMinimumTokens: 4096,
promptCacheMaximumCheckpoints: 4,
};
const EXPLICIT_CHECKPOINTS_4096_1H: ResolvedBedrockCompat = {
promptCacheMode: "explicit",
supportsLongPromptCacheRetention: true,
promptCacheMinimumTokens: 4096,
promptCacheMaximumCheckpoints: 4,
};
/**
* Explicit Nova cache points complement Bedrock's automatic prefix caching:
* AWS recommends them for consistent cache hits and input-cost savings. Keep
* this exact bundled set conservative rather than treating arbitrary Nova-like
* application profiles as checkpoint-capable.
*/
function detectedBedrockCompat(modelId: string): ResolvedBedrockCompat {
const id = modelId.toLowerCase();
if (id === "us.amazon.nova-lite-v1:0" || id === "us.amazon.nova-micro-v1:0" || id === "us.amazon.nova-pro-v1:0") {
return EXPLICIT_CHECKPOINTS_1024_5M;
}
// https://docs.aws.amazon.com/bedrock/latest/userguide/prompt-caching.html
// This list is deliberately sourced from AWS model cards, not cache pricing:
// https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards.html
if (
id.includes("anthropic.claude-opus-4-5") ||
id.includes("anthropic.claude-sonnet-4-5") ||
id.includes("anthropic.claude-haiku-4-5") ||
id.includes("anthropic.claude-opus-4-7") ||
id.includes("anthropic.claude-opus-4-8") ||
id.includes("anthropic.claude-sonnet-5")
) {
return EXPLICIT_CHECKPOINTS_4096_1H;
}
if (id.includes("anthropic.claude-opus-4-6")) {
return EXPLICIT_CHECKPOINTS_4096_5M;
}
if (id.includes("anthropic.claude-3-5-haiku")) {
return EXPLICIT_CHECKPOINTS_2048_5M;
}
if (id.includes("anthropic.claude-fable-5")) {
return EXPLICIT_CHECKPOINTS_1024_1H;
}
if (
id.includes("anthropic.claude-opus-4-1") ||
id.includes("anthropic.claude-opus-4-20250514") ||
id.includes("anthropic.claude-sonnet-4-20250514") ||
id.includes("anthropic.claude-sonnet-4-6") ||
id.includes("anthropic.claude-3-7-sonnet") ||
id.includes("anthropic.claude-3-5-sonnet-20241022-v2")
) {
return EXPLICIT_CHECKPOINTS_1024_5M;
}
return NO_EXPLICIT_CHECKPOINTS;
}
/** Resolve Bedrock Converse prompt-cache capabilities once per model. */
export function buildBedrockCompat(spec: ModelSpec<"bedrock-converse-stream">): ResolvedBedrockCompat {
const compat = { ...detectedBedrockCompat(spec.id) };
applyCompatOverrides(compat, spec.compat);
return compat;
}
+33 -6
View File
@@ -459,6 +459,29 @@ export interface AnthropicCompat {
escapeBuiltinToolNames?: boolean;
}
/**
* Compatibility settings for Bedrock Converse prompt caching. Cache pricing is
* deliberately not used to infer these request-shape capabilities.
*/
export interface BedrockCompat {
/** Whether this endpoint accepts no checkpoints, automatic caching, or explicit cachePoint blocks. */
promptCacheMode?: "none" | "automatic" | "explicit";
/** Whether explicit cachePoint blocks accept `ttl: "1h"`; omitted TTL means Bedrock's 5-minute default. */
supportsLongPromptCacheRetention?: boolean;
/** Minimum prompt-prefix tokens required for an effective checkpoint. Zero means no explicit checkpoints. */
promptCacheMinimumTokens?: number;
/** Maximum explicit cache checkpoints accepted in one request. Zero means no explicit checkpoints. */
promptCacheMaximumCheckpoints?: number;
}
/** Fully-resolved Bedrock Converse prompt-cache capabilities, materialized once by `buildModel`. */
export interface ResolvedBedrockCompat {
promptCacheMode: NonNullable<BedrockCompat["promptCacheMode"]>;
supportsLongPromptCacheRetention: boolean;
promptCacheMinimumTokens: number;
promptCacheMaximumCheckpoints: number;
}
/**
* OpenRouter provider routing preferences.
* Controls which upstream providers OpenRouter routes requests to.
@@ -681,9 +704,11 @@ export type CompatConfigOf<TApi extends Api> = TApi extends
? OpenAICompat
: TApi extends "anthropic-messages"
? AnthropicCompat
: TApi extends "devin-agent"
? DevinCompat
: undefined;
: TApi extends "bedrock-converse-stream"
? BedrockCompat
: TApi extends "devin-agent"
? DevinCompat
: undefined;
/** Resolved compat for a given API: complete record, materialized once by `buildModel`. */
export type CompatOf<TApi extends Api> = TApi extends "openrouter"
@@ -694,9 +719,11 @@ export type CompatOf<TApi extends Api> = TApi extends "openrouter"
? ResolvedOpenAIResponsesCompat
: TApi extends "anthropic-messages"
? ResolvedAnthropicCompat
: TApi extends "devin-agent"
? ResolvedDevinCompat
: undefined;
: TApi extends "bedrock-converse-stream"
? ResolvedBedrockCompat
: TApi extends "devin-agent"
? ResolvedDevinCompat
: undefined;
/** Provider-native compaction endpoint configuration for one model. */
export interface RemoteCompactionConfig<TApi extends Api = Api> {
@@ -0,0 +1,133 @@
import { describe, expect, test } from "bun:test";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
function bedrockSpec(
overrides: Partial<ModelSpec<"bedrock-converse-stream">> = {},
): ModelSpec<"bedrock-converse-stream"> {
return {
id: "anthropic.claude-opus-4-6-v1",
name: "Claude Opus 4.6",
api: "bedrock-converse-stream",
provider: "amazon-bedrock",
baseUrl: "https://bedrock-runtime.us-east-1.amazonaws.com",
reasoning: true,
input: ["text"],
cost: { input: 5, output: 25, cacheRead: 0.5, cacheWrite: 6.25 },
contextWindow: 1_000_000,
maxTokens: 128_000,
...overrides,
};
}
describe("Bedrock prompt-cache compat", () => {
test("resolves the AWS-documented capability for every cache-priced bundled Claude family", () => {
const cases = [
{
id: "anthropic.claude-3-5-haiku-20241022-v1:0",
minimumTokens: 2048,
supportsLongRetention: false,
},
// Current AWS docs do not advertise Converse cache checkpoints for this
// legacy v1 model, so catalog cache pricing alone must not enable them.
{
id: "anthropic.claude-3-5-sonnet-20240620-v1:0",
minimumTokens: 0,
supportsLongRetention: false,
},
{
id: "anthropic.claude-3-5-sonnet-20241022-v2:0",
minimumTokens: 1024,
supportsLongRetention: false,
},
{
id: "anthropic.claude-3-7-sonnet-20250219-v1:0",
minimumTokens: 1024,
supportsLongRetention: false,
},
{ id: "anthropic.claude-fable-5", minimumTokens: 1024, supportsLongRetention: true },
{
id: "anthropic.claude-haiku-4-5-20251001-v1:0",
minimumTokens: 4096,
supportsLongRetention: true,
},
{
id: "anthropic.claude-opus-4-1-20250805-v1:0",
minimumTokens: 1024,
supportsLongRetention: false,
},
{
id: "anthropic.claude-opus-4-20250514-v1:0",
minimumTokens: 1024,
supportsLongRetention: false,
},
{
id: "anthropic.claude-opus-4-5-20251101-v1:0",
minimumTokens: 4096,
supportsLongRetention: true,
},
{ id: "anthropic.claude-opus-4-6-v1", minimumTokens: 4096, supportsLongRetention: false },
{ id: "global.anthropic.claude-opus-4-7", minimumTokens: 4096, supportsLongRetention: true },
{ id: "us.anthropic.claude-opus-4-8", minimumTokens: 4096, supportsLongRetention: true },
{
id: "anthropic.claude-sonnet-4-20250514-v1:0",
minimumTokens: 1024,
supportsLongRetention: false,
},
{
id: "anthropic.claude-sonnet-4-5-20250929-v1:0",
minimumTokens: 4096,
supportsLongRetention: true,
},
{ id: "anthropic.claude-sonnet-4-6", minimumTokens: 1024, supportsLongRetention: false },
{ id: "us.anthropic.claude-sonnet-5", minimumTokens: 4096, supportsLongRetention: true },
] as const;
for (const { id, minimumTokens, supportsLongRetention } of cases) {
expect(buildModel(bedrockSpec({ id })).compat).toEqual({
promptCacheMode: minimumTokens === 0 ? "none" : "explicit",
supportsLongPromptCacheRetention: supportsLongRetention,
promptCacheMinimumTokens: minimumTokens,
promptCacheMaximumCheckpoints: minimumTokens === 0 ? 0 : 4,
});
}
});
test("models every bundled cache-capable Nova variant for explicit 5m checkpoints", () => {
for (const id of ["us.amazon.nova-lite-v1:0", "us.amazon.nova-micro-v1:0", "us.amazon.nova-pro-v1:0"] as const) {
const model = getBundledModel<"bedrock-converse-stream">("amazon-bedrock", id);
expect(model?.compat).toEqual({
promptCacheMode: "explicit",
supportsLongPromptCacheRetention: false,
promptCacheMinimumTokens: 1024,
promptCacheMaximumCheckpoints: 4,
});
}
});
test("keeps unknown routes conservative and honors sparse profile overrides", () => {
const unknown = buildModel(
bedrockSpec({ id: "arn:aws:bedrock:us-east-1:123:application-inference-profile/opaque" }),
);
expect(unknown.compat.promptCacheMode).toBe("none");
const sparse = {
promptCacheMode: "explicit" as const,
promptCacheMinimumTokens: 1024,
promptCacheMaximumCheckpoints: 4,
};
const configured = buildModel(
bedrockSpec({ id: "arn:aws:bedrock:us-east-1:123:application-inference-profile/opaque", compat: sparse }),
);
expect(configured.compat).toEqual({ ...unknown.compat, ...sparse });
expect(configured.compatConfig).toBe(sparse);
});
test("keeps bundled models memoized while materializing resolved compat", () => {
const first = getBundledModel<"bedrock-converse-stream">("amazon-bedrock", "anthropic.claude-opus-4-6-v1");
const second = getBundledModel<"bedrock-converse-stream">("amazon-bedrock", "anthropic.claude-opus-4-6-v1");
expect(first).toBe(second);
expect(first?.compat.promptCacheMode).toBe("explicit");
});
});
+3
View File
@@ -26,6 +26,9 @@
- Bound interactive bash live display write queue to prevent unbounded PTY chunk backlog ([#4240](https://github.com/can1357/oh-my-pi/issues/4240))
- All Markdown flavors (`.markdown`, `.mdx`, `.mdc`, `.mkd`, `.mdown`) now follow the `read.summarize.prose` setting like `.md`, so they read verbatim instead of being code-block summarized when prose summaries are off.
- xAI web search now uses `grok-4.5` (at low reasoning effort) instead of `grok-4.3`.
### Added
- Added `models.yml` Bedrock Converse prompt-cache capability overrides for bundled and opaque inference profiles.
### Fixed
@@ -73,8 +73,19 @@ export const OpenAICompatSchema = type({
"whenThinking?": OpenAICompatFieldsSchema,
});
const BedrockCompatSchema = type({
"promptCacheMode?": '"none" | "automatic" | "explicit"',
"supportsLongPromptCacheRetention?": "boolean",
"promptCacheMinimumTokens?": "number >= 0",
"promptCacheMaximumCheckpoints?": "number >= 0",
});
// Provider-level overrides can target bundled models whose API is not repeated
// in models.yml, so preserve the sparse compat shape for each supported API.
const ApiCompatSchema = OpenAICompatSchema.or(BedrockCompatSchema);
const ApiSchema = type(
'"openai-completions" | "openai-responses" | "openai-codex-responses" | "azure-openai-responses" | "anthropic-messages" | "google-generative-ai" | "google-gemini-cli" | "google-vertex"',
'"openai-completions" | "openai-responses" | "openai-codex-responses" | "azure-openai-responses" | "anthropic-messages" | "bedrock-converse-stream" | "google-generative-ai" | "google-gemini-cli" | "google-vertex"',
);
const EffortSchema = type('"minimal" | "low" | "medium" | "high" | "xhigh" | "max"');
@@ -173,7 +184,7 @@ const ModelDefinitionSchema = type({
"maxTokens?": "number",
"omitMaxOutputTokens?": "boolean",
"headers?": { "[string]": "string" },
"compat?": OpenAICompatSchema,
"compat?": ApiCompatSchema,
"contextPromotionTarget?": "string",
"compactionModel?": "string",
"remoteCompaction?": RemoteCompactionSchema,
@@ -222,7 +233,7 @@ export const ModelOverrideSchema = type({
"maxTokens?": "number",
"omitMaxOutputTokens?": "boolean",
"headers?": { "[string]": "string" },
"compat?": OpenAICompatSchema,
"compat?": ApiCompatSchema,
"contextPromotionTarget?": "string",
"compactionModel?": "string",
"remoteCompaction?": RemoteCompactionSchema,
@@ -263,7 +274,7 @@ const ProviderConfigSchema = type({
"apiKey?": "string",
"api?": ApiSchema,
"headers?": { "[string]": "string" },
"compat?": OpenAICompatSchema,
"compat?": ApiCompatSchema,
"remoteCompaction?": RemoteCompactionSchema,
"authHeader?": "boolean",
"auth?": ProviderAuthSchema,
@@ -33,6 +33,22 @@ describe("ModelRegistry default custom models config", () => {
expect(model?.baseUrl).toBe("https://yaml-default.example.com/v1");
});
test("loads Bedrock cache capabilities from a model override", () => {
writeBedrockCacheOverride();
const model = loadDefaultRegistryModel({
provider: "amazon-bedrock",
modelId: "us.anthropic.claude-opus-4-8",
});
expect(model?.compat).toEqual({
promptCacheMode: "explicit",
supportsLongPromptCacheRetention: false,
promptCacheMinimumTokens: 1024,
promptCacheMaximumCheckpoints: 4,
});
});
test("prefers default models.yml over models.yaml when both exist", () => {
writeModelsYaml("models.yml", {
provider: "yaml-precedence",
@@ -105,6 +121,12 @@ interface ModelSnapshot {
id: string;
name: string;
baseUrl: string | undefined;
compat: {
promptCacheMode: string;
supportsLongPromptCacheRetention: boolean;
promptCacheMinimumTokens: number;
promptCacheMaximumCheckpoints: number;
};
}
function writeModelsYaml(file: "models.yml" | "models.yaml", fixture: ProviderFixture): void {
@@ -133,6 +155,24 @@ function writeModelsYaml(file: "models.yml" | "models.yaml", fixture: ProviderFi
);
}
function writeBedrockCacheOverride(): void {
fs.writeFileSync(
path.join(tempDir.path(), "models.yml"),
[
"providers:",
" amazon-bedrock:",
" modelOverrides:",
" us.anthropic.claude-opus-4-8:",
" compat:",
" promptCacheMode: explicit",
" supportsLongPromptCacheRetention: false",
" promptCacheMinimumTokens: 1024",
" promptCacheMaximumCheckpoints: 4",
"",
].join("\n"),
);
}
function writeModelsJson(fixture: ProviderFixture): void {
fs.writeFileSync(
path.join(tempDir.path(), "models.json"),
@@ -173,6 +213,7 @@ function loadDefaultRegistryModel(lookup: ModelLookup): ModelSnapshot | undefine
id: model.id,
name: model.name,
baseUrl: model.baseUrl,
compat: model.compat,
} : null));
} finally {
authStorage.close();