feat: implemented anthropic keep-alive and migration support for updates

- Added Anthropic prompt-cache refresh scheduling and state management to keep prompts warm across idle sessions.
- Updated pricing models and database stats tracking to calculate and store cost-weighted cache savings.
- Integrated cache savings metrics and efficiency displays into the stats CLI, dashboard routes, and UI components.
- Added support for package renaming, manifest pointer tracking, and installation migration during CLI updates.
This commit is contained in:
can1357
2026-08-13 03:53:47 +02:00
parent b60bef961c
commit 1132c3e31c
35 changed files with 1488 additions and 518 deletions
+79 -1
View File
@@ -1,6 +1,6 @@
import { Database } from "bun:sqlite";
import { describe, expect, it } from "bun:test";
import { closeDb, getRecentRequests, initDb, insertMessageStats } from "@oh-my-pi/omp-stats/db";
import { closeDb, getOverallStats, getRecentRequests, initDb, insertMessageStats } from "@oh-my-pi/omp-stats/db";
import type { MessageStats } from "@oh-my-pi/omp-stats/types";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { getStatsDbPath } from "@oh-my-pi/pi-utils";
@@ -46,6 +46,32 @@ function expectedCodexGptCost() {
};
}
function createAnthropicCacheStats(entryId: string, cacheRead: number, cacheWrite: number): MessageStats {
const input = 1_000 - cacheRead - cacheWrite;
return {
sessionFile: "/tmp/anthropic-session.jsonl",
entryId,
folder: "/tmp/project",
model: "claude-sonnet-4-6",
provider: "anthropic",
api: "anthropic-messages",
timestamp: Date.now(),
duration: 1000,
ttft: 100,
stopReason: "stop",
errorMessage: null,
usage: {
input,
output: 0,
cacheRead,
cacheWrite,
totalTokens: 1_000,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
agentType: "main",
};
}
describe("stats GPT cost correction", () => {
it("stores catalog-derived cost when OpenAI Codex session usage has zero cost", async () => {
await initDb();
@@ -107,3 +133,55 @@ describe("stats GPT cost correction", () => {
expect(request?.usage.cost.total).toBeCloseTo(expectedCodexGptCost().total, 8);
});
});
describe("stats cache metrics", () => {
it("subtracts 5-minute writes from the savings produced by cache reads", async () => {
await initDb();
insertMessageStats([createAnthropicCacheStats("mixed-cache", 800, 100)]);
// 100 uncached + 800 reads at 0.1x + 100 writes at 1.25x = 305,
// versus 1,000 tokens at the uncached input rate.
expect(getOverallStats().cacheSavings).toBeCloseTo(0.695, 8);
expect(getOverallStats().cacheRate).toBeCloseTo(800 / 900, 8);
});
it("reports cache writes without reads as negative savings", async () => {
await initDb();
insertMessageStats([createAnthropicCacheStats("cache-write", 0, 1_000)]);
expect(getOverallStats().cacheSavings).toBeCloseTo(-0.25, 8);
});
it("charges 1-hour cache writes at their full overhead", async () => {
await initDb();
const stats = createAnthropicCacheStats("one-hour-write", 0, 1_000);
stats.usage.cost = {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0.006,
total: 0.006,
};
insertMessageStats([stats]);
expect(getOverallStats().cacheSavings).toBeCloseTo(-1, 8);
});
it("excludes unpriced custom models from the savings ratio", async () => {
await initDb();
const known = createAnthropicCacheStats("known", 800, 100);
const unpriced = createAnthropicCacheStats("unpriced", 0, 0);
unpriced.provider = "custom";
unpriced.model = "custom-model";
unpriced.usage.cost = {
input: 1,
output: 0,
cacheRead: 0,
cacheWrite: 0,
total: 1,
};
insertMessageStats([known, unpriced]);
expect(getOverallStats().cacheSavings).toBeCloseTo(0.695, 8);
});
});
@@ -14,6 +14,7 @@ const stats: AggregatedStats = {
totalCacheReadTokens: 300,
totalCacheWriteTokens: 40,
cacheRate: 0.75,
cacheSavings: 0.695,
totalCost: 0,
totalPremiumRequests: 0,
avgDuration: 1000,
@@ -31,6 +32,11 @@ describe("overview token metrics", () => {
expect(html).toContain("Cache Read");
expect(html).toContain("Conversation Total");
expect(html).toContain("Uncached input + cache reads + cache writes + output");
expect(html).toContain("Cache Rate");
expect(html).toContain("Cache Savings");
expect(html).toContain("75.0%");
expect(html).toContain("69.5%");
expect(html).toContain("cache writes can make this negative");
const expectedTotal = formatCompact(
stats.totalInputTokens +