feat: implemented anthropic keep-alive and migration support for updates
- Added Anthropic prompt-cache refresh scheduling and state management to keep prompts warm across idle sessions. - Updated pricing models and database stats tracking to calculate and store cost-weighted cache savings. - Integrated cache savings metrics and efficiency displays into the stats CLI, dashboard routes, and UI components. - Added support for package renaming, manifest pointer tracking, and installation migration during CLI updates.
This commit is contained in:
@@ -1,6 +1,6 @@
|
||||
import { Database } from "bun:sqlite";
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { closeDb, getRecentRequests, initDb, insertMessageStats } from "@oh-my-pi/omp-stats/db";
|
||||
import { closeDb, getOverallStats, getRecentRequests, initDb, insertMessageStats } from "@oh-my-pi/omp-stats/db";
|
||||
import type { MessageStats } from "@oh-my-pi/omp-stats/types";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import { getStatsDbPath } from "@oh-my-pi/pi-utils";
|
||||
@@ -46,6 +46,32 @@ function expectedCodexGptCost() {
|
||||
};
|
||||
}
|
||||
|
||||
function createAnthropicCacheStats(entryId: string, cacheRead: number, cacheWrite: number): MessageStats {
|
||||
const input = 1_000 - cacheRead - cacheWrite;
|
||||
return {
|
||||
sessionFile: "/tmp/anthropic-session.jsonl",
|
||||
entryId,
|
||||
folder: "/tmp/project",
|
||||
model: "claude-sonnet-4-6",
|
||||
provider: "anthropic",
|
||||
api: "anthropic-messages",
|
||||
timestamp: Date.now(),
|
||||
duration: 1000,
|
||||
ttft: 100,
|
||||
stopReason: "stop",
|
||||
errorMessage: null,
|
||||
usage: {
|
||||
input,
|
||||
output: 0,
|
||||
cacheRead,
|
||||
cacheWrite,
|
||||
totalTokens: 1_000,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
agentType: "main",
|
||||
};
|
||||
}
|
||||
|
||||
describe("stats GPT cost correction", () => {
|
||||
it("stores catalog-derived cost when OpenAI Codex session usage has zero cost", async () => {
|
||||
await initDb();
|
||||
@@ -107,3 +133,55 @@ describe("stats GPT cost correction", () => {
|
||||
expect(request?.usage.cost.total).toBeCloseTo(expectedCodexGptCost().total, 8);
|
||||
});
|
||||
});
|
||||
|
||||
describe("stats cache metrics", () => {
|
||||
it("subtracts 5-minute writes from the savings produced by cache reads", async () => {
|
||||
await initDb();
|
||||
insertMessageStats([createAnthropicCacheStats("mixed-cache", 800, 100)]);
|
||||
|
||||
// 100 uncached + 800 reads at 0.1x + 100 writes at 1.25x = 305,
|
||||
// versus 1,000 tokens at the uncached input rate.
|
||||
expect(getOverallStats().cacheSavings).toBeCloseTo(0.695, 8);
|
||||
expect(getOverallStats().cacheRate).toBeCloseTo(800 / 900, 8);
|
||||
});
|
||||
|
||||
it("reports cache writes without reads as negative savings", async () => {
|
||||
await initDb();
|
||||
insertMessageStats([createAnthropicCacheStats("cache-write", 0, 1_000)]);
|
||||
|
||||
expect(getOverallStats().cacheSavings).toBeCloseTo(-0.25, 8);
|
||||
});
|
||||
|
||||
it("charges 1-hour cache writes at their full overhead", async () => {
|
||||
await initDb();
|
||||
const stats = createAnthropicCacheStats("one-hour-write", 0, 1_000);
|
||||
stats.usage.cost = {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0.006,
|
||||
total: 0.006,
|
||||
};
|
||||
insertMessageStats([stats]);
|
||||
|
||||
expect(getOverallStats().cacheSavings).toBeCloseTo(-1, 8);
|
||||
});
|
||||
|
||||
it("excludes unpriced custom models from the savings ratio", async () => {
|
||||
await initDb();
|
||||
const known = createAnthropicCacheStats("known", 800, 100);
|
||||
const unpriced = createAnthropicCacheStats("unpriced", 0, 0);
|
||||
unpriced.provider = "custom";
|
||||
unpriced.model = "custom-model";
|
||||
unpriced.usage.cost = {
|
||||
input: 1,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
total: 1,
|
||||
};
|
||||
insertMessageStats([known, unpriced]);
|
||||
|
||||
expect(getOverallStats().cacheSavings).toBeCloseTo(0.695, 8);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -14,6 +14,7 @@ const stats: AggregatedStats = {
|
||||
totalCacheReadTokens: 300,
|
||||
totalCacheWriteTokens: 40,
|
||||
cacheRate: 0.75,
|
||||
cacheSavings: 0.695,
|
||||
totalCost: 0,
|
||||
totalPremiumRequests: 0,
|
||||
avgDuration: 1000,
|
||||
@@ -31,6 +32,11 @@ describe("overview token metrics", () => {
|
||||
expect(html).toContain("Cache Read");
|
||||
expect(html).toContain("Conversation Total");
|
||||
expect(html).toContain("Uncached input + cache reads + cache writes + output");
|
||||
expect(html).toContain("Cache Rate");
|
||||
expect(html).toContain("Cache Savings");
|
||||
expect(html).toContain("75.0%");
|
||||
expect(html).toContain("69.5%");
|
||||
expect(html).toContain("cache writes can make this negative");
|
||||
|
||||
const expectedTotal = formatCompact(
|
||||
stats.totalInputTokens +
|
||||
|
||||
Reference in New Issue
Block a user