feat(stats): added tool usage dashboard

- Parsed assistant `toolCall` blocks and `toolResult` messages into persisted `tool_calls` rows with one-shot historical backfill.
- Added tool aggregate queries, model breakdowns, call time-series data, and the `/api/stats/tools` dashboard endpoint.
- Added the `/#/tools` route with summary metrics, stacked calls-over-time chart, per-tool table, and model breakdown panel.
- Added end-to-end stats coverage for tool ingestion, result/error linkage, fork deduplication, incremental updates, and dashboard shaping.
This commit is contained in:
can1357
2026-07-02 08:31:18 +02:00
parent 41cc57c238
commit fbf3a6eb3f
16 changed files with 1315 additions and 4 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Added
- Added a Tools tab to the `omp stats` dashboard (`/#/tools`): per-tool call counts, error rates, result/argument payload sizes, per-model breakdown, and a stacked calls-over-time chart. Token and cost columns attribute each invoking turn's real provider usage evenly across that turn's tool calls. Existing databases re-parse sessions once on the next sync to backfill historical tool calls.
## [16.2.7] - 2026-06-30
### Fixed
+22 -1
View File
@@ -17,10 +17,15 @@ import {
getStatsByFolder,
getStatsByModel,
getTimeSeries,
getToolStats,
getToolStatsByModel,
getToolTimeSeries,
initDb,
insertMessageStats,
insertToolCalls,
insertUserMessageStats,
setFileOffset,
updateToolResults,
updateUserMessageLinks,
} from "./db";
import { getSessionEntry, listAllSessionFiles, type ParseSessionResult, parseSessionFile } from "./parser";
@@ -29,7 +34,7 @@ import type { SyncWorkerRequest, SyncWorkerResponse } from "./sync-worker";
// hidden argv mode, so the compiled binary and npm bundle only need one
// JavaScript entry. Standalone source `omp-stats` keeps using this package's
// own sync-worker source file.
import type { BehaviorDashboardStats, DashboardStats, MessageStats, RequestDetails } from "./types";
import type { BehaviorDashboardStats, DashboardStats, MessageStats, RequestDetails, ToolDashboardStats } from "./types";
/**
* Apply a freshly parsed result to the database. Runs entirely on the
@@ -39,6 +44,8 @@ function applyParseResult(sessionFile: string, lastModified: number, result: Par
if (result.stats.length > 0) insertMessageStats(result.stats);
if (result.userStats.length > 0) insertUserMessageStats(result.userStats);
if (result.userLinks.length > 0) updateUserMessageLinks(result.userLinks);
if (result.toolCalls.length > 0) insertToolCalls(result.toolCalls);
if (result.toolResults.length > 0) updateToolResults(result.toolResults);
setFileOffset(sessionFile, result.newOffset, lastModified);
return result.stats.length + result.userStats.length;
}
@@ -478,3 +485,17 @@ export async function getBehaviorDashboardStats(range?: string | null): Promise<
behaviorSeries: getBehaviorTimeSeries(cutoff),
};
}
/**
* Get the tools dashboard payload: per-tool totals, per-(tool, model)
* breakdown, and the call time series (bucketed like the model series).
*/
export async function getToolDashboardStats(range?: string | null): Promise<ToolDashboardStats> {
await initDb();
const { modelSeriesDays, modelSeriesBucketMs, cutoff } = getTimeRangeConfig(range);
return {
byTool: getToolStats(cutoff ?? undefined),
byToolModel: getToolStatsByModel(cutoff ?? undefined),
series: getToolTimeSeries(modelSeriesDays, cutoff, modelSeriesBucketMs),
};
}
+3
View File
@@ -11,6 +11,7 @@ import {
OverviewRoute,
ProjectsRoute,
RequestsRoute,
ToolsRoute,
} from "./routes";
import { RequestDrawer } from "./ui/RequestDrawer";
@@ -72,6 +73,8 @@ export default function App() {
);
case "models":
return <ModelsRoute active={isActive} range={range} refreshTrigger={refreshTrigger} />;
case "tools":
return <ToolsRoute active={isActive} range={range} refreshTrigger={refreshTrigger} />;
case "costs":
return <CostsRoute active={isActive} range={range} refreshTrigger={refreshTrigger} />;
case "behavior":
+8
View File
@@ -8,6 +8,7 @@ import type {
OverviewStats,
RequestDetails,
TimeRange,
ToolDashboardStats,
} from "./types";
const API_BASE = "/api";
@@ -92,3 +93,10 @@ export async function getGainDashboardStats(
if (project) params.set("project", project);
return fetchJson<GainDashboardStats>(`${API_BASE}/stats/gain?${params}`, { signal });
}
export async function getToolDashboardStats(
range: TimeRange = "24h",
signal?: AbortSignal,
): Promise<ToolDashboardStats> {
return fetchJson<ToolDashboardStats>(`${API_BASE}/stats/tools?range=${encodeURIComponent(range)}`, { signal });
}
+7 -1
View File
@@ -1,4 +1,4 @@
import { Activity, AlertCircle, Coins, Cpu, Folder, LayoutDashboard, Smile, TrendingUp } from "lucide-react";
import { Activity, AlertCircle, Coins, Cpu, Folder, LayoutDashboard, Smile, TrendingUp, Wrench } from "lucide-react";
import type React from "react";
export type DashboardSection =
@@ -6,6 +6,7 @@ export type DashboardSection =
| "requests"
| "errors"
| "models"
| "tools"
| "costs"
| "behavior"
| "projects"
@@ -39,6 +40,11 @@ export const routes: DashboardRoute[] = [
label: "Models",
icon: Cpu,
},
{
id: "tools",
label: "Tools",
icon: Wrench,
},
{
id: "costs",
label: "Costs",
@@ -7,6 +7,7 @@ const VALID_SECTIONS: DashboardSection[] = [
"requests",
"errors",
"models",
"tools",
"costs",
"behavior",
"projects",
@@ -8,6 +8,7 @@ import type {
FolderStats,
ModelPerformancePoint,
TimeRange,
ToolUsageStats,
} from "../types";
/** Fixed display order for the agent-token-share breakdown. */
@@ -231,3 +232,20 @@ export function buildFolderRows(folders: FolderStats[]): FolderRowView[] {
requestsPercentage: maxRequests > 0 ? (f.totalRequests / maxRequests) * 100 : 0,
}));
}
/** Table row for the Tools route: usage stats plus derived rates/shares. */
export interface ToolRowView extends ToolUsageStats {
/** errors / calls (0 for zero calls). */
errorRate: number;
/** Calls relative to the busiest tool, 0-100, for the share bar. */
callsPercentage: number;
}
export function buildToolRows(tools: ToolUsageStats[]): ToolRowView[] {
const maxCalls = tools.reduce((max, t) => Math.max(max, t.calls), 0);
return tools.map(t => ({
...t,
errorRate: t.calls > 0 ? t.errors / t.calls : 0,
callsPercentage: maxCalls > 0 ? (t.calls / maxCalls) * 100 : 0,
}));
}
@@ -0,0 +1,463 @@
import { useMemo, useState } from "react";
import { Line } from "react-chartjs-2";
import { getToolDashboardStats } from "../api";
import { CHART_THEMES, MODEL_COLORS } from "../components/chart-shared";
import { formatRangeTick, rangeMeta } from "../components/range-meta";
import { formatCompact, formatCost, formatInteger, formatPercent, formatRelativeTime } from "../data/formatters";
import { useResource } from "../data/useResource";
import { buildToolRows, type ToolRowView } from "../data/view-models";
import type { TimeRange, ToolModelStats, ToolTimeSeriesPoint, ToolUsageStats } from "../types";
import { AsyncBoundary, DataTable, Panel, StatusPill } from "../ui";
import { useSystemTheme } from "../useSystemTheme";
export interface ToolsRouteProps {
active: boolean;
range: TimeRange;
refreshTrigger: number;
}
export function ToolsRoute({ active, range, refreshTrigger }: ToolsRouteProps) {
const {
data: stats,
error,
loading,
} = useResource(["tools", range, refreshTrigger], signal => getToolDashboardStats(range, signal), {
pollMs: 30000,
enabled: active,
});
return (
<div className="stats-route-container space-y-6">
<AsyncBoundary loading={loading} error={error} data={stats} emptyText="No tool calls recorded for this range.">
{stats && (
<>
<ToolsSummaryPanel byTool={stats.byTool} />
<ToolCallsChart series={stats.series} timeRange={range} />
<ToolsTable byTool={stats.byTool} />
<ToolModelPanel byToolModel={stats.byToolModel} />
</>
)}
</AsyncBoundary>
</div>
);
}
// ---------------------------------------------------------------------------
// Summary metrics
// ---------------------------------------------------------------------------
function ToolsSummaryPanel({ byTool }: { byTool: ToolUsageStats[] }) {
const totals = useMemo(() => {
let calls = 0;
let errors = 0;
let tokens = 0;
let output = 0;
let cost = 0;
let resultChars = 0;
let argsChars = 0;
for (const t of byTool) {
calls += t.calls;
errors += t.errors;
tokens += t.totalTokensShare;
output += t.outputTokensShare;
cost += t.costShare;
resultChars += t.resultChars;
argsChars += t.argsChars;
}
return { calls, errors, tokens, output, cost, resultChars, argsChars, tools: byTool.length };
}, [byTool]);
return (
<Panel
title="Tool Usage"
subtitle="Tokens/cost are the invoking turns' real provider usage, split across each turn's tool calls"
>
<div className="stats-metric-cluster">
<div className="stats-metric-primary-grid">
<div className="stats-metric-card primary">
<div className="stats-metric-label">Tool Calls</div>
<div className="stats-metric-value">{formatInteger(totals.calls)}</div>
</div>
<div className="stats-metric-card primary">
<div className="stats-metric-label">Tools Used</div>
<div className="stats-metric-value">{formatInteger(totals.tools)}</div>
</div>
<div className="stats-metric-card primary">
<div className="stats-metric-label">Error Rate</div>
<div className="stats-metric-value">
{formatPercent(totals.calls > 0 ? totals.errors / totals.calls : 0)}
</div>
</div>
<div className="stats-metric-card primary">
<div className="stats-metric-label">Attributed Cost</div>
<div className="stats-metric-value">{formatCost(totals.cost)}</div>
</div>
</div>
<div className="stats-metric-secondary-grid">
<div className="stats-metric-card secondary">
<div className="stats-metric-label">Attributed Tokens</div>
<div className="stats-metric-value">{formatCompact(Math.round(totals.tokens))}</div>
</div>
<div className="stats-metric-card secondary">
<div className="stats-metric-label">Attributed Output</div>
<div className="stats-metric-value">{formatCompact(Math.round(totals.output))}</div>
</div>
<div className="stats-metric-card secondary">
<div className="stats-metric-label">Result Text</div>
<div className="stats-metric-value">{formatCompact(totals.resultChars)} chars</div>
</div>
<div className="stats-metric-card secondary">
<div className="stats-metric-label">Call Arguments</div>
<div className="stats-metric-value">{formatCompact(totals.argsChars)} chars</div>
</div>
</div>
</div>
</Panel>
);
}
// ---------------------------------------------------------------------------
// Calls over time (stacked by top tools)
// ---------------------------------------------------------------------------
const TOP_TOOLS = 6;
function buildToolCallSeries(points: ToolTimeSeriesPoint[]): {
buckets: number[];
tools: string[];
data: Map<number, Record<string, number>>;
} {
const totals = new Map<string, number>();
for (const p of points) totals.set(p.tool, (totals.get(p.tool) ?? 0) + p.calls);
const ranked = [...totals.entries()].sort((a, b) => b[1] - a[1]);
const top = ranked.slice(0, TOP_TOOLS).map(([tool]) => tool);
const topSet = new Set(top);
const hasOther = ranked.length > top.length;
const tools = hasOther ? [...top, "Other"] : top;
const buckets = [...new Set(points.map(p => p.timestamp))].sort((a, b) => a - b);
const data = new Map<number, Record<string, number>>();
for (const bucket of buckets) data.set(bucket, {});
for (const p of points) {
const label = topSet.has(p.tool) ? p.tool : "Other";
const row = data.get(p.timestamp);
if (row) row[label] = (row[label] ?? 0) + p.calls;
}
return { buckets, tools, data };
}
function ToolCallsChart({ series, timeRange }: { series: ToolTimeSeriesPoint[]; timeRange: TimeRange }) {
const theme = useSystemTheme();
const chartTheme = CHART_THEMES[theme];
const meta = rangeMeta(timeRange);
const chartSeries = useMemo(() => buildToolCallSeries(series), [series]);
const data = useMemo(
() => ({
labels: chartSeries.buckets.map(ts => formatRangeTick(ts, timeRange)),
datasets: chartSeries.tools.map((tool, index) => ({
label: tool,
data: chartSeries.buckets.map(bucket => chartSeries.data.get(bucket)?.[tool] ?? 0),
borderColor: MODEL_COLORS[index % MODEL_COLORS.length],
backgroundColor: `${MODEL_COLORS[index % MODEL_COLORS.length]}30`,
fill: true,
tension: 0.4,
pointRadius: 0,
pointHoverRadius: 4,
borderWidth: 2,
})),
}),
[chartSeries, timeRange],
);
const options = useMemo(
() => ({
responsive: true,
maintainAspectRatio: false,
interaction: { mode: "index" as const, intersect: false },
plugins: {
legend: {
position: "top" as const,
align: "start" as const,
labels: {
color: chartTheme.legendLabel,
usePointStyle: true,
padding: 16,
font: { size: 12 },
boxWidth: 8,
},
},
tooltip: {
backgroundColor: chartTheme.tooltipBackground,
titleColor: chartTheme.tooltipTitle,
bodyColor: chartTheme.tooltipBody,
borderColor: chartTheme.tooltipBorder,
borderWidth: 1,
padding: 12,
cornerRadius: 8,
callbacks: {
label: (context: { dataset: { label?: string }; parsed: { y: number | null } }) =>
`${context.dataset.label ?? ""}: ${formatInteger(context.parsed.y ?? 0)} calls`,
},
},
},
scales: {
x: {
stacked: true,
grid: { color: chartTheme.grid, drawBorder: false },
ticks: { color: chartTheme.tick, font: { size: 11 } },
},
y: {
stacked: true,
grid: { color: chartTheme.grid, drawBorder: false },
ticks: { color: chartTheme.tick, font: { size: 11 }, precision: 0 },
min: 0,
},
},
}),
[chartTheme],
);
return (
<Panel title="Calls Over Time" subtitle={`Tool calls over ${meta.windowLabel}, stacked by tool`}>
<div className="h-[280px]">
{chartSeries.buckets.length === 0 ? (
<div className="h-full flex items-center justify-center text-stats-muted text-sm">No data available</div>
) : (
<Line data={data} options={options} />
)}
</div>
</Panel>
);
}
// ---------------------------------------------------------------------------
// Per-tool table
// ---------------------------------------------------------------------------
function errorPillVariant(errorRate: number): "danger" | "warning" | "success" {
return errorRate > 0.1 ? "danger" : errorRate > 0 ? "warning" : "success";
}
function ToolsTable({ byTool }: { byTool: ToolUsageStats[] }) {
const rows = useMemo(() => buildToolRows(byTool), [byTool]);
const columns = useMemo(
() => [
{
key: "tool",
header: "Tool",
render: (item: ToolRowView) => (
<div className="stats-font-medium stats-text-primary font-mono truncate max-w-[280px]" title={item.tool}>
{item.tool}
</div>
),
},
{
key: "calls",
header: "Calls",
numeric: true,
render: (item: ToolRowView) => (
<div className="stats-text-right">
<div className="font-mono">{formatInteger(item.calls)}</div>
<div className="stats-progress-bar-track mt-1 ml-auto w-24 h-1">
<div
className="stats-progress-bar-fill"
data-variant="link"
style={{ width: `${item.callsPercentage}%` }}
/>
</div>
</div>
),
},
{
key: "errorRate",
header: "Error Rate",
numeric: true,
render: (item: ToolRowView) => (
<StatusPill variant={errorPillVariant(item.errorRate)}>{formatPercent(item.errorRate)}</StatusPill>
),
},
{
key: "tokens",
header: "Attr. Tokens",
numeric: true,
render: (item: ToolRowView) => (
<span className="font-mono" title="Invoking turns' total tokens, split across each turn's calls">
{formatCompact(Math.round(item.totalTokensShare))}
</span>
),
},
{
key: "cost",
header: "Attr. Cost",
numeric: true,
render: (item: ToolRowView) => <span className="font-mono">{formatCost(item.costShare)}</span>,
},
{
key: "resultChars",
header: "Result Text",
numeric: true,
render: (item: ToolRowView) => (
<span className="font-mono" title="Characters of tool-result text fed back into context">
{formatCompact(item.resultChars)}
</span>
),
},
{
key: "lastUsed",
header: "Last Used",
numeric: true,
render: (item: ToolRowView) => (
<span className="stats-text-secondary">{formatRelativeTime(item.lastUsed)}</span>
),
},
],
[],
);
const renderMobileCard = (item: ToolRowView) => (
<div className="stats-mobile-card">
<div className="stats-mobile-card-header mb-2">
<div className="stats-font-semibold stats-text-primary font-mono">{item.tool}</div>
<StatusPill variant={errorPillVariant(item.errorRate)}>{formatPercent(item.errorRate)} Err</StatusPill>
</div>
<div className="stats-mobile-card-grid">
<div>
<div className="stats-mobile-card-label">Calls</div>
<div className="stats-mobile-card-value font-mono">{formatInteger(item.calls)}</div>
</div>
<div>
<div className="stats-mobile-card-label">Attr. Tokens</div>
<div className="stats-mobile-card-value font-mono">
{formatCompact(Math.round(item.totalTokensShare))}
</div>
</div>
<div>
<div className="stats-mobile-card-label">Attr. Cost</div>
<div className="stats-mobile-card-value font-mono">{formatCost(item.costShare)}</div>
</div>
<div>
<div className="stats-mobile-card-label">Result Text</div>
<div className="stats-mobile-card-value font-mono">{formatCompact(item.resultChars)}</div>
</div>
</div>
</div>
);
return (
<Panel title="By Tool" subtitle="Usage per tool, most called first">
<DataTable
columns={columns}
data={rows}
keyExtractor={item => item.tool}
renderMobileCard={renderMobileCard}
emptyText="No tool calls recorded for this range."
/>
</Panel>
);
}
// ---------------------------------------------------------------------------
// Per-(tool, model) breakdown
// ---------------------------------------------------------------------------
function ToolModelPanel({ byToolModel }: { byToolModel: ToolModelStats[] }) {
const [tool, setTool] = useState<string | null>(null);
const tools = useMemo(() => [...new Set(byToolModel.map(row => row.tool))].sort(), [byToolModel]);
const rows = useMemo(() => {
const filtered = tool ? byToolModel.filter(row => row.tool === tool) : byToolModel;
return filtered.map(row => ({
...row,
errorRate: row.calls > 0 ? row.errors / row.calls : 0,
}));
}, [byToolModel, tool]);
const columns = useMemo(
() => [
{
key: "tool",
header: "Tool",
render: (item: ToolModelStats & { errorRate: number }) => (
<span className="stats-font-medium stats-text-primary font-mono">{item.tool}</span>
),
},
{
key: "model",
header: "Model",
render: (item: ToolModelStats & { errorRate: number }) => (
<div>
<div className="stats-text-primary">{item.model || "(unknown)"}</div>
<div className="stats-text-secondary text-xs">{item.provider}</div>
</div>
),
},
{
key: "calls",
header: "Calls",
numeric: true,
render: (item: ToolModelStats & { errorRate: number }) => (
<span className="font-mono">{formatInteger(item.calls)}</span>
),
},
{
key: "errorRate",
header: "Error Rate",
numeric: true,
render: (item: ToolModelStats & { errorRate: number }) => (
<StatusPill variant={errorPillVariant(item.errorRate)}>{formatPercent(item.errorRate)}</StatusPill>
),
},
{
key: "tokens",
header: "Attr. Tokens",
numeric: true,
render: (item: ToolModelStats & { errorRate: number }) => (
<span className="font-mono">{formatCompact(Math.round(item.totalTokensShare))}</span>
),
},
{
key: "cost",
header: "Attr. Cost",
numeric: true,
render: (item: ToolModelStats & { errorRate: number }) => (
<span className="font-mono">{formatCost(item.costShare)}</span>
),
},
],
[],
);
return (
<Panel title="By Model" subtitle="Which models call which tools">
<div className="mb-4" style={{ display: "flex", alignItems: "center", gap: "0.5rem" }}>
<span className="stats-text-secondary" style={{ fontSize: "0.875rem", whiteSpace: "nowrap" }}>
Tool
</span>
<select
className="stats-select"
value={tool ?? ""}
onChange={e => setTool(e.target.value || null)}
style={{ maxWidth: "320px", flex: 1 }}
>
<option value="">All tools</option>
{tools.map(name => (
<option key={name} value={name}>
{name}
</option>
))}
</select>
</div>
<DataTable
columns={columns}
data={rows}
keyExtractor={item => `${item.tool}::${item.model}::${item.provider}`}
emptyText="No tool calls recorded for this range."
/>
</Panel>
);
}
@@ -6,3 +6,4 @@ export * from "./ModelsRoute";
export * from "./OverviewRoute";
export * from "./ProjectsRoute";
export * from "./RequestsRoute";
export * from "./ToolsRoute";
+252
View File
@@ -19,6 +19,11 @@ import type {
ModelStats,
ModelTimeSeriesPoint,
TimeSeriesPoint,
ToolCallStats,
ToolModelStats,
ToolResultLink,
ToolTimeSeriesPoint,
ToolUsageStats,
UserMessageLink,
UserMessageStats,
} from "./types";
@@ -46,6 +51,7 @@ const USER_MESSAGE_LINKS_REPAIR_KEY = "user_message_links_v1";
const PRIORITY_PREMIUM_REQUESTS_BACKFILL_KEY = "premium_requests_priority_v1";
const AGENT_TYPE_BACKFILL_KEY = "agent_type_v1";
const FORK_DEDUPE_KEY = "fork_dedupe_v1";
const TOOL_CALLS_BACKFILL_KEY = "tool_calls_v1";
function shouldResetBackfill(value: string | undefined): boolean {
return value !== BACKFILL_COMPLETE && value !== BACKFILL_PENDING;
}
@@ -135,6 +141,27 @@ export async function initDb(): Promise<Database> {
CREATE INDEX IF NOT EXISTS idx_user_messages_timestamp ON user_messages(timestamp);
CREATE INDEX IF NOT EXISTS idx_user_messages_timestamp_model ON user_messages(timestamp, model, provider);
CREATE TABLE IF NOT EXISTS tool_calls (
id INTEGER PRIMARY KEY AUTOINCREMENT,
session_file TEXT NOT NULL,
entry_id TEXT NOT NULL,
tool_call_id TEXT NOT NULL,
folder TEXT NOT NULL,
tool_name TEXT NOT NULL,
model TEXT NOT NULL,
provider TEXT NOT NULL,
timestamp INTEGER NOT NULL,
agent_type TEXT NOT NULL DEFAULT 'main',
calls_in_turn INTEGER NOT NULL DEFAULT 1,
args_chars INTEGER NOT NULL DEFAULT 0,
result_chars INTEGER,
is_error INTEGER,
UNIQUE(session_file, tool_call_id)
);
CREATE INDEX IF NOT EXISTS idx_tool_calls_timestamp ON tool_calls(timestamp);
CREATE INDEX IF NOT EXISTS idx_tool_calls_tool_timestamp ON tool_calls(tool_name, timestamp);
CREATE TABLE IF NOT EXISTS meta (
key TEXT PRIMARY KEY,
value TEXT NOT NULL
@@ -215,6 +242,7 @@ export async function initDb(): Promise<Database> {
`);
}
backfillUserMessages(db);
backfillToolCalls(db);
repairUserMessageLinks(db);
backfillPriorityPremiumRequests(db);
backfillAgentType(db);
@@ -881,6 +909,27 @@ function backfillUserMessages(database: Database): void {
.run(USER_MESSAGES_BACKFILL_KEY, BACKFILL_PENDING);
}
/**
* One-shot wipe of `tool_calls` + `file_offsets` when the `tool_calls` table
* is introduced (or its schema version bumps), so the next sync re-parses
* every session and ingests historical tool calls. `messages` and
* `user_messages` re-inserts are idempotent, so the offset reset is safe.
* Same sentinel protocol as {@link backfillUserMessages}: the PENDING value
* written here prevents re-wiping on subsequent inits.
*/
function backfillToolCalls(database: Database): void {
const row = database.prepare("SELECT value FROM meta WHERE key = ?").get(TOOL_CALLS_BACKFILL_KEY) as
| { value: string }
| undefined;
if (!shouldResetBackfill(row?.value)) return;
database.run("DELETE FROM tool_calls");
database.run("DELETE FROM file_offsets");
database
.prepare("INSERT OR REPLACE INTO meta (key, value) VALUES (?, ?)")
.run(TOOL_CALLS_BACKFILL_KEY, BACKFILL_PENDING);
}
/**
* Reclassify pre-existing `messages` rows by agent type once, after the
* `agent_type` column is added to an older database (every prior row defaulted
@@ -1276,3 +1325,206 @@ export function getBehaviorByModel(cutoff?: number | null): BehaviorModelStats[]
lastTimestamp: row.last_timestamp ?? 0,
}));
}
/**
* Insert tool-call rows. Idempotent via UNIQUE(session_file, tool_call_id);
* the `WHERE NOT EXISTS` guard mirrors {@link insertMessageStats}: forked
* sessions deep-copy assistant entries (same `entry_id`, `timestamp`, and
* tool-call ids under a new file), so first-write-wins across the lineage
* keeps aggregates from double counting. Keyed on the assistant entry
* identity, not the call id alone — provider call ids are not a global
* namespace across unrelated sessions.
*/
export function insertToolCalls(calls: ToolCallStats[]): number {
if (!db || calls.length === 0) return 0;
const stmt = db.prepare(`
INSERT OR IGNORE INTO tool_calls (
session_file, entry_id, tool_call_id, folder, tool_name,
model, provider, timestamp, agent_type, calls_in_turn, args_chars
)
SELECT ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?
WHERE NOT EXISTS (
SELECT 1 FROM tool_calls
WHERE entry_id = ? AND timestamp = ? AND tool_call_id = ? AND session_file <> ?
)
`);
let inserted = 0;
const insert = db.transaction(() => {
for (const c of calls) {
const result = stmt.run(
c.sessionFile,
c.entryId,
c.toolCallId,
c.folder,
c.toolName,
c.model,
c.provider,
c.timestamp,
c.agentType,
c.callsInTurn,
c.argsChars,
// `WHERE NOT EXISTS` binds: skip when a different session_file
// already holds this (entry_id, timestamp, tool_call_id).
c.entryId,
c.timestamp,
c.toolCallId,
c.sessionFile,
);
if (result.changes > 0) inserted++;
}
});
insert();
return inserted;
}
/**
* Attach result size / error flag to persisted tool-call rows. Results can
* land in a later incremental sync pass than the call that produced them, so
* this is an UPDATE keyed by (session_file, tool_call_id). The `IS NULL`
* guard makes re-syncs idempotent; rows skipped by the fork guard simply
* never match.
*/
export function updateToolResults(links: ToolResultLink[]): number {
if (!db || links.length === 0) return 0;
const stmt = db.prepare(`
UPDATE tool_calls
SET result_chars = ?, is_error = ?
WHERE session_file = ? AND tool_call_id = ? AND result_chars IS NULL
`);
let updated = 0;
const apply = db.transaction(() => {
for (const link of links) {
const result = stmt.run(link.resultChars, link.isError ? 1 : 0, link.sessionFile, link.toolCallId);
updated += result.changes;
}
});
apply();
return updated;
}
/**
* Shared SELECT list for tool aggregates. Real provider usage comes from the
* invoking assistant turn (`messages` join) divided by `calls_in_turn`, so
* per-tool token/cost shares stay additive across tools.
*/
const TOOL_AGGREGATE_COLUMNS = `
COUNT(*) as calls,
SUM(CASE WHEN t.is_error = 1 THEN 1 ELSE 0 END) as errors,
SUM(t.args_chars) as args_chars,
SUM(COALESCE(t.result_chars, 0)) as result_chars,
SUM(COALESCE(m.total_tokens, 0) * 1.0 / t.calls_in_turn) as total_tokens_share,
SUM(COALESCE(m.output_tokens, 0) * 1.0 / t.calls_in_turn) as output_tokens_share,
SUM(COALESCE(m.cost_total, 0) / t.calls_in_turn) as cost_share,
MAX(t.timestamp) as last_used
`;
interface ToolAggregateRow {
tool_name: string;
model?: string;
provider?: string;
calls: number;
errors: number;
args_chars: number | null;
result_chars: number | null;
total_tokens_share: number | null;
output_tokens_share: number | null;
cost_share: number | null;
last_used: number;
}
function rowToToolUsage(row: ToolAggregateRow): ToolUsageStats {
return {
tool: row.tool_name,
calls: row.calls,
errors: row.errors,
argsChars: row.args_chars ?? 0,
resultChars: row.result_chars ?? 0,
totalTokensShare: row.total_tokens_share ?? 0,
outputTokensShare: row.output_tokens_share ?? 0,
costShare: row.cost_share ?? 0,
lastUsed: row.last_used,
};
}
/**
* Get tool usage aggregated by tool name.
*/
export function getToolStats(cutoff?: number): ToolUsageStats[] {
if (!db) return [];
const hasCutoff = cutoff !== undefined && cutoff > 0;
const stmt = db.prepare(`
SELECT t.tool_name, ${TOOL_AGGREGATE_COLUMNS}
FROM tool_calls t
LEFT JOIN messages m ON m.session_file = t.session_file AND m.entry_id = t.entry_id
${hasCutoff ? "WHERE t.timestamp >= ?" : ""}
GROUP BY t.tool_name
ORDER BY calls DESC
`);
const rows = (hasCutoff ? stmt.all(cutoff) : stmt.all()) as ToolAggregateRow[];
return rows.map(rowToToolUsage);
}
/**
* Get tool usage aggregated by (tool, model, provider).
*/
export function getToolStatsByModel(cutoff?: number): ToolModelStats[] {
if (!db) return [];
const hasCutoff = cutoff !== undefined && cutoff > 0;
const stmt = db.prepare(`
SELECT t.tool_name, t.model, t.provider, ${TOOL_AGGREGATE_COLUMNS}
FROM tool_calls t
LEFT JOIN messages m ON m.session_file = t.session_file AND m.entry_id = t.entry_id
${hasCutoff ? "WHERE t.timestamp >= ?" : ""}
GROUP BY t.tool_name, t.model, t.provider
ORDER BY calls DESC
`);
const rows = (hasCutoff ? stmt.all(cutoff) : stmt.all()) as ToolAggregateRow[];
return rows.map(row => ({
...rowToToolUsage(row),
model: row.model ?? "",
provider: row.provider ?? "",
}));
}
/**
* Get tool-call time series (one point per bucket per tool).
*/
export function getToolTimeSeries(
days = 14,
cutoff?: number | null,
bucketMs = 24 * 60 * 60 * 1000,
): ToolTimeSeriesPoint[] {
if (!db) return [];
const hasCutoff = cutoff !== null;
const seriesCutoff = hasCutoff ? (cutoff ?? Date.now() - days * 24 * 60 * 60 * 1000) : 0;
const stmt = db.prepare(`
SELECT
(timestamp / ?) * ? as bucket,
tool_name,
COUNT(*) as calls,
SUM(CASE WHEN is_error = 1 THEN 1 ELSE 0 END) as errors
FROM tool_calls
${hasCutoff ? "WHERE timestamp >= ?" : ""}
GROUP BY bucket, tool_name
ORDER BY bucket ASC
`);
const rowsRaw = hasCutoff ? stmt.all(bucketMs, bucketMs, seriesCutoff) : stmt.all(bucketMs, bucketMs);
const rows = rowsRaw as Array<{ bucket: number; tool_name: string; calls: number; errors: number }>;
return rows.map(row => ({
timestamp: row.bucket,
tool: row.tool_name,
calls: row.calls,
errors: row.errors,
}));
}
+5
View File
@@ -8,6 +8,7 @@ import { startServer } from "./server";
export {
getDashboardStats,
getToolDashboardStats,
getTotalMessageCount,
type SyncOptions,
type SyncProgress,
@@ -32,6 +33,10 @@ export type {
ModelStats,
ModelTimeSeriesPoint,
TimeSeriesPoint,
ToolDashboardStats,
ToolModelStats,
ToolTimeSeriesPoint,
ToolUsageStats,
} from "./types";
/**
+84 -2
View File
@@ -6,6 +6,7 @@ import {
getPriorityPremiumRequests,
resolveModelServiceTier,
type ServiceTierByFamily,
type ToolResultMessage,
} from "@oh-my-pi/pi-ai";
import { getSessionsDir, isEnoent } from "@oh-my-pi/pi-utils";
import type {
@@ -14,6 +15,8 @@ import type {
SessionEntry,
SessionMessageEntry,
SessionServiceTierChangeEntry,
ToolCallStats,
ToolResultLink,
UserMessageLink,
UserMessageStats,
} from "./types";
@@ -85,6 +88,14 @@ function isServiceTierChange(entry: SessionEntry): entry is SessionServiceTierCh
return entry.type === "service_tier_change";
}
/**
* Check if an entry is a tool-result message.
*/
function isToolResultMessage(entry: SessionEntry): entry is SessionMessageEntry {
if (entry.type !== "message") return false;
return (entry as SessionMessageEntry).message?.role === "toolResult";
}
/**
* Extract plain text from a user message content payload.
*/
@@ -171,6 +182,66 @@ function extractStats(
};
}
/**
* Extract one {@link ToolCallStats} per `toolCall` content block of an
* assistant message. Returns an empty array for turns without tool calls.
*/
function extractToolCalls(
sessionFile: string,
folder: string,
entry: SessionMessageEntry,
agentType: AgentType,
): ToolCallStats[] {
const msg = entry.message as AssistantMessage;
if (msg?.role !== "assistant" || !Array.isArray(msg.content)) return [];
const blocks = msg.content.filter(block => block.type === "toolCall");
if (blocks.length === 0) return [];
return blocks.map(block => {
let argsChars = 0;
try {
argsChars = JSON.stringify(block.arguments ?? {}).length;
} catch {
// Non-serializable arguments (shouldn't happen in persisted JSONL); size unknown.
}
return {
sessionFile,
entryId: entry.id,
toolCallId: block.id,
folder,
toolName: block.name,
model: msg.model,
provider: msg.provider,
timestamp: msg.timestamp,
agentType,
callsInTurn: blocks.length,
argsChars,
};
});
}
/**
* Build the result linkage for a `toolResult` entry: text characters fed back
* into context plus the error flag, keyed to the originating call.
*/
function extractToolResultLink(sessionFile: string, entry: SessionMessageEntry): ToolResultLink | null {
const msg = entry.message as ToolResultMessage;
if (msg.role !== "toolResult" || typeof msg.toolCallId !== "string" || msg.toolCallId.length === 0) return null;
let resultChars = 0;
if (Array.isArray(msg.content)) {
for (const block of msg.content) {
if (block.type === "text" && typeof block.text === "string") resultChars += block.text.length;
}
}
return {
sessionFile,
toolCallId: msg.toolCallId,
resultChars,
isError: msg.isError === true,
};
}
const LF = 0x0a;
const CR = 0x0d;
const jsonLineDecoder = new TextDecoder();
@@ -241,6 +312,8 @@ export interface ParseSessionResult {
stats: MessageStats[];
userStats: UserMessageStats[];
userLinks: UserMessageLink[];
toolCalls: ToolCallStats[];
toolResults: ToolResultLink[];
newOffset: number;
}
export async function parseSessionFile(sessionPath: string, fromOffset = 0): Promise<ParseSessionResult> {
@@ -248,7 +321,8 @@ export async function parseSessionFile(sessionPath: string, fromOffset = 0): Pro
try {
bytes = await Bun.file(sessionPath).bytes();
} catch (err) {
if (isEnoent(err)) return { stats: [], userStats: [], userLinks: [], newOffset: fromOffset };
if (isEnoent(err))
return { stats: [], userStats: [], userLinks: [], toolCalls: [], toolResults: [], newOffset: fromOffset };
throw err;
}
@@ -257,6 +331,8 @@ export async function parseSessionFile(sessionPath: string, fromOffset = 0): Pro
const stats: MessageStats[] = [];
const userStats: UserMessageStats[] = [];
const userLinks: UserMessageLink[] = [];
const toolCalls: ToolCallStats[] = [];
const toolResults: ToolResultLink[] = [];
const userByEntryId = new Map<string, UserMessageStats>();
const start = Math.max(0, Math.min(fromOffset, bytes.length));
const unprocessed = bytes.subarray(start);
@@ -278,9 +354,15 @@ export async function parseSessionFile(sessionPath: string, fromOffset = 0): Pro
}
continue;
}
if (isToolResultMessage(entry)) {
const link = extractToolResultLink(sessionPath, entry);
if (link) toolResults.push(link);
continue;
}
if (isAssistantMessage(entry)) {
const msgStats = extractStats(sessionPath, folder, entry, currentServiceTier, agentType);
if (msgStats) stats.push(msgStats);
toolCalls.push(...extractToolCalls(sessionPath, folder, entry, agentType));
// Link assistant's responding model back to the user message it answered.
const parentId = (entry as SessionMessageEntry).parentId;
if (parentId) {
@@ -303,7 +385,7 @@ export async function parseSessionFile(sessionPath: string, fromOffset = 0): Pro
}
}
return { stats, userStats, userLinks, newOffset: start + read };
return { stats, userStats, userLinks, toolCalls, toolResults, newOffset: start + read };
}
/**
+6
View File
@@ -13,6 +13,7 @@ import {
getRecentErrors,
getRecentRequests,
getRequestDetails,
getToolDashboardStats,
getTotalMessageCount,
syncAllSessions,
} from "./aggregator";
@@ -215,6 +216,11 @@ async function handleApi(req: Request): Promise<Response> {
return Response.json(stats);
}
if (path === "/api/stats/tools") {
const stats = await getToolDashboardStats(range);
return Response.json(stats);
}
if (path === "/api/stats/recent") {
const limit = url.searchParams.get("limit");
const stats = await getRecentRequests(limit ? parseInt(limit, 10) : undefined);
+52
View File
@@ -269,3 +269,55 @@ export interface GainDashboardStats {
/** All distinct projects seen in the data, for the selector. */
projects: string[];
}
/**
* Aggregated usage for a single tool over the active range.
*
* Token/cost fields are the *real* provider usage of the assistant turns that
* invoked the tool, split evenly across that turn's tool calls so the numbers
* stay additive (a turn with 3 calls contributes a third of its usage to each
* tool). Payload fields (`argsChars`/`resultChars`) are raw character counts
* of the serialized arguments and the text fed back into context — a size
* proxy, not provider-counted tokens.
*/
export interface ToolUsageStats {
/** Tool name as recorded on the tool call. */
tool: string;
/** Number of tool calls. */
calls: number;
/** Calls whose result came back with `isError`. */
errors: number;
/** Serialized tool-call argument characters. */
argsChars: number;
/** Text characters of tool results fed back into context. */
resultChars: number;
/** Total provider tokens of invoking turns, attributed per call share. */
totalTokensShare: number;
/** Output tokens of invoking turns, attributed per call share. */
outputTokensShare: number;
/** Cost (USD) of invoking turns, attributed per call share. */
costShare: number;
/** Unix ms of the most recent call in range. */
lastUsed: number;
}
/** Per-(tool, model) breakdown with the same attribution as {@link ToolUsageStats}. */
export interface ToolModelStats extends ToolUsageStats {
model: string;
provider: string;
}
/** Tool-call time-series point (one bucket per tool). */
export interface ToolTimeSeriesPoint {
timestamp: number;
tool: string;
calls: number;
errors: number;
}
/** Complete tools dashboard payload. */
export interface ToolDashboardStats {
byTool: ToolUsageStats[];
byToolModel: ToolModelStats[];
series: ToolTimeSeriesPoint[];
}
+43
View File
@@ -126,3 +126,46 @@ export interface UserMessageLink {
model: string;
provider: string;
}
/**
* One tool call extracted from an assistant message's `toolCall` content
* blocks. `callsInTurn` records how many calls that assistant turn contained
* so aggregation can split the turn's real provider usage evenly per call.
*/
export interface ToolCallStats {
/** Session file path */
sessionFile: string;
/** Assistant-message entry ID that emitted the call */
entryId: string;
/** Provider-assigned tool call ID (unique within a session) */
toolCallId: string;
/** Folder/project path (extracted from session filename) */
folder: string;
/** Tool name */
toolName: string;
/** Model that emitted the call */
model: string;
/** Provider name */
provider: string;
/** Assistant-message timestamp (Unix ms) */
timestamp: number;
/** Which agent produced the call */
agentType: AgentType;
/** Total tool calls in the same assistant turn (>= 1) */
callsInTurn: number;
/** Serialized argument characters */
argsChars: number;
}
/**
* Result linkage emitted when the parser sees a `toolResult` message entry.
* Applied as an UPDATE on the persisted tool-call row — results can land in a
* later incremental sync pass than the call that produced them.
*/
export interface ToolResultLink {
sessionFile: string;
toolCallId: string;
/** Text characters fed back into context */
resultChars: number;
isError: boolean;
}
+346
View File
@@ -0,0 +1,346 @@
import { describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as path from "node:path";
import { getToolDashboardStats, syncAllSessions } from "@oh-my-pi/omp-stats/aggregator";
import { getToolStats, getToolStatsByModel } from "@oh-my-pi/omp-stats/db";
import type { ToolUsageStats } from "@oh-my-pi/omp-stats/types";
import { getSessionsDir } from "@oh-my-pi/pi-utils";
import { installStatsTestIsolation } from "./helpers/temp-agent";
installStatsTestIsolation("@pi-stats-tool-stats-");
const FOLDER_SLUG = "--tmp--tool-stats";
const MODEL = "gpt-5.4";
const PROVIDER = "openai";
const TS1 = "2026-06-24T10:00:00.000Z";
const TS2 = "2026-06-24T10:05:00.000Z";
// Turn 1: two toolCall blocks (grep + read) sharing one provider request.
const TURN1_TOTAL_TOKENS = 100;
const TURN1_OUTPUT_TOKENS = 20;
const TURN1_COST = 0.01;
// Turn 2: a single grep toolCall owning the whole request.
const TURN2_TOTAL_TOKENS = 40;
const TURN2_OUTPUT_TOKENS = 8;
const TURN2_COST = 0.004;
const GREP_ARGS_1 = { pattern: "x" };
const READ_ARGS = { path: "/tmp/f" };
const GREP_ARGS_2 = { pattern: "yz" };
const GREP_RESULT_1 = "grep hit: src/index.ts:42";
const READ_ERROR_RESULT = "read failed: ENOENT";
const GREP_RESULT_2 = "ok";
interface ToolCallBlock {
id: string;
name: string;
arguments: Record<string, unknown>;
}
interface AssistantTurnOptions {
entryId: string;
parentId?: string | null;
timestamp: string;
toolCalls: ToolCallBlock[];
totalTokens: number;
outputTokens: number;
costTotal: number;
}
function buildAssistantEntry(opts: AssistantTurnOptions) {
return {
type: "message",
id: opts.entryId,
parentId: opts.parentId ?? null,
timestamp: opts.timestamp,
message: {
role: "assistant",
content: [
{ type: "text", text: "ok" },
...opts.toolCalls.map(call => ({
type: "toolCall",
id: call.id,
name: call.name,
arguments: call.arguments,
})),
],
api: "openai-responses",
provider: PROVIDER,
model: MODEL,
usage: {
input: 10,
output: opts.outputTokens,
cacheRead: 0,
cacheWrite: 0,
totalTokens: opts.totalTokens,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: opts.costTotal },
},
stopReason: "toolUse",
timestamp: Date.parse(opts.timestamp),
duration: 10,
ttft: 5,
},
};
}
interface ToolResultOptions {
entryId: string;
parentId: string;
timestamp: string;
toolCallId: string;
toolName: string;
text: string;
isError?: boolean;
}
function buildToolResultEntry(opts: ToolResultOptions) {
return {
type: "message",
id: opts.entryId,
parentId: opts.parentId,
timestamp: opts.timestamp,
message: {
role: "toolResult",
toolCallId: opts.toolCallId,
toolName: opts.toolName,
content: [{ type: "text", text: opts.text }],
isError: opts.isError ?? false,
timestamp: Date.parse(opts.timestamp),
},
};
}
async function writeSessionFile(
fileName: string,
header: { id: string; parentSession?: string },
entries: unknown[],
): Promise<string> {
const sessionDir = path.join(getSessionsDir(), FOLDER_SLUG);
await fs.mkdir(sessionDir, { recursive: true });
const sessionFile = path.join(sessionDir, fileName);
const headerEntry = {
type: "session",
version: 3,
id: header.id,
timestamp: new Date().toISOString(),
cwd: "/tmp/project",
...(header.parentSession ? { parentSession: header.parentSession } : {}),
};
const lines = [headerEntry, ...entries].map(entry => JSON.stringify(entry)).join("\n");
await Bun.write(sessionFile, `${lines}\n`);
return sessionFile;
}
/**
* Standard fixture: turn 1 calls grep+read (grep succeeds, read errors),
* turn 2 calls grep again and succeeds.
*/
function buildStandardEntries(): unknown[] {
return [
buildAssistantEntry({
entryId: "asst-1",
timestamp: TS1,
toolCalls: [
{ id: "call-1", name: "grep", arguments: GREP_ARGS_1 },
{ id: "call-2", name: "read", arguments: READ_ARGS },
],
totalTokens: TURN1_TOTAL_TOKENS,
outputTokens: TURN1_OUTPUT_TOKENS,
costTotal: TURN1_COST,
}),
buildToolResultEntry({
entryId: "tr-1",
parentId: "asst-1",
timestamp: TS1,
toolCallId: "call-1",
toolName: "grep",
text: GREP_RESULT_1,
}),
buildToolResultEntry({
entryId: "tr-2",
parentId: "asst-1",
timestamp: TS1,
toolCallId: "call-2",
toolName: "read",
text: READ_ERROR_RESULT,
isError: true,
}),
buildAssistantEntry({
entryId: "asst-2",
parentId: "tr-2",
timestamp: TS2,
toolCalls: [{ id: "call-3", name: "grep", arguments: GREP_ARGS_2 }],
totalTokens: TURN2_TOTAL_TOKENS,
outputTokens: TURN2_OUTPUT_TOKENS,
costTotal: TURN2_COST,
}),
buildToolResultEntry({
entryId: "tr-3",
parentId: "asst-2",
timestamp: TS2,
toolCallId: "call-3",
toolName: "grep",
text: GREP_RESULT_2,
}),
];
}
function toolRow(rows: ToolUsageStats[], tool: string): ToolUsageStats {
const row = rows.find(r => r.tool === tool);
if (!row) throw new Error(`missing aggregate row for tool "${tool}"`);
return row;
}
describe("tool usage stats pipeline", () => {
it("ingests tool calls and results end-to-end and splits turn usage across calls", async () => {
await writeSessionFile("session.jsonl", { id: "sess0001" }, buildStandardEntries());
await syncAllSessions({ workers: 1 });
const stats = getToolStats();
expect(stats).toHaveLength(2);
const grep = toolRow(stats, "grep");
expect(grep.calls).toBe(2);
expect(grep.errors).toBe(0);
expect(grep.resultChars).toBe(GREP_RESULT_1.length + GREP_RESULT_2.length);
expect(grep.argsChars).toBe(JSON.stringify(GREP_ARGS_1).length + JSON.stringify(GREP_ARGS_2).length);
// Turn 1's request is split across its two toolCall blocks; turn 2 is
// grep's alone: 100/2 + 40 = 90, 20/2 + 8 = 18, 0.01/2 + 0.004 = 0.009.
expect(grep.totalTokensShare).toBeCloseTo(90, 6);
expect(grep.outputTokensShare).toBeCloseTo(18, 6);
expect(grep.costShare).toBeCloseTo(0.009, 8);
expect(grep.lastUsed).toBe(Date.parse(TS2));
const read = toolRow(stats, "read");
expect(read.calls).toBe(1);
expect(read.errors).toBe(1);
expect(read.resultChars).toBe(READ_ERROR_RESULT.length);
expect(read.argsChars).toBe(JSON.stringify(READ_ARGS).length);
expect(read.totalTokensShare).toBeCloseTo(50, 6);
expect(read.outputTokensShare).toBeCloseTo(10, 6);
expect(read.costShare).toBeCloseTo(0.005, 8);
expect(read.lastUsed).toBe(Date.parse(TS1));
// Per-model breakdown carries the fixture model/provider with the same split.
const byModel = getToolStatsByModel();
expect(byModel).toHaveLength(2);
for (const row of byModel) {
expect(row.model).toBe(MODEL);
expect(row.provider).toBe(PROVIDER);
}
expect(toolRow(byModel, "grep").calls).toBe(2);
expect(toolRow(byModel, "grep").totalTokensShare).toBeCloseTo(90, 6);
expect(toolRow(byModel, "read").calls).toBe(1);
expect(toolRow(byModel, "read").totalTokensShare).toBeCloseTo(50, 6);
// Dashboard payload reuses the same aggregates and buckets the calls.
const dashboard = await getToolDashboardStats("all");
expect(dashboard.byTool).toEqual(stats);
expect(dashboard.series.length).toBeGreaterThan(0);
const seriesCalls = new Map<string, number>();
const seriesErrors = new Map<string, number>();
for (const point of dashboard.series) {
seriesCalls.set(point.tool, (seriesCalls.get(point.tool) ?? 0) + point.calls);
seriesErrors.set(point.tool, (seriesErrors.get(point.tool) ?? 0) + point.errors);
}
expect(seriesCalls.get("grep")).toBe(2);
expect(seriesCalls.get("read")).toBe(1);
expect(seriesErrors.get("grep")).toBe(0);
expect(seriesErrors.get("read")).toBe(1);
});
it("links a result that lands in a later sync pass without duplicating the call", async () => {
const lateResultText = "late failure output";
const sessionFile = await writeSessionFile("session.jsonl", { id: "sess0002" }, [
buildAssistantEntry({
entryId: "asst-1",
timestamp: TS1,
toolCalls: [{ id: "call-1", name: "grep", arguments: GREP_ARGS_1 }],
totalTokens: TURN2_TOTAL_TOKENS,
outputTokens: TURN2_OUTPUT_TOKENS,
costTotal: TURN2_COST,
}),
]);
await syncAllSessions({ workers: 1 });
// First pass: the call is recorded but no result has arrived yet.
const pending = getToolStats();
expect(pending).toHaveLength(1);
expect(pending[0].tool).toBe("grep");
expect(pending[0].calls).toBe(1);
expect(pending[0].resultChars).toBe(0);
expect(pending[0].errors).toBe(0);
// The toolResult is appended after the first pass consumed the file.
const lateResult = buildToolResultEntry({
entryId: "tr-1",
parentId: "asst-1",
timestamp: TS2,
toolCallId: "call-1",
toolName: "grep",
text: lateResultText,
isError: true,
});
await fs.appendFile(sessionFile, `${JSON.stringify(lateResult)}\n`);
// Guarantee the stored mtime is strictly older than the file's, so the
// incremental sync re-reads it regardless of filesystem granularity.
const bumped = new Date(Date.now() + 1_000);
await fs.utimes(sessionFile, bumped, bumped);
await syncAllSessions({ workers: 1 });
const linked = getToolStats();
expect(linked).toHaveLength(1);
expect(linked[0].tool).toBe("grep");
expect(linked[0].calls).toBe(1);
expect(linked[0].resultChars).toBe(lateResultText.length);
expect(linked[0].errors).toBe(1);
});
it("does not double-count tool calls copied into a forked session file", async () => {
const entries = buildStandardEntries();
const parentFile = await writeSessionFile("01_parent.jsonl", { id: "parent00" }, entries);
// `createBranchedSession` deep-copies the parent's entries (same entry
// ids, timestamps, and tool call ids) into the child file.
await writeSessionFile("02_fork.jsonl", { id: "fork0000", parentSession: parentFile }, entries);
await syncAllSessions({ workers: 1 });
const stats = getToolStats();
expect(stats).toHaveLength(2);
const grep = toolRow(stats, "grep");
expect(grep.calls).toBe(2);
expect(grep.errors).toBe(0);
expect(grep.resultChars).toBe(GREP_RESULT_1.length + GREP_RESULT_2.length);
expect(grep.totalTokensShare).toBeCloseTo(90, 6);
const read = toolRow(stats, "read");
expect(read.calls).toBe(1);
expect(read.errors).toBe(1);
expect(read.resultChars).toBe(READ_ERROR_RESULT.length);
expect(read.totalTokensShare).toBeCloseTo(50, 6);
});
it("keeps tool aggregates stable across repeated syncs of unchanged data", async () => {
const sessionFile = await writeSessionFile("session.jsonl", { id: "sess0003" }, buildStandardEntries());
await syncAllSessions({ workers: 1 });
const first = getToolStats();
expect(toolRow(first, "grep").calls).toBe(2);
expect(toolRow(first, "read").calls).toBe(1);
// Plain re-sync: the offset table short-circuits the unchanged file.
await syncAllSessions({ workers: 1 });
expect(getToolStats()).toEqual(first);
// Bumped mtime with identical content: the file is re-examined from its
// stored offset and must not re-ingest anything.
const bumped = new Date(Date.now() + 1_000);
await fs.utimes(sessionFile, bumped, bumped);
await syncAllSessions({ workers: 1 });
expect(getToolStats()).toEqual(first);
});
});