From 27430f1768f7748d022baed19596ff7eeb22d800 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 4 May 2026 02:29:49 +0200 Subject: [PATCH] feat(session-stats): added session-stats rolling buckets and csv export - Added timestamp and model fields to message and tool-call structs with serde defaults. - Added timestamp parsing and bucket-format helpers, including clamped percentile calculation. - Added rolling-window call grouping and CSV export via new `--by`, `--top`, `--tool`, and `--calls-csv` options. - Added `TOOL_CALLS_CSV` env fallback and help text updates describing new bucketed percentile output. --- scripts/session-stats/Cargo.toml | 1 + scripts/session-stats/src/cmd_tools.rs | 284 +++++++++++++++++++++++-- scripts/session-stats/src/common.rs | 88 ++++++++ scripts/session-stats/src/main.rs | 7 +- 4 files changed, 365 insertions(+), 15 deletions(-) diff --git a/scripts/session-stats/Cargo.toml b/scripts/session-stats/Cargo.toml index 4e873e5f3..6a0530c8e 100644 --- a/scripts/session-stats/Cargo.toml +++ b/scripts/session-stats/Cargo.toml @@ -12,6 +12,7 @@ name = "session-stats" path = "src/main.rs" [dependencies] +chrono = { version = "0.4", default-features = false, features = ["std", "clock"] } anyhow = "1" csv = "1" dirs = "5" diff --git a/scripts/session-stats/src/cmd_tools.rs b/scripts/session-stats/src/cmd_tools.rs index 688d80282..ebf70dac2 100644 --- a/scripts/session-stats/src/cmd_tools.rs +++ b/scripts/session-stats/src/cmd_tools.rs @@ -13,7 +13,13 @@ //! user TEXT — user-authored text content //! //! Output: grand totals + per-tool breakdown sorted by total (arg+res) tokens. -//! Optional CSV at `$TOOL_USAGE_CSV`. +//! Optional CSV at `$TOOL_USAGE_CSV` (per-tool totals) or +//! `--calls-csv PATH` / `$TOOL_CALLS_CSV` (one row per tool call). +//! +//! Pass `--by ` to bucket per-call data into rolling windows +//! and surface per-tool tokens/call (avg + p50 + p95) over time, so you can +//! spot regressions in tool efficiency. The buckets do not align to calendar +//! boundaries; they are pure `floor(unix_secs / N)` slices. use crate::common::*; use anyhow::{Context, Result, bail}; @@ -44,11 +50,35 @@ struct SessionTotals { struct FileResult { totals: SessionTotals, tools: HashMap, + calls: Vec, +} + +#[derive(Clone)] +struct CallRecord { + ts: i64, + tool: String, + session: String, + model: String, + arg_tok: i32, + res_tok: i32, +} + +struct PendingCall { + tool: String, + ts: i64, + arg_tok: i32, + model: String, } pub fn run(args: Vec) -> Result<()> { let mut limit: usize = 1_000; let mut workers: usize = 0; + let mut by: Option = None; + let mut top: usize = 12; + let mut tool_filter: Option = None; + let mut calls_csv: Option = std::env::var("TOOL_CALLS_CSV") + .ok() + .filter(|s| !s.is_empty()); let mut iter = args.into_iter(); while let Some(a) = iter.next() { @@ -67,12 +97,42 @@ pub fn run(args: Vec) -> Result<()> { .parse() .context("-j value")?; } + "--by" => { + let spec = iter.next().context("--by requires a bucket spec")?; + by = Some(parse_bucket(&spec)?); + } + "--top" => { + top = iter + .next() + .context("--top requires a value")? + .parse() + .context("--top value")?; + } + "--tool" => { + tool_filter = Some(iter.next().context("--tool requires a name")?); + } + "--calls-csv" => { + calls_csv = Some(iter.next().context("--calls-csv requires a path")?); + } "-h" | "--help" => { eprintln!( - "usage: session-stats tools [-n N] [-j workers]\n\ - \n\ - Aggregates per-tool token usage across the most-recent N session\n\ - jsonl files (default 1000). Tokenizer: o200k_base." +"usage: session-stats tools [-n N] [-j workers] [--by SPEC] [--top N] + [--tool NAME] [--calls-csv PATH] + +Aggregates per-tool token usage across the most-recent N session +jsonl files (default 1000). Tokenizer: o200k_base. + + --by SPEC bucket per-call data into rolling windows. SPEC is one of: + hour, day, week, month, or {{h,d,w}} (e.g. 7d, 12h, 2w). + Buckets are pure floor(unix_secs / N); they do not align to + calendar boundaries. + --top N limit per-tool series to the N most-called tools (default 12). + --tool NAME show only this tool in the bucketed series. + --calls-csv PATH + emit one CSV row per tool call (ts, session, tool, model, + arg_tok, res_tok). Env fallback: TOOL_CALLS_CSV. + +TOOL_USAGE_CSV env still emits the per-tool grand-totals CSV." ); return Ok(()); } @@ -94,6 +154,7 @@ pub fn run(args: Vec) -> Result<()> { let sessions = results.len(); let mut grand = SessionTotals::default(); let mut tools: HashMap = HashMap::new(); + let mut all_calls: Vec = Vec::new(); for r in results { grand.arg_tok += r.totals.arg_tok; grand.res_tok += r.totals.res_tok; @@ -109,12 +170,26 @@ pub fn run(args: Vec) -> Result<()> { dst.arg_tok += t.arg_tok; dst.res_tok += t.res_tok; } + all_calls.extend(r.calls); + } + + if let Some(ref name) = tool_filter { + all_calls.retain(|c| &c.tool == name); } print_grand(&grand, sessions); println!(); print_table(&tools); write_csv(&tools)?; + + if let Some(bucket_secs) = by { + println!(); + print_buckets(&all_calls, bucket_secs, top, tool_filter.as_deref()); + } + if let Some(path) = calls_csv { + write_calls_csv(&all_calls, &path)?; + eprintln!("wrote {} calls to {path}", commas(all_calls.len() as i64)); + } Ok(()) } @@ -130,9 +205,12 @@ fn process_file(path: &Path) -> Option { let mut totals = SessionTotals::default(); let mut tools: HashMap = HashMap::new(); - // Pending arg attribution: when a result arrives we credit the tool listed - // here; otherwise we fall back to message.toolName on the result event. - let mut pending: HashMap = HashMap::new(); + let mut calls: Vec = Vec::new(); + let session = session_id_from_path(path); + // Pending call records: keyed by toolCallId so the matching toolResult + // can finalize a CallRecord with both arg+res tokens. Falls back to + // message.toolName when the id is absent (legacy sessions). + let mut pending: HashMap = HashMap::new(); for line in reader.lines() { let Ok(line) = line else { continue }; @@ -165,7 +243,15 @@ fn process_file(path: &Path) -> Option { let t = tools.entry(name.clone()).or_default(); t.calls += 1; t.arg_tok += tok; - pending.insert(it.id, name); + pending.insert( + it.id, + PendingCall { + tool: name, + ts: parse_ts(&ev.timestamp), + arg_tok: clamp_i32(tok), + model: m.model.clone(), + }, + ); } "thinking" => { totals.thinking_tok += count_tokens(&it.thinking) as i64; @@ -182,12 +268,24 @@ fn process_file(path: &Path) -> Option { let tok = count_tokens(&text) as i64; totals.res_tok += tok; totals.n_results += 1; - let name = pending - .remove(&m.tool_call_id) + let pc = pending.remove(&m.tool_call_id); + let name = pc + .as_ref() + .map(|p| p.tool.clone()) .unwrap_or_else(|| normalize_tool(&m.tool_name)); - let t = tools.entry(name).or_default(); + let t = tools.entry(name.clone()).or_default(); t.results += 1; t.res_tok += tok; + if let Some(p) = pc { + calls.push(CallRecord { + ts: p.ts, + tool: name, + session: session.clone(), + model: p.model, + arg_tok: p.arg_tok, + res_tok: clamp_i32(tok), + }); + } } "user" => { for it in items { @@ -200,7 +298,7 @@ fn process_file(path: &Path) -> Option { } } - Some(FileResult { totals, tools }) + Some(FileResult { totals, tools, calls }) } use serde_json::value::RawValue; @@ -442,3 +540,163 @@ fn write_csv(tools: &HashMap) -> Result<()> { w.flush()?; Ok(()) } + +// ---- per-call helpers ---- + +fn session_id_from_path(path: &Path) -> String { + path.file_stem() + .and_then(|s| s.to_str()) + .unwrap_or("") + .to_string() +} + +fn clamp_i32(n: i64) -> i32 { + n.clamp(0, i32::MAX as i64) as i32 +} + +fn print_buckets( + calls: &[CallRecord], + bucket_secs: i64, + top: usize, + tool_filter: Option<&str>, +) { + if calls.is_empty() { + println!("(no per-call records — no toolCall/toolResult pairs found)"); + return; + } + + // Pick the tools to show: top-N by call count, ignoring records with ts==0 + // (events that lacked a parseable timestamp). + let mut per_tool: HashMap<&str, i64> = HashMap::new(); + for c in calls { + if c.ts == 0 { + continue; + } + *per_tool.entry(c.tool.as_str()).or_default() += 1; + } + let mut ranked: Vec<(&str, i64)> = per_tool.into_iter().collect(); + ranked.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(b.0))); + if tool_filter.is_none() { + ranked.truncate(top); + } + + let bucket_human = humanize_bucket(bucket_secs); + println!( + "=== per-call tokens by {} bucket (rolling, no calendar alignment) ===", + bucket_human + ); + println!( + "showing {} tool{} (sorted by call count). totals/calls match per-call records only;", + ranked.len(), + if ranked.len() == 1 { "" } else { "s" } + ); + println!( + "calls without a matching toolResult or without a parseable timestamp are skipped here." + ); + + // Group calls per (tool, bucket_id). + let mut by_bucket: HashMap<(&str, i64), Vec<&CallRecord>> = HashMap::new(); + for c in calls { + if c.ts == 0 { + continue; + } + if !ranked.iter().any(|(t, _)| *t == c.tool.as_str()) { + continue; + } + let bid = c.ts.div_euclid(bucket_secs); + by_bucket.entry((c.tool.as_str(), bid)).or_default().push(c); + } + + for (tool, total_calls) in ranked { + let mut bids: Vec = by_bucket + .keys() + .filter(|(t, _)| *t == tool) + .map(|(_, b)| *b) + .collect(); + if bids.is_empty() { + continue; + } + bids.sort(); + + println!(); + println!("=== {tool} ({} calls) ===", commas(total_calls)); + println!( + "{:<22} {:>7} {:>9} {:>9} {:>9} {:>9} {:>9}", + "window", "calls", "avg_arg", "avg_res", "avg_tot", "p50_tot", "p95_tot" + ); + let dashes = 22 + 1 + 7 + 5 * (1 + 9); + println!("{}", "-".repeat(dashes)); + + for bid in bids { + let records = by_bucket.get(&(tool, bid)).expect("present"); + let n = records.len() as i64; + let mut sum_arg = 0i64; + let mut sum_res = 0i64; + let mut totals: Vec = Vec::with_capacity(records.len()); + for r in records { + sum_arg += r.arg_tok as i64; + sum_res += r.res_tok as i64; + totals.push(r.arg_tok as i64 + r.res_tok as i64); + } + let avg_arg = sum_arg as f64 / n as f64; + let avg_res = sum_res as f64 / n as f64; + let avg_tot = avg_arg + avg_res; + let p50 = percentile(&mut totals.clone(), 50.0); + let p95 = percentile(&mut totals, 95.0); + let label = bucket_label(bid * bucket_secs, bucket_secs); + println!( + "{:<22} {:>7} {:>9} {:>9} {:>9} {:>9} {:>9}", + label, + commas(n), + commas(avg_arg.round() as i64), + commas(avg_res.round() as i64), + commas(avg_tot.round() as i64), + commas(p50.round() as i64), + commas(p95.round() as i64), + ); + } + } +} + +fn humanize_bucket(bucket_secs: i64) -> String { + if bucket_secs % (7 * 86_400) == 0 { + let n = bucket_secs / (7 * 86_400); + if n == 1 { "7d".into() } else { format!("{n}w") } + } else if bucket_secs % 86_400 == 0 { + format!("{}d", bucket_secs / 86_400) + } else if bucket_secs % 3600 == 0 { + format!("{}h", bucket_secs / 3600) + } else { + format!("{}s", bucket_secs) + } +} + +fn write_calls_csv(calls: &[CallRecord], path: &str) -> Result<()> { + let f = File::create(path).with_context(|| format!("create {path}"))?; + let mut w = csv::Writer::from_writer(f); + w.write_record([ + "ts_unix", + "ts_iso", + "session", + "tool", + "model", + "arg_tok", + "res_tok", + "total_tok", + ])?; + for c in calls { + let total = c.arg_tok as i64 + c.res_tok as i64; + w.write_record([ + &c.ts.to_string(), + &format_iso(c.ts), + &c.session, + &c.tool, + &c.model, + &c.arg_tok.to_string(), + &c.res_tok.to_string(), + &total.to_string(), + ])?; + } + w.flush()?; + Ok(()) +} diff --git a/scripts/session-stats/src/common.rs b/scripts/session-stats/src/common.rs index a5279f5a3..2e363ef00 100644 --- a/scripts/session-stats/src/common.rs +++ b/scripts/session-stats/src/common.rs @@ -20,6 +20,8 @@ pub struct RawEvent { pub kind: String, #[serde(default)] pub message: Option>, + #[serde(default)] + pub timestamp: String, } #[derive(Deserialize)] @@ -32,6 +34,8 @@ pub struct Message { pub tool_name: String, #[serde(default, rename = "toolCallId")] pub tool_call_id: String, + #[serde(default)] + pub model: String, } #[derive(Deserialize)] @@ -254,3 +258,87 @@ where .collect() }) } + +// ---- timestamp / bucket helpers ---- + +/// Parses an RFC3339 timestamp into unix seconds. Returns 0 on failure. +pub fn parse_ts(s: &str) -> i64 { + if s.is_empty() { + return 0; + } + chrono::DateTime::parse_from_rfc3339(s) + .map(|dt| dt.timestamp()) + .unwrap_or(0) +} + +/// Parses a bucket spec like "h", "day", "week", "month", "1h", "12h", "7d", "2w". +/// Returns the bucket size in seconds. +pub fn parse_bucket(spec: &str) -> anyhow::Result { + use anyhow::bail; + let s = spec.trim(); + let secs: i64 = match s { + "" => bail!("empty bucket spec"), + "hour" | "h" | "1h" => 3600, + "day" | "d" | "1d" => 86_400, + "week" | "w" | "1w" | "7d" => 7 * 86_400, + "month" | "mo" | "1mo" | "30d" => 30 * 86_400, + other => { + let bytes = other.as_bytes(); + let last = *bytes.last().unwrap(); + let unit_secs: i64 = match last { + b'h' => 3600, + b'd' => 86_400, + b'w' => 7 * 86_400, + _ => bail!("bad bucket spec {spec:?} (use h/d/w or e.g. 7d, 12h)"), + }; + let n: i64 = other[..other.len() - 1] + .parse() + .map_err(|_| anyhow::anyhow!("bad bucket count in {spec:?}"))?; + if n <= 0 { + bail!("bucket count must be > 0"); + } + n * unit_secs + } + }; + Ok(secs) +} + +/// Returns a label for the bucket starting at `start_secs`. +/// Buckets shorter than a day include the hour; multi-day buckets show start..end (exclusive). +pub fn bucket_label(start_secs: i64, bucket_secs: i64) -> String { + use chrono::TimeZone; + let start = chrono::Utc.timestamp_opt(start_secs, 0).single(); + let end = chrono::Utc.timestamp_opt(start_secs + bucket_secs - 1, 0).single(); + match (start, end) { + (Some(s), _) if bucket_secs < 86_400 => s.format("%Y-%m-%d %H:00").to_string(), + (Some(s), _) if bucket_secs == 86_400 => s.format("%Y-%m-%d").to_string(), + (Some(s), Some(e)) => format!( + "{}..{}", + s.format("%Y-%m-%d"), + e.format("%Y-%m-%d") + ), + _ => start_secs.to_string(), + } +} + +/// Formats a unix timestamp as an RFC3339 string (UTC). +pub fn format_iso(unix_secs: i64) -> String { + use chrono::TimeZone; + chrono::Utc + .timestamp_opt(unix_secs, 0) + .single() + .map(|dt| dt.format("%Y-%m-%dT%H:%M:%SZ").to_string()) + .unwrap_or_default() +} + +/// Computes a percentile over an integer slice. The slice is sorted in place. +/// `p` is 0..=100. +pub fn percentile(v: &mut [i64], p: f64) -> f64 { + if v.is_empty() { + return 0.0; + } + v.sort_unstable(); + let p = p.clamp(0.0, 100.0); + let idx = ((p / 100.0) * (v.len() as f64 - 1.0)).round() as usize; + v[idx.min(v.len() - 1)] as f64 +} diff --git a/scripts/session-stats/src/main.rs b/scripts/session-stats/src/main.rs index f75bfcc7f..1b5593d15 100644 --- a/scripts/session-stats/src/main.rs +++ b/scripts/session-stats/src/main.rs @@ -22,8 +22,11 @@ fn usage() { audit edit-tool usage across N most-recent sessions (default 1000). Optional date filters (e.g. 2026-04-28) further narrow the set. - tools [-n N] per-tool token totals across the N most-recent - session jsonl files (default 1000). + tools [-n N] [--by SPEC] [--top N] [--tool NAME] [--calls-csv PATH] + per-tool token totals across the N most-recent + session jsonl files (default 1000). With --by, + also bucket per-call tokens (avg + p50/p95) into + rolling windows so you can spot regressions. Token counting uses the o200k_base tokenizer (the GPT-4o / Claude-adjacent BPE). Walk root: ~/.omp/agent/sessions/"