feat(session-stats): added Python session-stats analyze/sync commands
- Added a Python `analyze.py` CLI with `tools`, `edits`, and `followups` session-stats subcommands. - Added `sync.py` ingestion with `~/.omp/stats.db`, migration SQL, and incremental JSONL resume logic. - Replaced `stats:run` in `package.json` with `stats:sync` and new `stats:edits`, `stats:tools`, and `stats:followups` scripts. - Removed the Rust session-stats crate files (`Cargo.toml`, `main.rs`, `common.rs`, `cmd_*.rs`) and their old command logic. - Removed `.gitignore` ignores for `scripts/session-stats/Cargo.lock` and `scripts/session-stats/edit-analysis.csv`. - Updated eval tool guidance to show Python `asyncio.run(...)` usage.
This commit is contained in:
@@ -1,32 +0,0 @@
|
||||
[package]
|
||||
name = "session-stats"
|
||||
version = "0.1.0"
|
||||
edition = "2024"
|
||||
publish = false
|
||||
|
||||
# Standalone crate: do not inherit from the parent workspace.
|
||||
[workspace]
|
||||
|
||||
[[bin]]
|
||||
name = "session-stats"
|
||||
path = "src/main.rs"
|
||||
|
||||
[dependencies]
|
||||
chrono = { version = "0.4", default-features = false, features = ["std", "clock"] }
|
||||
anyhow = "1"
|
||||
csv = "1"
|
||||
dirs = "6"
|
||||
rayon = "1.12"
|
||||
regex = "1"
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde_json = { version = "1", features = ["raw_value"] }
|
||||
tiktoken-rs = "0.11"
|
||||
walkdir = "2"
|
||||
|
||||
[profile.release]
|
||||
opt-level = 3
|
||||
lto = "thin"
|
||||
codegen-units = 1
|
||||
debug = "line-tables-only"
|
||||
split-debuginfo = "off"
|
||||
strip = "none"
|
||||
@@ -1,82 +1,70 @@
|
||||
# session-stats
|
||||
|
||||
Ad-hoc analyses over the local agent session corpus
|
||||
(`~/.omp/agent/sessions/`). Single Rust binary with subcommands.
|
||||
|
||||
## Subcommands
|
||||
|
||||
### `edits` — edit-tool reliability audit
|
||||
|
||||
Audits how agents have used the `edit` / `ast_edit` / `write` tools.
|
||||
|
||||
For each call we:
|
||||
|
||||
- detect the **argument-schema family** in use (the edit tool has shipped many
|
||||
shapes over time: `oldText/newText`, `op+pos+end+lines`, `loc+content`,
|
||||
`loc+splice/pre/post/sed`, etc.);
|
||||
- record the locator shape and verb combination (for the current schema);
|
||||
- pair the call with its `toolResult` and classify the outcome
|
||||
(`success` / `truncated` / `aborted` / `fail:anchor-stale` /
|
||||
`fail:no-match` / `fail:parse` / `fail:no-enclosing-block` / …).
|
||||
|
||||
Output: markdown-ish report on stdout plus per-call CSV at `$EDIT_ANALYSIS_CSV`
|
||||
(default `./edit-analysis.csv`).
|
||||
|
||||
### `tools` — per-tool token budget
|
||||
|
||||
Aggregates token usage across the most-recent N sessions. Buckets:
|
||||
|
||||
- `tool ARGS` — assistant tool-call argument JSON
|
||||
- `tool RESULTS` — tool result content text
|
||||
- `assistant THINKING` — assistant `thinking` blocks
|
||||
- `assistant TEXT` — assistant prose
|
||||
- `user TEXT` — user-authored text content
|
||||
|
||||
Token counting uses **`o200k_base`** via `tiktoken-rs` (the GPT-4o / GPT-5
|
||||
family BPE — well-defined offline and within ~5-10% of Claude's own counts in
|
||||
aggregate across English/code).
|
||||
|
||||
Output: grand totals + per-tool breakdown sorted by total (arg+res) tokens.
|
||||
Optional CSV at `$TOOL_USAGE_CSV`.
|
||||
|
||||
## Usage
|
||||
|
||||
```sh
|
||||
# Edit audit on the most-recent sessions.
|
||||
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits
|
||||
|
||||
# Edit audit on the 200 most-recent sessions.
|
||||
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits -n 200
|
||||
|
||||
# Edit audit on a specific date.
|
||||
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits 2026-04-28
|
||||
|
||||
# Tool token budget on the 1000 most-recent sessions.
|
||||
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 1000
|
||||
|
||||
# Tool token budget on every jsonl on disk.
|
||||
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 0
|
||||
|
||||
# Dump per-tool CSV alongside the report.
|
||||
TOOL_USAGE_CSV=tools.csv \
|
||||
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 200
|
||||
```
|
||||
|
||||
The walk root is `~/.omp/agent/sessions/`. Subagent jsonls
|
||||
(`<session-id>/<n>-<name>.jsonl`) count as their own session and are included
|
||||
in the recency window independently.
|
||||
(`~/.omp/agent/sessions/`). SQLite-backed; data is synced once into the same
|
||||
`~/.omp/stats.db` that `packages/stats` uses, then queried by short Python
|
||||
scripts.
|
||||
|
||||
## Layout
|
||||
|
||||
```
|
||||
scripts/session-stats/
|
||||
Cargo.toml
|
||||
src/
|
||||
main.rs # subcommand dispatch
|
||||
common.rs # shared JSONL shapes, walk, tokenizer, formatting helpers
|
||||
cmd_edits.rs # edits subcommand
|
||||
cmd_tools.rs # tools subcommand
|
||||
sync.py # walks ~/.omp/agent/sessions/ and populates ss_* tables
|
||||
analyze.py # tools | edits | followups subcommands over the synced db
|
||||
```
|
||||
|
||||
The crate is a standalone Cargo project (it carries its own `[workspace]`
|
||||
declaration) so it does not perturb the main workspace's lockfile.
|
||||
## One-time prep
|
||||
|
||||
```sh
|
||||
pip install tiktoken
|
||||
```
|
||||
|
||||
## Sync
|
||||
|
||||
```sh
|
||||
bun run stats:sync # incremental
|
||||
python3 scripts/session-stats/sync.py --workers 16 --full # rebuild all
|
||||
python3 scripts/session-stats/sync.py --limit 200 # newest 200 only
|
||||
```
|
||||
|
||||
The sync is incremental: per-file `mtime`, `size`, `byte_offset`, and
|
||||
`parser_version` are tracked in `ss_sessions`. Re-runs only parse new bytes
|
||||
and only re-tokenize / re-classify what changed. A bump of `EDIT_PARSER_VERSION`
|
||||
in `sync.py` invalidates `ss_edit_*` rows on next sync.
|
||||
|
||||
Tokenization is `o200k_base` (GPT-4o / GPT-5 family) via tiktoken — well
|
||||
within ~5–10% of Claude's BPE in aggregate.
|
||||
|
||||
## Schema
|
||||
|
||||
All tables are prefixed `ss_` to avoid collision with `packages/stats`.
|
||||
|
||||
|Table|Granularity|
|
||||
|---|---|
|
||||
|`ss_sessions`|one row per `.jsonl`; carries sync state + session metadata|
|
||||
|`ss_tool_calls`|one row per `toolCall` content block (`arg_json`, `arg_tokens`)|
|
||||
|`ss_tool_results`|one row per `toolResult` message (`result_text`, `result_tokens`, `is_error`)|
|
||||
|`ss_assistant_msgs`|per assistant message text + thinking blobs and token counts|
|
||||
|`ss_user_msgs`|per user message text and token count|
|
||||
|`ss_edit_calls`|per `edit` call: `success`, `warnings`, `raw_input_len`|
|
||||
|`ss_edit_sections`|per `@PATH` section in an edit; precomputed `longest_repeat_*`, `dup_anchors`|
|
||||
|
||||
Indexes on `(tool_name, timestamp)` and `(session_file, seq)` make per-tool
|
||||
aggregations and ordered session walks cheap.
|
||||
|
||||
## Analyses
|
||||
|
||||
```sh
|
||||
bun run stats:tools # per-tool token totals
|
||||
bun run stats:tools -- --by d --top 8 # bucket by day, top 8 tools each
|
||||
bun run stats:edits # edit-tool reliability audit
|
||||
bun run stats:followups # five hashline-edit detectors
|
||||
bun run stats:followups -- --max-fix 2 --min-dup 8 --show 20
|
||||
```
|
||||
|
||||
All three accept `-n N` / `--folder SUBSTR` to scope the query.
|
||||
|
||||
The Rust crate that previously lived here was retired in favor of this
|
||||
SQLite-backed flow. The schema persists everything the analyses used to
|
||||
recompute on every run (token counts, hashline parse output, success flags),
|
||||
so subsequent invocations are sub-second over the full corpus.
|
||||
|
||||
@@ -0,0 +1,861 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Analyses over the session-stats sqlite tables (`ss_*`) populated by sync.py.
|
||||
|
||||
Subcommands:
|
||||
tools — per-tool token totals (port of cmd_tools.rs)
|
||||
edits — edit-tool reliability audit (port of cmd_edits.rs)
|
||||
followups — five hashline-edit detectors (port of cmd_followups.rs)
|
||||
|
||||
Each subcommand reads from ~/.omp/stats.db. Run sync.py first.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import sqlite3
|
||||
import sys
|
||||
from collections import Counter, defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
DB_PATH = Path.home() / ".omp" / "stats.db"
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Shared helpers
|
||||
|
||||
def open_ro() -> sqlite3.Connection:
|
||||
if not DB_PATH.exists():
|
||||
sys.exit(f"db not found: {DB_PATH}. Run sync.py first.")
|
||||
conn = sqlite3.connect(f"file:{DB_PATH}?mode=ro", uri=True)
|
||||
conn.row_factory = sqlite3.Row
|
||||
return conn
|
||||
|
||||
|
||||
def commas(n: int) -> str:
|
||||
return f"{n:,}"
|
||||
|
||||
|
||||
def pct(part: int, total: int) -> float:
|
||||
return 0.0 if total == 0 else (100.0 * part / total)
|
||||
|
||||
|
||||
def truncate_line(s: str, n: int) -> str:
|
||||
s = s.replace("\n", " | ")
|
||||
if len(s) <= n:
|
||||
return s
|
||||
return s[: n - 1] + "…"
|
||||
|
||||
|
||||
def parse_bucket(spec: str) -> int:
|
||||
"""`h`,`d`,`w`,`m`,`<N>h`,`<N>d`,`<N>w` -> seconds."""
|
||||
units = {"h": 3600, "d": 86400, "w": 604800, "m": 2592000}
|
||||
if spec in units:
|
||||
return units[spec]
|
||||
if spec[-1] in units and spec[:-1].isdigit():
|
||||
return int(spec[:-1]) * units[spec[-1]]
|
||||
if spec == "hour":
|
||||
return 3600
|
||||
if spec == "day":
|
||||
return 86400
|
||||
if spec == "week":
|
||||
return 604800
|
||||
raise ValueError(f"bad --by spec: {spec}")
|
||||
|
||||
|
||||
def percentile(values: list[int], p: float) -> float:
|
||||
if not values:
|
||||
return 0.0
|
||||
s = sorted(values)
|
||||
k = (len(s) - 1) * (p / 100.0)
|
||||
lo, hi = int(k), min(int(k) + 1, len(s) - 1)
|
||||
if lo == hi:
|
||||
return float(s[lo])
|
||||
return s[lo] + (s[hi] - s[lo]) * (k - lo)
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# `tools` — per-tool token totals (cmd_tools.rs port)
|
||||
|
||||
TOOLS_AGGREGATE_SQL = """
|
||||
WITH per_tool AS (
|
||||
SELECT
|
||||
c.tool_name,
|
||||
COUNT(*) AS calls,
|
||||
IFNULL(SUM(c.arg_tokens), 0) AS arg_tok
|
||||
FROM ss_tool_calls c
|
||||
GROUP BY c.tool_name
|
||||
),
|
||||
per_tool_res AS (
|
||||
SELECT
|
||||
r.tool_name,
|
||||
COUNT(*) AS results,
|
||||
IFNULL(SUM(r.result_tokens),0) AS res_tok
|
||||
FROM ss_tool_results r
|
||||
GROUP BY r.tool_name
|
||||
)
|
||||
SELECT
|
||||
COALESCE(p.tool_name, q.tool_name) AS tool_name,
|
||||
IFNULL(p.calls, 0) AS calls,
|
||||
IFNULL(q.results, 0) AS results,
|
||||
IFNULL(p.arg_tok, 0) AS arg_tok,
|
||||
IFNULL(q.res_tok, 0) AS res_tok
|
||||
FROM per_tool p FULL OUTER JOIN per_tool_res q USING (tool_name)
|
||||
ORDER BY (IFNULL(p.arg_tok, 0) + IFNULL(q.res_tok, 0)) DESC
|
||||
"""
|
||||
|
||||
|
||||
def cmd_tools(args: argparse.Namespace) -> int:
|
||||
conn = open_ro()
|
||||
where_session, where_args = _session_filter_clause(conn, args)
|
||||
|
||||
def with_session(table_alias: str) -> tuple[str, tuple]:
|
||||
"""Returns ('AND <alias>.session_file IN (...)', params) or ('', ())."""
|
||||
if not where_args:
|
||||
return "", ()
|
||||
ph = ",".join("?" * len(where_args))
|
||||
return f"AND {table_alias}.session_file IN ({ph})", where_args
|
||||
|
||||
sf_clause_c, sf_params_c = with_session("c")
|
||||
sf_clause_r, sf_params_r = with_session("r")
|
||||
sf_clause_a, sf_params_a = with_session("a")
|
||||
sf_clause_u, sf_params_u = with_session("u")
|
||||
|
||||
# Grand totals (each subquery applies its own session filter).
|
||||
grand = conn.execute(
|
||||
f"""
|
||||
SELECT
|
||||
(SELECT IFNULL(SUM(c.arg_tokens),0) FROM ss_tool_calls c WHERE 1=1 {sf_clause_c}) AS tool_args,
|
||||
(SELECT IFNULL(SUM(r.result_tokens),0) FROM ss_tool_results r WHERE 1=1 {sf_clause_r}) AS tool_res,
|
||||
(SELECT IFNULL(SUM(a.thinking_tokens),0) FROM ss_assistant_msgs a WHERE 1=1 {sf_clause_a}) AS thinking,
|
||||
(SELECT IFNULL(SUM(a.text_tokens),0) FROM ss_assistant_msgs a WHERE 1=1 {sf_clause_a}) AS asst_text,
|
||||
(SELECT IFNULL(SUM(u.text_tokens),0) FROM ss_user_msgs u WHERE 1=1 {sf_clause_u}) AS user_text,
|
||||
(SELECT COUNT(*) FROM ss_tool_calls c WHERE 1=1 {sf_clause_c}) AS n_calls,
|
||||
(SELECT COUNT(*) FROM ss_tool_results r WHERE 1=1 {sf_clause_r}) AS n_results
|
||||
""",
|
||||
sf_params_c + sf_params_r + sf_params_a + sf_params_a + sf_params_u + sf_params_c + sf_params_r,
|
||||
).fetchone()
|
||||
n_sessions = conn.execute(
|
||||
f"SELECT COUNT(*) FROM ss_sessions {where_session}", where_args
|
||||
).fetchone()[0]
|
||||
g = grand
|
||||
|
||||
grand_total = g["tool_args"] + g["tool_res"] + g["thinking"] + g["asst_text"] + g["user_text"]
|
||||
print("=== grand totals ===")
|
||||
print(f"sessions: {commas(n_sessions)}")
|
||||
print(f"tool calls / results: {commas(g['n_calls'])} / {commas(g['n_results'])}")
|
||||
print(f"tool ARGS tokens: {commas(g['tool_args']):>14} ({pct(g['tool_args'], grand_total):5.1f}%)")
|
||||
print(f"tool RESULTS tokens: {commas(g['tool_res']):>14} ({pct(g['tool_res'], grand_total):5.1f}%)")
|
||||
print(f"assistant THINKING: {commas(g['thinking']):>14} ({pct(g['thinking'], grand_total):5.1f}%)")
|
||||
print(f"assistant TEXT: {commas(g['asst_text']):>14} ({pct(g['asst_text'], grand_total):5.1f}%)")
|
||||
print(f"user TEXT: {commas(g['user_text']):>14} ({pct(g['user_text'], grand_total):5.1f}%)")
|
||||
print(f"total: {commas(grand_total):>14}")
|
||||
|
||||
# Per-tool table.
|
||||
rows = conn.execute(
|
||||
f"""
|
||||
WITH per_tool AS (
|
||||
SELECT c.tool_name, COUNT(*) AS calls, IFNULL(SUM(c.arg_tokens),0) AS arg_tok
|
||||
FROM ss_tool_calls c WHERE 1=1 {sf_clause_c}
|
||||
GROUP BY c.tool_name
|
||||
),
|
||||
per_tool_res AS (
|
||||
SELECT r.tool_name, COUNT(*) AS results, IFNULL(SUM(r.result_tokens),0) AS res_tok
|
||||
FROM ss_tool_results r WHERE 1=1 {sf_clause_r}
|
||||
GROUP BY r.tool_name
|
||||
)
|
||||
SELECT
|
||||
COALESCE(p.tool_name, q.tool_name) AS tool_name,
|
||||
IFNULL(p.calls,0) AS calls, IFNULL(q.results,0) AS results,
|
||||
IFNULL(p.arg_tok,0) AS arg_tok, IFNULL(q.res_tok,0) AS res_tok
|
||||
FROM per_tool p FULL OUTER JOIN per_tool_res q USING (tool_name)
|
||||
ORDER BY (IFNULL(p.arg_tok,0) + IFNULL(q.res_tok,0)) DESC
|
||||
""",
|
||||
sf_params_c + sf_params_r,
|
||||
).fetchall()
|
||||
|
||||
print("\n=== per-tool tokens ===")
|
||||
print(f"{'tool':<24} {'calls':>7} {'args':>14} {'results':>14} {'total':>14}")
|
||||
print("-" * 78)
|
||||
for r in rows:
|
||||
total = r["arg_tok"] + r["res_tok"]
|
||||
print(
|
||||
f"{r['tool_name']:<24} {r['calls']:>7} "
|
||||
f"{commas(r['arg_tok']):>14} {commas(r['res_tok']):>14} {commas(total):>14}"
|
||||
)
|
||||
|
||||
if args.by:
|
||||
bucket = parse_bucket(args.by)
|
||||
_print_buckets(conn, bucket, args.top, args.tool)
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
def _session_filter_clause(conn, args) -> tuple[str, tuple]:
|
||||
"""Builds an optional WHERE clause for session_file filtering by --limit / --folder."""
|
||||
clauses, params = [], []
|
||||
if args.folder:
|
||||
clauses.append("folder LIKE ?")
|
||||
params.append(f"%{args.folder}%")
|
||||
if args.limit > 0:
|
||||
# Resolve to a concrete session_file IN (...) so other tables can reuse it.
|
||||
rows = conn.execute(
|
||||
f"""
|
||||
SELECT session_file FROM ss_sessions
|
||||
{('WHERE ' + ' AND '.join(clauses)) if clauses else ''}
|
||||
ORDER BY mtime DESC LIMIT ?
|
||||
""",
|
||||
(*params, args.limit),
|
||||
).fetchall()
|
||||
files = [r[0] for r in rows]
|
||||
if not files:
|
||||
return ("WHERE 0", ())
|
||||
placeholders = ",".join("?" * len(files))
|
||||
return (f"WHERE session_file IN ({placeholders})", tuple(files))
|
||||
if clauses:
|
||||
return ("WHERE " + " AND ".join(clauses), tuple(params))
|
||||
return ("", ())
|
||||
|
||||
|
||||
def _print_buckets(conn, bucket_secs: int, top: int, tool_filter: str | None) -> None:
|
||||
where = "WHERE c.tool_name = ?" if tool_filter else ""
|
||||
params = (tool_filter,) if tool_filter else ()
|
||||
rows = conn.execute(
|
||||
f"""
|
||||
SELECT
|
||||
(c.timestamp / 1000 / ?) * ? AS bucket,
|
||||
c.tool_name,
|
||||
COUNT(*) AS calls,
|
||||
IFNULL(SUM(c.arg_tokens), 0) AS arg_tok,
|
||||
IFNULL(SUM(r.result_tokens), 0) AS res_tok
|
||||
FROM ss_tool_calls c
|
||||
LEFT JOIN ss_tool_results r
|
||||
ON r.session_file = c.session_file AND r.call_id = c.call_id
|
||||
{where}
|
||||
GROUP BY bucket, c.tool_name
|
||||
ORDER BY bucket DESC
|
||||
""",
|
||||
(bucket_secs, bucket_secs) + params,
|
||||
).fetchall()
|
||||
|
||||
by_bucket: dict[int, list[sqlite3.Row]] = defaultdict(list)
|
||||
for r in rows:
|
||||
by_bucket[r["bucket"]].append(r)
|
||||
|
||||
print(f"\n=== per-tool tokens, bucketed by {bucket_secs}s "
|
||||
f"({'all tools' if not tool_filter else tool_filter}) ===")
|
||||
for bucket in sorted(by_bucket.keys(), reverse=True)[:20]:
|
||||
from datetime import datetime, timezone
|
||||
label = datetime.fromtimestamp(bucket, tz=timezone.utc).strftime("%Y-%m-%d %H:%MZ")
|
||||
print(f"\n[{label}]")
|
||||
ranked = sorted(by_bucket[bucket], key=lambda r: -(r["arg_tok"] + r["res_tok"]))
|
||||
for r in ranked[:top]:
|
||||
tot = r["arg_tok"] + r["res_tok"]
|
||||
print(f" {r['tool_name']:<22} {r['calls']:>5}c "
|
||||
f"args={commas(r['arg_tok']):>12} res={commas(r['res_tok']):>12} "
|
||||
f"tot={commas(tot):>12}")
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# `edits` — edit-tool reliability audit (cmd_edits.rs port)
|
||||
|
||||
_RE_TRUNCATED = re.compile(r"\[Output truncated", re.I)
|
||||
_RE_ABORTED = re.compile(
|
||||
r"Tool execution was aborted|Request was aborted|cancelled|canceled by user", re.I
|
||||
)
|
||||
_RE_SUCCESS = re.compile(
|
||||
r"^(Updated|Successfully (wrote|replaced|edited|deleted|inserted)|Replaced|"
|
||||
r"Applied|Deleted|Created|Wrote|edit applied|Edited|Inserted|OK\b)",
|
||||
re.I,
|
||||
)
|
||||
_RE_ANCHOR_STALE = re.compile(
|
||||
r"(Edit rejected:.*line[s]? .* changed since the last read|"
|
||||
r"line[s]? ha(s|ve) changed since last read)",
|
||||
re.I,
|
||||
)
|
||||
_RE_ANCHOR_MISSING = re.compile(
|
||||
r"anchor .* (not found|unknown|missing)|loc requires the full anchor", re.I
|
||||
)
|
||||
_RE_NO_ENCLOSING = re.compile(r"No enclosing .* block", re.I)
|
||||
_RE_PARSE_ERROR = re.compile(r"parse|syntax error|unbalanced|unexpected token", re.I)
|
||||
_RE_SSR_NO_MATCH = re.compile(
|
||||
r"0 matches|no replacements|no match found|No replacements made|Failed to find expected lines",
|
||||
re.I,
|
||||
)
|
||||
_RE_FILE_NOT_READ = re.compile(r"must be read first|has not been read|not yet read", re.I)
|
||||
_RE_FILE_CHANGED = re.compile(r"file has been (modified|changed) externally", re.I)
|
||||
_RE_PERM_DENIED = re.compile(r"permission denied|not allowed", re.I)
|
||||
_RE_GENERIC_REJECTED = re.compile(r"\b(rejected|failed|error|invalid)\b", re.I)
|
||||
|
||||
|
||||
def classify_edit_result(text: str) -> str:
|
||||
t = (text or "").strip()
|
||||
if not t:
|
||||
return "empty"
|
||||
first = t.split("\n", 1)[0]
|
||||
if _RE_TRUNCATED.search(first):
|
||||
return "truncated"
|
||||
if _RE_ABORTED.search(t):
|
||||
return "aborted"
|
||||
if _RE_SUCCESS.match(first):
|
||||
return "success"
|
||||
if _RE_ANCHOR_STALE.search(t):
|
||||
return "fail:anchor-stale"
|
||||
if _RE_NO_ENCLOSING.search(t):
|
||||
return "fail:no-enclosing-block"
|
||||
if _RE_ANCHOR_MISSING.search(t):
|
||||
return "fail:anchor-missing"
|
||||
if _RE_PARSE_ERROR.search(t):
|
||||
return "fail:parse"
|
||||
if _RE_SSR_NO_MATCH.search(t):
|
||||
return "fail:no-match"
|
||||
if _RE_FILE_NOT_READ.search(t):
|
||||
return "fail:file-not-read"
|
||||
if _RE_FILE_CHANGED.search(t):
|
||||
return "fail:file-changed"
|
||||
if _RE_PERM_DENIED.search(t):
|
||||
return "fail:perm"
|
||||
if _RE_GENERIC_REJECTED.search(first):
|
||||
return "fail:other"
|
||||
return "unknown"
|
||||
|
||||
|
||||
_ANCHOR_BARE = re.compile(r"^[a-zA-Z]?[0-9]+[a-z]{2}$")
|
||||
|
||||
|
||||
def _detect_edit_format(tool_name: str, args_obj: dict | None) -> str:
|
||||
if tool_name == "write":
|
||||
return "write"
|
||||
if tool_name == "ast_edit":
|
||||
return "ast_edit"
|
||||
if not isinstance(args_obj, dict):
|
||||
return "unknown"
|
||||
has = lambda k: k in args_obj # noqa: E731
|
||||
if has("oldText") and has("newText"):
|
||||
return "oldText/newText"
|
||||
if has("old_text") and has("new_text"):
|
||||
return "old_text/new_text"
|
||||
if has("diff") and has("op"):
|
||||
return "diff+op"
|
||||
if has("diff") and has("operation"):
|
||||
return "diff+operation"
|
||||
if has("diff"):
|
||||
return "diff"
|
||||
if has("replace") or has("insert"):
|
||||
return "replace/insert"
|
||||
if has("input") and isinstance(args_obj.get("input"), str):
|
||||
return "hashline"
|
||||
edits = args_obj.get("edits")
|
||||
if isinstance(edits, list) and edits and isinstance(edits[0], dict):
|
||||
first = edits[0]
|
||||
fh = lambda k: k in first # noqa: E731
|
||||
if fh("loc") and (fh("splice") or fh("pre") or fh("post") or fh("sed")):
|
||||
return "loc+splice/pre/post/sed"
|
||||
if fh("loc") and fh("content"):
|
||||
return "loc+content"
|
||||
if fh("set_line"):
|
||||
return "set_line"
|
||||
if fh("insert_after"):
|
||||
return "insert_after"
|
||||
if fh("op") and fh("pos") and fh("end") and fh("lines"):
|
||||
return "op+pos+end+lines"
|
||||
if fh("op") and fh("pos") and fh("lines"):
|
||||
return "op+pos+lines"
|
||||
if fh("op") and fh("sel") and fh("content"):
|
||||
return "op+sel+content"
|
||||
if fh("all") and (fh("new_text") or fh("old_text")):
|
||||
return "per-edit:old_text/new_text"
|
||||
return "edits[" + ",".join(sorted(first.keys())) + "]"
|
||||
return ",".join(sorted(args_obj.keys()))
|
||||
|
||||
|
||||
def _loc_shape(loc: str) -> str:
|
||||
if not loc:
|
||||
return "empty"
|
||||
if loc == "$":
|
||||
return "$file"
|
||||
if ":" in loc and not loc.startswith("$"):
|
||||
rest = loc.rsplit(":", 1)[1]
|
||||
else:
|
||||
rest = loc
|
||||
if rest.startswith("(") and rest.endswith(")"):
|
||||
return "bracket-(body)"
|
||||
if rest.startswith("[") and rest.endswith("]"):
|
||||
return "bracket-[block]"
|
||||
if rest.startswith("(") or rest.startswith("["):
|
||||
return "bracket-tail"
|
||||
if rest.endswith(")") or rest.endswith("]"):
|
||||
return "bracket-head"
|
||||
if _ANCHOR_BARE.match(rest):
|
||||
return "bare-anchor"
|
||||
return "other"
|
||||
|
||||
|
||||
def _classify_edit_args(tool_name: str, args_obj: dict | None) -> tuple[str, list[str], list[str]]:
|
||||
"""Returns (format, verbs, loc_shapes)."""
|
||||
fmt = _detect_edit_format(tool_name, args_obj)
|
||||
verbs: list[str] = []
|
||||
loc_shapes: list[str] = []
|
||||
if tool_name == "write":
|
||||
verbs.append("write")
|
||||
elif tool_name == "edit" and isinstance(args_obj, dict):
|
||||
edits = args_obj.get("edits")
|
||||
if isinstance(edits, list):
|
||||
for op in edits:
|
||||
if not isinstance(op, dict):
|
||||
continue
|
||||
loc_val = op.get("loc")
|
||||
loc_shapes.append(_loc_shape(loc_val if isinstance(loc_val, str) else ""))
|
||||
v: list[str] = []
|
||||
if op.get("splice"):
|
||||
v.append("splice")
|
||||
if op.get("pre"):
|
||||
v.append("pre")
|
||||
if op.get("post"):
|
||||
v.append("post")
|
||||
if op.get("sed"):
|
||||
v.append("sed")
|
||||
if not v:
|
||||
v.append("none")
|
||||
verbs.append("+".join(v))
|
||||
return fmt, verbs, loc_shapes
|
||||
|
||||
|
||||
def cmd_edits(args: argparse.Namespace) -> int:
|
||||
conn = open_ro()
|
||||
where_session, where_args = _session_filter_clause(conn, args)
|
||||
sf_clause = "AND c.session_file IN (" + ",".join("?" * len(where_args)) + ")" if where_args else ""
|
||||
|
||||
rows = conn.execute(
|
||||
f"""
|
||||
SELECT
|
||||
c.session_file, c.call_id, c.tool_name, c.arg_json, c.timestamp,
|
||||
r.result_text, r.is_error
|
||||
FROM ss_tool_calls c
|
||||
LEFT JOIN ss_tool_results r
|
||||
ON r.session_file = c.session_file AND r.call_id = c.call_id
|
||||
WHERE c.tool_name IN ('edit','ast_edit','write') {sf_clause}
|
||||
ORDER BY c.timestamp
|
||||
""",
|
||||
where_args,
|
||||
).fetchall()
|
||||
|
||||
if not rows:
|
||||
print("no edit-family tool calls found")
|
||||
return 0
|
||||
|
||||
by_tool: Counter = Counter()
|
||||
by_format: Counter = Counter()
|
||||
status_by_tool: dict[str, Counter] = defaultdict(Counter)
|
||||
status_by_format: dict[str, Counter] = defaultdict(Counter)
|
||||
verb_count: Counter = Counter()
|
||||
loc_count: Counter = Counter()
|
||||
fails_by_verb: dict[str, Counter] = defaultdict(Counter)
|
||||
fails_by_loc: dict[str, Counter] = defaultdict(Counter)
|
||||
sessions = set()
|
||||
failed_samples: list[tuple[str, str, list[str], list[str], str]] = []
|
||||
|
||||
for r in rows:
|
||||
sessions.add(r["session_file"])
|
||||
tool = r["tool_name"]
|
||||
try:
|
||||
args_obj = json.loads(r["arg_json"]) if r["arg_json"] else None
|
||||
except Exception:
|
||||
args_obj = None
|
||||
fmt, verbs, locs = _classify_edit_args(tool, args_obj)
|
||||
status = classify_edit_result(r["result_text"] or "")
|
||||
|
||||
by_tool[tool] += 1
|
||||
by_format[fmt] += 1
|
||||
status_by_tool[tool][status] += 1
|
||||
status_by_format[fmt][status] += 1
|
||||
for v in verbs:
|
||||
verb_count[v] += 1
|
||||
fails_by_verb[v][status] += 1
|
||||
for l in locs:
|
||||
loc_count[l] += 1
|
||||
fails_by_loc[l][status] += 1
|
||||
|
||||
if status.startswith("fail") and len(failed_samples) < 8:
|
||||
text = r["result_text"] or ""
|
||||
first = text.split("\n\n", 1)[0]
|
||||
failed_samples.append((tool, status, verbs, locs, truncate_line(first, 220)))
|
||||
|
||||
print("# Edit-tool usage")
|
||||
print(f"\nTotal tool calls: {len(rows)} (across {len(sessions)} sessions)")
|
||||
|
||||
_print_counter("\n## By tool", by_tool)
|
||||
print("\n## Outcome by tool")
|
||||
for tool in sorted(by_tool):
|
||||
print(f"\n {tool} ({by_tool[tool]} calls):")
|
||||
for st, n in sorted(status_by_tool[tool].items(), key=lambda kv: -kv[1]):
|
||||
print(f" {st:<28} {n}")
|
||||
|
||||
_print_counter("\n## edit verb distribution (per sub-edit)", verb_count)
|
||||
_print_counter("\n## edit locator shape distribution", loc_count)
|
||||
|
||||
print("\n## Failure rate per verb shape")
|
||||
for v, _ in verb_count.most_common():
|
||||
total, failed = _fail_totals(fails_by_verb[v])
|
||||
print(f" {v:<20} {failed}/{total} failed ({pct(failed, total):.0f}%)")
|
||||
|
||||
print("\n## Failure rate per locator shape")
|
||||
for l, _ in loc_count.most_common():
|
||||
total, failed = _fail_totals(fails_by_loc[l])
|
||||
print(f" {l:<20} {failed}/{total} failed ({pct(failed, total):.0f}%)")
|
||||
|
||||
_print_counter("\n## edit-tool argument-format usage", by_format)
|
||||
|
||||
print("\n## Failure rate per argument format")
|
||||
for f, _ in by_format.most_common():
|
||||
total, failed = _fail_totals(status_by_format[f])
|
||||
print(f" {f:<32} {failed:>6}/{total:<6} failed ({pct(failed, total):.0f}%)")
|
||||
|
||||
print("\n## Failure breakdown per top format")
|
||||
for f, _ in by_format.most_common(8):
|
||||
print(f"\n {f} ({by_format[f]} total)")
|
||||
for st, n in sorted(status_by_format[f].items(), key=lambda kv: -kv[1]):
|
||||
print(f" {st:<28} {n}")
|
||||
|
||||
print("\n## Sample failed edits")
|
||||
for tool, status, verbs, locs, snippet in failed_samples:
|
||||
print(f"\n— {tool} [{status}] verbs={verbs} loc={locs}\n result: {snippet}")
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
def _print_counter(header: str, c: Counter) -> None:
|
||||
print(header)
|
||||
for k, v in c.most_common():
|
||||
print(f" {k:<32} {v}")
|
||||
|
||||
|
||||
def _fail_totals(c: Counter) -> tuple[int, int]:
|
||||
total = sum(c.values())
|
||||
failed = sum(v for k, v in c.items() if k.startswith("fail"))
|
||||
return total, failed
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# `followups` — five hashline-edit detectors (cmd_followups.rs port)
|
||||
|
||||
_CLOSER_LINE_RE = re.compile(r"^\s*[\])}]+[;,]?\s*$")
|
||||
|
||||
_FIX_PATTERNS = [
|
||||
"remove-single-closer",
|
||||
"add-single-closer",
|
||||
"one-line-modify",
|
||||
"pure-delete-1",
|
||||
"pure-insert-1",
|
||||
"small-other",
|
||||
]
|
||||
_FIX_PRIORITY = {p: i for i, p in enumerate(_FIX_PATTERNS)}
|
||||
|
||||
|
||||
def _classify_fix(deleted: int, payload_lines: list[str]) -> str:
|
||||
closer = len(payload_lines) == 1 and bool(_CLOSER_LINE_RE.match(payload_lines[0]))
|
||||
if deleted == 0 and closer:
|
||||
return "add-single-closer"
|
||||
if deleted == 1 and not payload_lines:
|
||||
return "remove-single-closer"
|
||||
if deleted == 0 and len(payload_lines) == 1:
|
||||
return "pure-insert-1"
|
||||
if deleted == 1 and len(payload_lines) == 1:
|
||||
return "one-line-modify"
|
||||
if deleted >= 1 and not payload_lines:
|
||||
return "pure-delete-1"
|
||||
return "small-other"
|
||||
|
||||
|
||||
def cmd_followups(args: argparse.Namespace) -> int:
|
||||
conn = open_ro()
|
||||
where_session, where_args = _session_filter_clause(conn, args)
|
||||
sf_clause = "AND c.session_file IN (" + ",".join("?" * len(where_args)) + ")" if where_args else ""
|
||||
|
||||
# All edit calls + their sections, ordered per session.
|
||||
call_rows = conn.execute(
|
||||
f"""
|
||||
SELECT c.session_file, c.call_id, c.seq, c.timestamp, c.raw_input_len,
|
||||
c.success, c.warnings
|
||||
FROM ss_edit_calls c
|
||||
WHERE 1=1 {sf_clause}
|
||||
ORDER BY c.session_file, c.seq
|
||||
""",
|
||||
where_args,
|
||||
).fetchall()
|
||||
|
||||
section_rows = conn.execute(
|
||||
f"""
|
||||
SELECT s.session_file, s.call_id, s.seq, s.section_idx, s.target_file,
|
||||
s.op_count, s.deleted_lines, s.payload_count, s.change_size,
|
||||
s.min_line, s.max_line, s.payload_blocks,
|
||||
s.longest_repeat_len, s.longest_repeat_block_idx,
|
||||
s.longest_repeat_sample, s.dup_anchors
|
||||
FROM ss_edit_sections s
|
||||
JOIN ss_edit_calls c USING (session_file, call_id)
|
||||
WHERE 1=1 {sf_clause.replace('c.session_file', 's.session_file')}
|
||||
ORDER BY s.session_file, s.seq, s.section_idx
|
||||
""",
|
||||
where_args,
|
||||
).fetchall()
|
||||
|
||||
# Index sections by (session_file, call_id).
|
||||
sec_by_call: dict[tuple[str, str], list[sqlite3.Row]] = defaultdict(list)
|
||||
for s in section_rows:
|
||||
sec_by_call[(s["session_file"], s["call_id"])].append(s)
|
||||
|
||||
# Build per-(session, target_file) ordered list of (call_meta, section).
|
||||
by_session_file: dict[tuple[str, str], list[tuple[sqlite3.Row, sqlite3.Row]]] = defaultdict(list)
|
||||
total_successful_edits = 0
|
||||
warning_hits: list[dict] = []
|
||||
payload_dups: list[dict] = []
|
||||
anchor_dups: list[dict] = []
|
||||
|
||||
for c in call_rows:
|
||||
if c["success"] == 1:
|
||||
total_successful_edits += 1
|
||||
warns = json.loads(c["warnings"] or "[]")
|
||||
seen: list[str] = []
|
||||
for w in warns:
|
||||
if w not in seen:
|
||||
seen.append(w)
|
||||
if seen:
|
||||
files_csv = ",".join(
|
||||
s["target_file"] for s in sec_by_call.get((c["session_file"], c["call_id"]), [])
|
||||
)
|
||||
for kind in seen:
|
||||
warning_hits.append({
|
||||
"session": c["session_file"],
|
||||
"call_id": c["call_id"],
|
||||
"kind": kind,
|
||||
"files": files_csv,
|
||||
"input_len": c["raw_input_len"],
|
||||
})
|
||||
if c["success"] != 1:
|
||||
continue
|
||||
for s in sec_by_call.get((c["session_file"], c["call_id"]), []):
|
||||
by_session_file[(c["session_file"], s["target_file"])].append((c, s))
|
||||
# Payload self-dup
|
||||
if s["longest_repeat_len"] >= 4:
|
||||
payload_dups.append({
|
||||
"session": c["session_file"],
|
||||
"call_id": c["call_id"],
|
||||
"file": s["target_file"],
|
||||
"block_len": _block_len(s, s["longest_repeat_block_idx"]),
|
||||
"repeat_len": s["longest_repeat_len"],
|
||||
"sample": s["longest_repeat_sample"] or "",
|
||||
})
|
||||
# Anchor reuse
|
||||
try:
|
||||
dups = json.loads(s["dup_anchors"] or "[]")
|
||||
except Exception:
|
||||
dups = []
|
||||
for d in dups:
|
||||
anchor_dups.append({
|
||||
"session": c["session_file"],
|
||||
"call_id": c["call_id"],
|
||||
"files": d[2] if len(d) > 2 else s["target_file"],
|
||||
"anchor": d[0],
|
||||
"count": d[1],
|
||||
})
|
||||
|
||||
# (1) small-fix follow-ups + (3) same-locus re-edits.
|
||||
fix_hits: list[dict] = []
|
||||
locus_hits: list[dict] = []
|
||||
for (session, target), entries in by_session_file.items():
|
||||
for i in range(len(entries) - 1):
|
||||
ac, asec = entries[i]
|
||||
bc, bsec = entries[i + 1]
|
||||
if ac["call_id"] == bc["call_id"]:
|
||||
continue
|
||||
first_size = asec["change_size"]
|
||||
second_size = bsec["change_size"]
|
||||
gap = max(0, (bc["timestamp"] - ac["timestamp"]) // 1000)
|
||||
|
||||
# (1) small fix on big edit
|
||||
if 0 < second_size <= args.max_fix and first_size > 2:
|
||||
pl = _flatten_payload(bsec)
|
||||
pattern = _classify_fix(bsec["deleted_lines"], pl)
|
||||
summary = _render_section_summary(bsec, pl)
|
||||
fix_hits.append({
|
||||
"session": session, "file": target,
|
||||
"first_call_id": ac["call_id"], "second_call_id": bc["call_id"],
|
||||
"first_size": first_size, "first_input_len": ac["raw_input_len"],
|
||||
"second_size": second_size,
|
||||
"pattern": pattern, "second_summary": summary, "gap_secs": gap,
|
||||
})
|
||||
|
||||
# (3) same-locus re-edit (both > max-fix)
|
||||
if (
|
||||
first_size > 2 and second_size > args.max_fix
|
||||
and asec["min_line"] is not None and asec["max_line"] is not None
|
||||
and bsec["min_line"] is not None and bsec["max_line"] is not None
|
||||
):
|
||||
a_lo, a_hi = asec["min_line"], asec["max_line"]
|
||||
b_lo, b_hi = bsec["min_line"], bsec["max_line"]
|
||||
if max(a_lo, b_lo) <= min(a_hi, b_hi):
|
||||
locus_hits.append({
|
||||
"session": session, "file": target,
|
||||
"first_call_id": ac["call_id"], "second_call_id": bc["call_id"],
|
||||
"first_range": (a_lo, a_hi), "second_range": (b_lo, b_hi),
|
||||
"first_size": first_size, "second_size": second_size, "gap_secs": gap,
|
||||
})
|
||||
|
||||
if args.max_gap > 0:
|
||||
fix_hits = [h for h in fix_hits if h["gap_secs"] <= args.max_gap]
|
||||
locus_hits = [h for h in locus_hits if h["gap_secs"] <= args.max_gap]
|
||||
if args.pattern:
|
||||
fix_hits = [h for h in fix_hits if h["pattern"] == args.pattern]
|
||||
|
||||
fix_hits.sort(key=lambda h: (_FIX_PRIORITY.get(h["pattern"], 99), -h["first_input_len"]))
|
||||
locus_hits.sort(key=lambda h: -h["first_size"])
|
||||
payload_dups = [p for p in payload_dups if p["repeat_len"] >= args.min_dup]
|
||||
payload_dups.sort(key=lambda p: -p["repeat_len"])
|
||||
anchor_dups.sort(key=lambda a: -a["count"])
|
||||
|
||||
# ---- print ----
|
||||
by_pattern = Counter(h["pattern"] for h in fix_hits)
|
||||
|
||||
print("=== heuristic followup hits ===")
|
||||
print(f"total hits: {commas(len(fix_hits))}")
|
||||
print("\nby pattern:")
|
||||
for label, n in by_pattern.most_common():
|
||||
print(f" {label:<22} {n:>6}")
|
||||
|
||||
shown = min(args.show, len(fix_hits))
|
||||
print(f"\n=== top {shown} hits (by first-edit input size) ===")
|
||||
for h in fix_hits[:shown]:
|
||||
print(
|
||||
f"[{h['pattern']}] {h['file']} first={h['first_size']}L "
|
||||
f"({h['first_input_len']}B) → second={h['second_size']}L gap={h['gap_secs']}s"
|
||||
)
|
||||
print(f" session={h['session']}")
|
||||
print(f" first_call={h['first_call_id']} second_call={h['second_call_id']}")
|
||||
print(f" fix: {h['second_summary']}")
|
||||
|
||||
print("\n=== tool self-corrections ===")
|
||||
print("(emitted as warnings on otherwise-successful edits — the tool caught what the model wrote)")
|
||||
by_kind = Counter(w["kind"] for w in warning_hits)
|
||||
for kind, n in by_kind.most_common():
|
||||
suffix = (f" ({pct(n, total_successful_edits):.2f}% of "
|
||||
f"{commas(total_successful_edits)} successful edits)") if total_successful_edits else ""
|
||||
print(f" {kind:<16} {n:>6}{suffix}")
|
||||
|
||||
warn_show = min(args.show, len(warning_hits))
|
||||
if warn_show:
|
||||
print(f"\n--- top {warn_show} self-correction examples (by input size) ---")
|
||||
sorted_warns = sorted(warning_hits, key=lambda w: -w["input_len"])
|
||||
for w in sorted_warns[:warn_show]:
|
||||
print(f"[{w['kind']}] {truncate_line(w['files'], 80)} ({w['input_len']}B)")
|
||||
print(f" session={w['session']} call={w['call_id']}")
|
||||
|
||||
print("\n=== same-locus re-edits (overlapping anchor ranges, both > max-fix) ===")
|
||||
print(f"hits: {commas(len(locus_hits))}")
|
||||
locus_show = min(args.show, len(locus_hits))
|
||||
for h in locus_hits[:locus_show]:
|
||||
print(
|
||||
f"{h['file']} first={h['first_range'][0]}..{h['first_range'][1]} "
|
||||
f"({h['first_size']}L) → second={h['second_range'][0]}..{h['second_range'][1]} "
|
||||
f"({h['second_size']}L) gap={h['gap_secs']}s"
|
||||
)
|
||||
print(f" session={h['session']}")
|
||||
print(f" first_call={h['first_call_id']} second_call={h['second_call_id']}")
|
||||
|
||||
print("\n=== payload self-duplication (model pasted same N-line chunk twice in one payload) ===")
|
||||
print(
|
||||
f"hits with repeat_len >= {args.min_dup}: {commas(len(payload_dups))} "
|
||||
f"({pct(len(payload_dups), total_successful_edits):.2f}% of "
|
||||
f"{commas(total_successful_edits)} successful edits)"
|
||||
)
|
||||
for p in payload_dups[: args.show]:
|
||||
print(f"k={p['repeat_len']} block={p['block_len']}L {p['file']}")
|
||||
print(f" session={p['session']} call={p['call_id']}")
|
||||
print(f" sample: {p['sample']}")
|
||||
|
||||
print("\n=== same-anchor reused by multiple ops in one input ===")
|
||||
print(f"hits: {commas(len(anchor_dups))}")
|
||||
for a in anchor_dups[: args.show]:
|
||||
print(
|
||||
f"anchor {a['anchor']} referenced {a['count']}x "
|
||||
f"files={truncate_line(a['files'], 80)}"
|
||||
)
|
||||
print(f" session={a['session']} call={a['call_id']}")
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
def _flatten_payload(section_row: sqlite3.Row) -> list[str]:
|
||||
try:
|
||||
blocks = json.loads(section_row["payload_blocks"] or "[]")
|
||||
except Exception:
|
||||
return []
|
||||
out: list[str] = []
|
||||
for b in blocks:
|
||||
out.extend(b)
|
||||
return out
|
||||
|
||||
|
||||
def _block_len(section_row: sqlite3.Row, idx: int | None) -> int:
|
||||
if idx is None:
|
||||
return 0
|
||||
try:
|
||||
blocks = json.loads(section_row["payload_blocks"] or "[]")
|
||||
except Exception:
|
||||
return 0
|
||||
if 0 <= idx < len(blocks):
|
||||
return len(blocks[idx])
|
||||
return 0
|
||||
|
||||
|
||||
def _render_section_summary(section_row: sqlite3.Row, payload_lines: list[str]) -> str:
|
||||
bits: list[str] = []
|
||||
deleted = section_row["deleted_lines"]
|
||||
if deleted > 0:
|
||||
bits.append(f"-{deleted}")
|
||||
if payload_lines:
|
||||
bits.append(f"+{len(payload_lines)}")
|
||||
out = " / ".join(bits)
|
||||
if payload_lines:
|
||||
out += f" | {truncate_line(payload_lines[0], 80)}"
|
||||
return out
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Entry point
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(description="session-stats analyses (sqlite-backed)")
|
||||
sub = ap.add_subparsers(dest="cmd", required=True)
|
||||
|
||||
common = argparse.ArgumentParser(add_help=False)
|
||||
common.add_argument("-n", "--limit", type=int, default=0,
|
||||
help="restrict to N most-recent sessions (0 = all)")
|
||||
common.add_argument("--folder", default=None,
|
||||
help="filter sessions whose folder contains this substring")
|
||||
|
||||
ap_tools = sub.add_parser("tools", parents=[common], help="per-tool token totals")
|
||||
ap_tools.add_argument("--by", default=None,
|
||||
help="bucket per-call data: h, d, w, m, or <N>{h,d,w}")
|
||||
ap_tools.add_argument("--top", type=int, default=10, help="top tools per bucket")
|
||||
ap_tools.add_argument("--tool", default=None, help="restrict bucket view to one tool")
|
||||
ap_tools.set_defaults(func=cmd_tools)
|
||||
|
||||
ap_edits = sub.add_parser("edits", parents=[common], help="edit reliability audit")
|
||||
ap_edits.set_defaults(func=cmd_edits)
|
||||
|
||||
ap_fu = sub.add_parser("followups", parents=[common], help="hashline edit followup detectors")
|
||||
ap_fu.add_argument("--max-fix", type=int, default=2)
|
||||
ap_fu.add_argument("--max-gap", type=int, default=0, help="cap seconds between paired edits")
|
||||
ap_fu.add_argument("--min-dup", type=int, default=8, help="min payload-dup repeat length")
|
||||
ap_fu.add_argument("--pattern", default=None, help="filter (1) to a single FixPattern")
|
||||
ap_fu.add_argument("--show", type=int, default=60)
|
||||
ap_fu.set_defaults(func=cmd_followups)
|
||||
|
||||
args = ap.parse_args()
|
||||
return args.func(args)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,639 +0,0 @@
|
||||
//! `edits` subcommand — audits how agents have used the edit / ast_edit /
|
||||
//! write tools across session jsonl files.
|
||||
//!
|
||||
//! For every edit-family toolCall we record:
|
||||
//! - which argument-schema family is in use (the edit tool has shipped many
|
||||
//! shapes over time: oldText/newText, op+pos+end+lines, loc+content,
|
||||
//! loc+splice/pre/post/sed, etc.);
|
||||
//! - the locator shape and verb combination (for the current
|
||||
//! loc+splice/pre/post/sed schema);
|
||||
//! then pair the call with its toolResult and classify success / failure
|
||||
//! category (anchor-stale, no-match, parse, etc.).
|
||||
//!
|
||||
//! Output: markdown-ish report on stdout plus a per-call CSV at
|
||||
//! `$EDIT_ANALYSIS_CSV` (default `./edit-analysis.csv`).
|
||||
|
||||
use crate::common::*;
|
||||
use anyhow::{Context, Result, bail};
|
||||
use regex::Regex;
|
||||
use serde::Deserialize;
|
||||
use serde_json::Value;
|
||||
use serde_json::value::RawValue;
|
||||
use std::collections::HashMap;
|
||||
use std::fs::File;
|
||||
use std::io::{BufRead, BufReader};
|
||||
use std::path::Path;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
#[derive(Default, Clone)]
|
||||
struct EditEntry {
|
||||
file: String,
|
||||
call_id: String,
|
||||
tool_name: String,
|
||||
num_edits: i64,
|
||||
/// splice/pre/post/sed per sub-edit
|
||||
verbs: Vec<String>,
|
||||
/// bare-anchor / bracket-(body) / ...
|
||||
loc_shapes: Vec<String>,
|
||||
/// edit-tool argument schema family
|
||||
format: String,
|
||||
result_raw: String,
|
||||
/// "success" / "fail:..." / etc.
|
||||
status: String,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Default)]
|
||||
struct EditOp {
|
||||
#[serde(default)]
|
||||
loc: String,
|
||||
#[serde(default)]
|
||||
splice: Option<Box<RawValue>>,
|
||||
#[serde(default)]
|
||||
pre: Option<Box<RawValue>>,
|
||||
#[serde(default)]
|
||||
post: Option<Box<RawValue>>,
|
||||
#[serde(default)]
|
||||
sed: Option<Box<RawValue>>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Default)]
|
||||
struct EditArgs {
|
||||
#[serde(default)]
|
||||
edits: Vec<EditOp>,
|
||||
#[serde(default)]
|
||||
ops: Vec<Box<RawValue>>,
|
||||
}
|
||||
|
||||
pub fn run(args: Vec<String>) -> Result<()> {
|
||||
let mut limit: usize = 1_000;
|
||||
let mut workers: usize = 0;
|
||||
let mut date_filters: Vec<String> = Vec::new();
|
||||
|
||||
let mut iter = args.into_iter();
|
||||
while let Some(a) = iter.next() {
|
||||
match a.as_str() {
|
||||
"-n" => {
|
||||
limit = iter
|
||||
.next()
|
||||
.context("-n requires a value")?
|
||||
.parse()
|
||||
.context("-n value")?;
|
||||
}
|
||||
"-j" => {
|
||||
workers = iter
|
||||
.next()
|
||||
.context("-j requires a value")?
|
||||
.parse()
|
||||
.context("-j value")?;
|
||||
}
|
||||
"-h" | "--help" => {
|
||||
eprintln!(
|
||||
"usage: session-stats edits [-n N] [-j workers] [date prefix ...]"
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
other if other.starts_with('-') => bail!("unknown flag: {other}"),
|
||||
other => date_filters.push(other.to_string()),
|
||||
}
|
||||
}
|
||||
|
||||
let files = collect_sessions(&WalkOpts {
|
||||
date_filters,
|
||||
limit_most_recent: limit,
|
||||
})?;
|
||||
eprintln!("scanning {} session files", files.len());
|
||||
|
||||
let mut entries: Vec<EditEntry> = parallel_collect(&files, workers, 5_000, |p| {
|
||||
Some(process_file(p))
|
||||
})
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.collect();
|
||||
|
||||
// Stable ordering for sample output.
|
||||
entries.sort_by(|a, b| a.file.cmp(&b.file));
|
||||
|
||||
report_edits(&entries);
|
||||
write_csv(&entries)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn process_file(path: &Path) -> Vec<EditEntry> {
|
||||
let f = match File::open(path) {
|
||||
Ok(f) => f,
|
||||
Err(e) => {
|
||||
eprintln!("open {}: {e}", path.display());
|
||||
return Vec::new();
|
||||
}
|
||||
};
|
||||
let reader = BufReader::with_capacity(64 * 1024, f);
|
||||
let path_str = path.to_string_lossy().into_owned();
|
||||
|
||||
let mut calls: HashMap<String, EditEntry> = HashMap::new();
|
||||
let mut order: Vec<String> = Vec::new();
|
||||
|
||||
for line in reader.lines() {
|
||||
let Ok(line) = line else { continue };
|
||||
if line.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let Ok(ev) = serde_json::from_str::<RawEvent>(&line) else {
|
||||
continue;
|
||||
};
|
||||
if ev.kind != "message" {
|
||||
continue;
|
||||
}
|
||||
let Some(msg_raw) = ev.message else { continue };
|
||||
let Ok(m) = serde_json::from_str::<Message>(msg_raw.get()) else {
|
||||
continue;
|
||||
};
|
||||
let Some(content_raw) = m.content else { continue };
|
||||
let items = parse_content(&content_raw);
|
||||
|
||||
match m.role.as_str() {
|
||||
"assistant" => {
|
||||
for it in items {
|
||||
if it.kind != "toolCall" || !is_edit_tool(&it.name) {
|
||||
continue;
|
||||
}
|
||||
let raw = it.arguments.as_deref();
|
||||
let mut e = classify_edit_args(&it.name, raw);
|
||||
e.file.clone_from(&path_str);
|
||||
e.call_id.clone_from(&it.id);
|
||||
e.tool_name.clone_from(&it.name);
|
||||
let id = it.id.clone();
|
||||
calls.insert(id.clone(), e);
|
||||
order.push(id);
|
||||
}
|
||||
}
|
||||
"toolResult" => {
|
||||
if !is_edit_tool(&m.tool_name) {
|
||||
continue;
|
||||
}
|
||||
let Some(e) = calls.get_mut(&m.tool_call_id) else {
|
||||
continue;
|
||||
};
|
||||
let text = join_text(&items);
|
||||
e.status = classify_edit_result(&text);
|
||||
e.result_raw = text;
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
let mut out: Vec<EditEntry> = Vec::with_capacity(order.len());
|
||||
for id in order {
|
||||
if let Some(e) = calls.remove(&id) {
|
||||
out.push(e);
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
fn is_edit_tool(name: &str) -> bool {
|
||||
matches!(
|
||||
name.to_ascii_lowercase().as_str(),
|
||||
"edit" | "ast_edit" | "write"
|
||||
)
|
||||
}
|
||||
|
||||
// ---- argument classification ----
|
||||
|
||||
static ANCHOR_BARE: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"^[a-zA-Z]?[0-9]+[a-z]{2}$").expect("anchor_bare"));
|
||||
|
||||
fn classify_edit_args(name: &str, raw: Option<&RawValue>) -> EditEntry {
|
||||
let mut e = EditEntry {
|
||||
format: detect_edit_format(name, raw),
|
||||
..EditEntry::default()
|
||||
};
|
||||
let lname = name.to_ascii_lowercase();
|
||||
match lname.as_str() {
|
||||
"edit" => {
|
||||
let a: EditArgs = raw
|
||||
.and_then(|r| serde_json::from_str(r.get()).ok())
|
||||
.unwrap_or_default();
|
||||
e.num_edits = a.edits.len() as i64;
|
||||
for op in &a.edits {
|
||||
e.loc_shapes.push(loc_shape(&op.loc));
|
||||
let mut verbs: Vec<&str> = Vec::new();
|
||||
if !is_null_or_empty(op.splice.as_deref()) {
|
||||
verbs.push("splice");
|
||||
}
|
||||
if !is_null_or_empty(op.pre.as_deref()) {
|
||||
verbs.push("pre");
|
||||
}
|
||||
if !is_null_or_empty(op.post.as_deref()) {
|
||||
verbs.push("post");
|
||||
}
|
||||
if !is_null_or_empty(op.sed.as_deref()) {
|
||||
verbs.push("sed");
|
||||
}
|
||||
if verbs.is_empty() {
|
||||
verbs.push("none");
|
||||
}
|
||||
e.verbs.push(verbs.join("+"));
|
||||
}
|
||||
}
|
||||
"ast_edit" => {
|
||||
let a: EditArgs = raw
|
||||
.and_then(|r| serde_json::from_str(r.get()).ok())
|
||||
.unwrap_or_default();
|
||||
e.num_edits = a.ops.len() as i64;
|
||||
}
|
||||
"write" => {
|
||||
e.num_edits = 1;
|
||||
e.verbs.push("write".to_string());
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
e
|
||||
}
|
||||
|
||||
/// Looks at top-level argument keys (and the first sub-edit for the `edit`
|
||||
/// tool) to identify which schema is in use. Older sessions used many
|
||||
/// incompatible schemas.
|
||||
fn detect_edit_format(name: &str, raw: Option<&RawValue>) -> String {
|
||||
match name.to_ascii_lowercase().as_str() {
|
||||
"write" => return "write".to_string(),
|
||||
"ast_edit" => return "ast_edit".to_string(),
|
||||
_ => {}
|
||||
}
|
||||
let Some(raw) = raw else {
|
||||
return "unknown".to_string();
|
||||
};
|
||||
let top: HashMap<String, Value> = match serde_json::from_str(raw.get()) {
|
||||
Ok(v) => v,
|
||||
Err(_) => return "unknown".to_string(),
|
||||
};
|
||||
let has = |k: &str| top.contains_key(k);
|
||||
|
||||
if has("oldText") && has("newText") {
|
||||
return "oldText/newText".to_string();
|
||||
}
|
||||
if has("old_text") && has("new_text") {
|
||||
return "old_text/new_text".to_string();
|
||||
}
|
||||
if has("diff") && has("op") {
|
||||
return "diff+op".to_string();
|
||||
}
|
||||
if has("diff") && has("operation") {
|
||||
return "diff+operation".to_string();
|
||||
}
|
||||
if has("diff") {
|
||||
return "diff".to_string();
|
||||
}
|
||||
if has("replace") || has("insert") {
|
||||
return "replace/insert".to_string();
|
||||
}
|
||||
|
||||
if let Some(edits_val) = top.get("edits")
|
||||
&& let Some(arr) = edits_val.as_array()
|
||||
&& let Some(first) = arr.first().and_then(Value::as_object)
|
||||
{
|
||||
let fh = |k: &str| first.contains_key(k);
|
||||
if fh("loc") && (fh("splice") || fh("pre") || fh("post") || fh("sed")) {
|
||||
return "loc+splice/pre/post/sed".to_string();
|
||||
}
|
||||
if fh("loc") && fh("content") {
|
||||
return "loc+content".to_string();
|
||||
}
|
||||
if fh("set_line") {
|
||||
return "set_line".to_string();
|
||||
}
|
||||
if fh("insert_after") {
|
||||
return "insert_after".to_string();
|
||||
}
|
||||
if fh("op") && fh("pos") && fh("end") && fh("lines") {
|
||||
return "op+pos+end+lines".to_string();
|
||||
}
|
||||
if fh("op") && fh("pos") && fh("lines") {
|
||||
return "op+pos+lines".to_string();
|
||||
}
|
||||
if fh("op") && fh("sel") && fh("content") {
|
||||
return "op+sel+content".to_string();
|
||||
}
|
||||
if fh("all") && (fh("new_text") || fh("old_text")) {
|
||||
return "per-edit:old_text/new_text".to_string();
|
||||
}
|
||||
let mut keys: Vec<&str> = first.keys().map(String::as_str).collect();
|
||||
keys.sort_unstable();
|
||||
return format!("edits[{}]", keys.join(","));
|
||||
}
|
||||
|
||||
let mut keys: Vec<&str> = top.keys().map(String::as_str).collect();
|
||||
keys.sort_unstable();
|
||||
keys.join(",")
|
||||
}
|
||||
|
||||
fn is_null_or_empty(b: Option<&RawValue>) -> bool {
|
||||
let Some(b) = b else { return true };
|
||||
let s = b.get().trim();
|
||||
s.is_empty() || s == "null"
|
||||
}
|
||||
|
||||
fn loc_shape(loc: &str) -> String {
|
||||
if loc.is_empty() {
|
||||
return "empty".to_string();
|
||||
}
|
||||
if loc == "$" {
|
||||
return "$file".to_string();
|
||||
}
|
||||
let rest = if let Some(i) = loc.rfind(':')
|
||||
&& !loc.starts_with('$')
|
||||
{
|
||||
&loc[i + 1..]
|
||||
} else {
|
||||
loc
|
||||
};
|
||||
if rest.starts_with('(') && rest.ends_with(')') {
|
||||
return "bracket-(body)".to_string();
|
||||
}
|
||||
if rest.starts_with('[') && rest.ends_with(']') {
|
||||
return "bracket-[block]".to_string();
|
||||
}
|
||||
if rest.starts_with('(') || rest.starts_with('[') {
|
||||
return "bracket-tail".to_string();
|
||||
}
|
||||
if rest.ends_with(')') || rest.ends_with(']') {
|
||||
return "bracket-head".to_string();
|
||||
}
|
||||
if ANCHOR_BARE.is_match(rest) {
|
||||
return "bare-anchor".to_string();
|
||||
}
|
||||
"other".to_string()
|
||||
}
|
||||
|
||||
// ---- result classification ----
|
||||
|
||||
macro_rules! re {
|
||||
($pat:expr) => {
|
||||
LazyLock::new(|| Regex::new($pat).expect("compile result regex"))
|
||||
};
|
||||
}
|
||||
|
||||
static RE_ANCHOR_STALE: LazyLock<Regex> = re!(
|
||||
r"(?i)(Edit rejected:.*line[s]? .* changed since the last read|line[s]? ha(s|ve) changed since last read)"
|
||||
);
|
||||
static RE_ANCHOR_MISSING: LazyLock<Regex> =
|
||||
re!(r"(?i)anchor .* (not found|unknown|missing)|loc requires the full anchor");
|
||||
static RE_NO_ENCLOSING: LazyLock<Regex> = re!(r"(?i)No enclosing .* block");
|
||||
static RE_PARSE_ERROR: LazyLock<Regex> =
|
||||
re!(r"(?i)parse|syntax error|unbalanced|unexpected token");
|
||||
static RE_SSR_NO_MATCH: LazyLock<Regex> = re!(
|
||||
r"(?i)0 matches|no replacements|no match found|No replacements made|Failed to find expected lines"
|
||||
);
|
||||
static RE_FILE_NOT_READ: LazyLock<Regex> =
|
||||
re!(r"(?i)must be read first|has not been read|not yet read");
|
||||
static RE_FILE_CHANGED: LazyLock<Regex> =
|
||||
re!(r"(?i)file has been (modified|changed) externally");
|
||||
static RE_PERM_DENIED: LazyLock<Regex> = re!(r"(?i)permission denied|not allowed");
|
||||
static RE_GENERIC_REJECTED: LazyLock<Regex> =
|
||||
re!(r"(?i)\b(rejected|failed|error|invalid)\b");
|
||||
static RE_TRUNCATED: LazyLock<Regex> = re!(r"(?i)\[Output truncated");
|
||||
static RE_ABORTED: LazyLock<Regex> = re!(
|
||||
r"(?i)Tool execution was aborted|Request was aborted|cancelled|canceled by user"
|
||||
);
|
||||
static RE_SUCCESS: LazyLock<Regex> = re!(
|
||||
r"(?i)^(Updated|Successfully (wrote|replaced|edited|deleted|inserted)|Replaced|Applied|Deleted|Created|Wrote|edit applied|Edited|Inserted|OK\b)"
|
||||
);
|
||||
|
||||
fn classify_edit_result(text: &str) -> String {
|
||||
let t = text.trim();
|
||||
if t.is_empty() {
|
||||
return "empty".to_string();
|
||||
}
|
||||
let first = t.split_once('\n').map_or(t, |(a, _)| a);
|
||||
|
||||
if RE_TRUNCATED.is_match(first) {
|
||||
return "truncated".to_string();
|
||||
}
|
||||
if RE_ABORTED.is_match(t) {
|
||||
return "aborted".to_string();
|
||||
}
|
||||
if RE_SUCCESS.is_match(first) {
|
||||
return "success".to_string();
|
||||
}
|
||||
if RE_ANCHOR_STALE.is_match(t) {
|
||||
return "fail:anchor-stale".to_string();
|
||||
}
|
||||
if RE_NO_ENCLOSING.is_match(t) {
|
||||
return "fail:no-enclosing-block".to_string();
|
||||
}
|
||||
if RE_ANCHOR_MISSING.is_match(t) {
|
||||
return "fail:anchor-missing".to_string();
|
||||
}
|
||||
if RE_PARSE_ERROR.is_match(t) {
|
||||
return "fail:parse".to_string();
|
||||
}
|
||||
if RE_SSR_NO_MATCH.is_match(t) {
|
||||
return "fail:no-match".to_string();
|
||||
}
|
||||
if RE_FILE_NOT_READ.is_match(t) {
|
||||
return "fail:file-not-read".to_string();
|
||||
}
|
||||
if RE_FILE_CHANGED.is_match(t) {
|
||||
return "fail:file-changed".to_string();
|
||||
}
|
||||
if RE_PERM_DENIED.is_match(t) {
|
||||
return "fail:perm".to_string();
|
||||
}
|
||||
if RE_GENERIC_REJECTED.is_match(first) {
|
||||
return "fail:other".to_string();
|
||||
}
|
||||
"unknown".to_string()
|
||||
}
|
||||
|
||||
// ---- reporting ----
|
||||
|
||||
fn report_edits(entries: &[EditEntry]) {
|
||||
if entries.is_empty() {
|
||||
println!("no edit-family tool calls found in matched sessions");
|
||||
return;
|
||||
}
|
||||
|
||||
let mut by_tool: HashMap<String, i64> = HashMap::new();
|
||||
let mut by_format: HashMap<String, i64> = HashMap::new();
|
||||
let mut status_by_format: HashMap<String, HashMap<String, i64>> = HashMap::new();
|
||||
let mut status_by_tool: HashMap<String, HashMap<String, i64>> = HashMap::new();
|
||||
let mut verb_count: HashMap<String, i64> = HashMap::new();
|
||||
let mut loc_count: HashMap<String, i64> = HashMap::new();
|
||||
let mut fails_by_verb: HashMap<String, HashMap<String, i64>> = HashMap::new();
|
||||
let mut fails_by_loc: HashMap<String, HashMap<String, i64>> = HashMap::new();
|
||||
|
||||
for e in entries {
|
||||
*by_tool.entry(e.tool_name.clone()).or_insert(0) += 1;
|
||||
*status_by_tool
|
||||
.entry(e.tool_name.clone())
|
||||
.or_default()
|
||||
.entry(e.status.clone())
|
||||
.or_insert(0) += 1;
|
||||
*by_format.entry(e.format.clone()).or_insert(0) += 1;
|
||||
*status_by_format
|
||||
.entry(e.format.clone())
|
||||
.or_default()
|
||||
.entry(e.status.clone())
|
||||
.or_insert(0) += 1;
|
||||
for v in &e.verbs {
|
||||
*verb_count.entry(v.clone()).or_insert(0) += 1;
|
||||
*fails_by_verb
|
||||
.entry(v.clone())
|
||||
.or_default()
|
||||
.entry(e.status.clone())
|
||||
.or_insert(0) += 1;
|
||||
}
|
||||
for l in &e.loc_shapes {
|
||||
*loc_count.entry(l.clone()).or_insert(0) += 1;
|
||||
*fails_by_loc
|
||||
.entry(l.clone())
|
||||
.or_default()
|
||||
.entry(e.status.clone())
|
||||
.or_insert(0) += 1;
|
||||
}
|
||||
}
|
||||
|
||||
println!("# Edit-tool usage");
|
||||
println!(
|
||||
"\nTotal tool calls: {} (across {} sessions)",
|
||||
entries.len(),
|
||||
count_edit_sessions(entries)
|
||||
);
|
||||
|
||||
println!("\n## By tool");
|
||||
print_sorted(&by_tool);
|
||||
|
||||
println!("\n## Outcome by tool");
|
||||
let mut tools: Vec<&String> = by_tool.keys().collect();
|
||||
tools.sort();
|
||||
for t in tools {
|
||||
println!("\n {t} ({} calls):", by_tool[t.as_str()]);
|
||||
if let Some(m) = status_by_tool.get(t.as_str()) {
|
||||
print_sorted_indent(m, " ");
|
||||
}
|
||||
}
|
||||
|
||||
println!("\n## edit verb distribution (per sub-edit)");
|
||||
print_sorted(&verb_count);
|
||||
|
||||
println!("\n## edit locator shape distribution");
|
||||
print_sorted(&loc_count);
|
||||
|
||||
println!("\n## Failure rate per verb shape");
|
||||
for v in sorted_by_count(&verb_count) {
|
||||
let (total, failed) = fail_totals(fails_by_verb.get(v.as_str()));
|
||||
println!(
|
||||
" {v:<20} {failed}/{total} failed ({:.0}%)",
|
||||
pct(failed, total)
|
||||
);
|
||||
}
|
||||
|
||||
println!("\n## Failure rate per locator shape");
|
||||
for l in sorted_by_count(&loc_count) {
|
||||
let (total, failed) = fail_totals(fails_by_loc.get(l.as_str()));
|
||||
println!(
|
||||
" {l:<20} {failed}/{total} failed ({:.0}%)",
|
||||
pct(failed, total)
|
||||
);
|
||||
}
|
||||
|
||||
println!("\n## edit-tool argument-format usage");
|
||||
print_sorted(&by_format);
|
||||
|
||||
println!("\n## Failure rate per argument format");
|
||||
for fname in sorted_by_count(&by_format) {
|
||||
let (total, failed) = fail_totals(status_by_format.get(fname.as_str()));
|
||||
println!(
|
||||
" {fname:<32} {failed:>6}/{total:<6} failed ({:.0}%)",
|
||||
pct(failed, total)
|
||||
);
|
||||
}
|
||||
|
||||
println!("\n## Failure breakdown per top format");
|
||||
for fname in sorted_by_count(&by_format).into_iter().take(8) {
|
||||
println!("\n {fname} ({} total)", by_format[fname.as_str()]);
|
||||
if let Some(m) = status_by_format.get(fname.as_str()) {
|
||||
print_sorted_indent(m, " ");
|
||||
}
|
||||
}
|
||||
|
||||
println!("\n## Sample failed edits");
|
||||
let mut shown = 0;
|
||||
for e in entries {
|
||||
if !e.status.starts_with("fail") {
|
||||
continue;
|
||||
}
|
||||
let first = e
|
||||
.result_raw
|
||||
.split_once("\n\n")
|
||||
.map_or(e.result_raw.as_str(), |(a, _)| a);
|
||||
println!(
|
||||
"\n— {} [{}] verbs={:?} loc={:?}\n result: {}",
|
||||
e.tool_name,
|
||||
e.status,
|
||||
e.verbs,
|
||||
e.loc_shapes,
|
||||
truncate_line(first, 220)
|
||||
);
|
||||
shown += 1;
|
||||
if shown >= 8 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn fail_totals(m: Option<&HashMap<String, i64>>) -> (i64, i64) {
|
||||
let Some(m) = m else { return (0, 0) };
|
||||
let mut total = 0i64;
|
||||
let mut failed = 0i64;
|
||||
for (status, n) in m {
|
||||
total += n;
|
||||
if status.starts_with("fail") {
|
||||
failed += n;
|
||||
}
|
||||
}
|
||||
(total, failed)
|
||||
}
|
||||
|
||||
fn count_edit_sessions(entries: &[EditEntry]) -> usize {
|
||||
let mut s: std::collections::HashSet<&str> = std::collections::HashSet::new();
|
||||
for e in entries {
|
||||
s.insert(&e.file);
|
||||
}
|
||||
s.len()
|
||||
}
|
||||
|
||||
fn write_csv(entries: &[EditEntry]) -> Result<()> {
|
||||
let path = std::env::var("EDIT_ANALYSIS_CSV").unwrap_or_else(|_| "edit-analysis.csv".to_string());
|
||||
let f = File::create(&path).with_context(|| format!("create {path}"))?;
|
||||
let mut w = csv::Writer::from_writer(f);
|
||||
w.write_record([
|
||||
"session",
|
||||
"tool",
|
||||
"status",
|
||||
"num_edits",
|
||||
"verbs",
|
||||
"loc_shapes",
|
||||
"result_first_line",
|
||||
])?;
|
||||
for e in entries {
|
||||
let first = e
|
||||
.result_raw
|
||||
.split_once('\n')
|
||||
.map_or(e.result_raw.as_str(), |(a, _)| a);
|
||||
let session = Path::new(&e.file)
|
||||
.file_name()
|
||||
.and_then(|s| s.to_str())
|
||||
.unwrap_or(&e.file);
|
||||
w.write_record([
|
||||
session,
|
||||
&e.tool_name,
|
||||
&e.status,
|
||||
&e.num_edits.to_string(),
|
||||
&e.verbs.join(","),
|
||||
&e.loc_shapes.join(","),
|
||||
&truncate_line(first, 200),
|
||||
])?;
|
||||
}
|
||||
w.flush()?;
|
||||
Ok(())
|
||||
}
|
||||
@@ -1,972 +0,0 @@
|
||||
//! `followups` subcommand — heuristic for buggy hashline edits.
|
||||
//!
|
||||
//! Premise: when a hashline `edit` introduces a duplicate `}`, drops a line,
|
||||
//! or otherwise breaks adjacent context, the next thing the agent does is
|
||||
//! usually a tiny corrective edit on the same file. So we flag, per session,
|
||||
//! pairs of edits where:
|
||||
//!
|
||||
//! - both are hashline `edit` toolCalls (input begins with a `@PATH`),
|
||||
//! - both target the same file,
|
||||
//! - the FIRST edit succeeded,
|
||||
//! - the SECOND edit is small (<= --max-fix lines changed, default 2),
|
||||
//! - the second edit is the next edit (in the same session) on that file,
|
||||
//! and there is no other edit in between (so retries-after-failure are
|
||||
//! not counted).
|
||||
//!
|
||||
//! For each hit we classify the small follow-up's pattern: adds a single
|
||||
//! closing brace/paren/bracket, removes one (likely duplicate), pure single
|
||||
//! insert, pure single delete, or one-line tweak.
|
||||
|
||||
use crate::common::*;
|
||||
use anyhow::{Context, Result, bail};
|
||||
use regex::Regex;
|
||||
use serde::Deserialize;
|
||||
use std::collections::HashMap;
|
||||
use std::fs::File;
|
||||
use std::io::{BufRead, BufReader};
|
||||
use std::path::Path;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
// ---- parsed shapes --------------------------------------------------------
|
||||
|
||||
#[derive(Clone, Default)]
|
||||
struct EditSection {
|
||||
file: String,
|
||||
payload_lines: Vec<String>,
|
||||
/// Payload lines grouped by op. Each inner vec is one op's payload (`+`,
|
||||
/// `<`, or `=` followed by `~TEXT` lines), in source order. Used to detect
|
||||
/// intra-block self-similarity (model duplicating an N-line chunk).
|
||||
payload_blocks: Vec<Vec<String>>,
|
||||
/// Raw anchor refs (`123ab`) from ops in this section, in source order.
|
||||
/// Used to detect the same anchor being referenced by multiple ops in the
|
||||
/// same call.
|
||||
op_anchors: Vec<String>,
|
||||
/// Total line count covered by `- A..B` and `= A..B` ranges (range size).
|
||||
deleted_lines: i64,
|
||||
op_count: i64,
|
||||
/// Min/max anchor line touched by ops in this section. `None` for a
|
||||
/// section whose only ops target BOF/EOF (no concrete line).
|
||||
min_line: Option<i64>,
|
||||
max_line: Option<i64>,
|
||||
}
|
||||
|
||||
impl EditSection {
|
||||
fn change_size(&self) -> i64 {
|
||||
self.payload_lines.len() as i64 + self.deleted_lines
|
||||
}
|
||||
fn touch(&mut self, line: i64) {
|
||||
self.min_line = Some(self.min_line.map_or(line, |m| m.min(line)));
|
||||
self.max_line = Some(self.max_line.map_or(line, |m| m.max(line)));
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
struct EditCall {
|
||||
ts: i64,
|
||||
call_id: String,
|
||||
sections: Vec<EditSection>,
|
||||
success: bool,
|
||||
raw_input_len: usize,
|
||||
/// Tool-emitted warnings extracted from a successful result text. Each
|
||||
/// entry is one of `auto-rebased` / `auto-absorbed` / `auto-dropped`.
|
||||
warnings: Vec<&'static str>,
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Default)]
|
||||
struct EditArgs {
|
||||
#[serde(default)]
|
||||
input: Option<String>,
|
||||
}
|
||||
|
||||
// ---- per-section parsing --------------------------------------------------
|
||||
|
||||
// Op headers. We parse loosely: anything that doesn't match a known op or a
|
||||
// `~` payload line is ignored (blank line, comment, etc.).
|
||||
//
|
||||
// Range sizes: `LINEhash..LINEhash` -> end_line - start_line + 1 (clamped to
|
||||
// >= 1). For single-anchor `- A` we count 1.
|
||||
static RANGE_RE: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$").expect("RANGE_RE"));
|
||||
static SINGLE_ANCHOR_RE: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"^\s*(\d+)[a-z*]+\s*$").expect("SINGLE_ANCHOR_RE"));
|
||||
|
||||
/// Returns (range_size, optional (start_line, end_line)). Size is at least 1.
|
||||
fn parse_range(raw: &str) -> (i64, Option<(i64, i64)>) {
|
||||
let Some(caps) = RANGE_RE.captures(raw.trim()) else {
|
||||
return (1, None);
|
||||
};
|
||||
let start: i64 = caps.get(1).and_then(|m| m.as_str().parse().ok()).unwrap_or(0);
|
||||
let end: i64 = caps
|
||||
.get(2)
|
||||
.and_then(|m| m.as_str().parse().ok())
|
||||
.unwrap_or(start);
|
||||
let size = (end - start + 1).max(1);
|
||||
let lines = if start > 0 { Some((start, end.max(start))) } else { None };
|
||||
(size, lines)
|
||||
}
|
||||
|
||||
fn parse_anchor_line(raw: &str) -> Option<i64> {
|
||||
let caps = SINGLE_ANCHOR_RE.captures(raw.trim())?;
|
||||
caps.get(1)?.as_str().parse().ok()
|
||||
}
|
||||
|
||||
/// Parses a single hashline `input` arg into per-file sections. Returns an
|
||||
/// empty vec if the input doesn't begin with `@PATH` (vim-mode edits etc.).
|
||||
fn parse_hashline_input(input: &str) -> Vec<EditSection> {
|
||||
let mut sections: Vec<EditSection> = Vec::new();
|
||||
let mut cur: Option<EditSection> = None;
|
||||
// Index of the currently-open payload block in cur.payload_blocks, or
|
||||
// `None` if no op is awaiting payload.
|
||||
let mut open_block: Option<usize> = None;
|
||||
|
||||
fn open_new_block(s: &mut EditSection, open_block: &mut Option<usize>) {
|
||||
s.payload_blocks.push(Vec::new());
|
||||
*open_block = Some(s.payload_blocks.len() - 1);
|
||||
}
|
||||
fn close_block(open_block: &mut Option<usize>) {
|
||||
*open_block = None;
|
||||
}
|
||||
|
||||
for raw_line in input.split('\n') {
|
||||
let line = raw_line.strip_suffix('\r').unwrap_or(raw_line);
|
||||
|
||||
// File header: `@<path>`. Always starts a new section.
|
||||
if let Some(rest) = line.strip_prefix('@') {
|
||||
if let Some(s) = cur.take() {
|
||||
sections.push(s);
|
||||
}
|
||||
cur = Some(EditSection { file: rest.trim().to_string(), ..Default::default() });
|
||||
open_block = None;
|
||||
continue;
|
||||
}
|
||||
|
||||
let Some(s) = cur.as_mut() else {
|
||||
// Stray content before any `@PATH` — not a hashline input.
|
||||
continue;
|
||||
};
|
||||
|
||||
// Payload lines (`~...`) belong to whatever op was last opened.
|
||||
if let Some(payload) = line.strip_prefix('~') {
|
||||
s.payload_lines.push(payload.to_string());
|
||||
if open_block.is_none() {
|
||||
open_new_block(s, &mut open_block);
|
||||
}
|
||||
if let Some(idx) = open_block {
|
||||
s.payload_blocks[idx].push(payload.to_string());
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
let trimmed = line.trim_start();
|
||||
let op_byte = trimmed.as_bytes().first().copied();
|
||||
match op_byte {
|
||||
Some(b'+') | Some(b'<') => {
|
||||
let body = trimmed[1..].trim_start();
|
||||
let (anchor_part, tail) = match body.split_once('~') {
|
||||
Some((a, t)) => (a, Some(t)),
|
||||
None => (body, None),
|
||||
};
|
||||
let anchor_trimmed = anchor_part.trim();
|
||||
if !anchor_trimmed.is_empty() && anchor_trimmed != "BOF" && anchor_trimmed != "EOF" {
|
||||
s.op_anchors.push(anchor_trimmed.to_string());
|
||||
}
|
||||
if let Some(line) = parse_anchor_line(anchor_part) {
|
||||
s.touch(line);
|
||||
}
|
||||
if let Some(tail) = tail {
|
||||
// Inline `+ ANCHOR~text`: replaces a single line; not a
|
||||
// payload-collecting op.
|
||||
s.payload_lines.push(tail.to_string());
|
||||
s.deleted_lines += 1;
|
||||
close_block(&mut open_block);
|
||||
} else {
|
||||
open_new_block(s, &mut open_block);
|
||||
}
|
||||
s.op_count += 1;
|
||||
}
|
||||
Some(b'-') | Some(b'=') => {
|
||||
let body = trimmed[1..].trim_start();
|
||||
let (size, lines) = parse_range(body);
|
||||
s.deleted_lines += size;
|
||||
if let Some((lo, hi)) = lines {
|
||||
s.touch(lo);
|
||||
s.touch(hi);
|
||||
}
|
||||
// Collect raw anchor refs (`A` or `A..B`) for dup detection.
|
||||
for part in body.trim().split("..") {
|
||||
let t = part.trim();
|
||||
if !t.is_empty() {
|
||||
s.op_anchors.push(t.to_string());
|
||||
}
|
||||
}
|
||||
s.op_count += 1;
|
||||
if op_byte == Some(b'=') {
|
||||
open_new_block(s, &mut open_block);
|
||||
} else {
|
||||
close_block(&mut open_block);
|
||||
}
|
||||
}
|
||||
_ => {
|
||||
// blank / unrecognized — leave payload state alone.
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(s) = cur.take() {
|
||||
sections.push(s);
|
||||
}
|
||||
sections
|
||||
}
|
||||
|
||||
// ---- success classification ----------------------------------------------
|
||||
|
||||
static RE_FAILURE_HEAD: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(
|
||||
r"(?i)^(edit rejected|error\b|failed\b|invalid\b|unrecognized\b|cannot\b|no enclosing|file has been (modified|changed)|file has not been read|permission denied|tool execution was aborted|request was aborted|cancelled|canceled|line \d+:|expected|unexpected|patch failed|no replacements|0 matches)",
|
||||
)
|
||||
.expect("RE_FAILURE_HEAD")
|
||||
});
|
||||
|
||||
fn looks_successful(text: &str) -> bool {
|
||||
let head = text.lines().find(|l| !l.trim().is_empty()).unwrap_or("");
|
||||
if head.is_empty() {
|
||||
return false;
|
||||
}
|
||||
!RE_FAILURE_HEAD.is_match(head.trim_start())
|
||||
}
|
||||
|
||||
fn extract_warnings(text: &str) -> Vec<&'static str> {
|
||||
let mut out: Vec<&'static str> = Vec::new();
|
||||
for line in text.lines() {
|
||||
let t = line.trim_start();
|
||||
if t.starts_with("Auto-rebased anchor") {
|
||||
out.push("auto-rebased");
|
||||
} else if t.starts_with("Auto-absorbed") {
|
||||
out.push("auto-absorbed");
|
||||
} else if t.starts_with("Auto-dropped") {
|
||||
out.push("auto-dropped");
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
// ---- followup pattern classification -------------------------------------
|
||||
|
||||
#[derive(Clone, Copy, PartialEq, Eq, Hash)]
|
||||
enum FixPattern {
|
||||
AddSingleCloser, // payload contains exactly one line that's pure }/]/)
|
||||
RemoveSingleCloser, // single delete of a pure }/]/) line
|
||||
PureInsertOneLine,
|
||||
PureDeleteOneLine,
|
||||
OneLineModify,
|
||||
SmallOther,
|
||||
}
|
||||
|
||||
impl FixPattern {
|
||||
fn label(&self) -> &'static str {
|
||||
match self {
|
||||
FixPattern::AddSingleCloser => "add-single-closer",
|
||||
FixPattern::RemoveSingleCloser => "remove-single-closer",
|
||||
FixPattern::PureInsertOneLine => "pure-insert-1",
|
||||
FixPattern::PureDeleteOneLine => "pure-delete-1",
|
||||
FixPattern::OneLineModify => "one-line-modify",
|
||||
FixPattern::SmallOther => "small-other",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn pattern_priority(p: FixPattern) -> u8 {
|
||||
match p {
|
||||
FixPattern::RemoveSingleCloser => 0,
|
||||
FixPattern::AddSingleCloser => 1,
|
||||
FixPattern::OneLineModify => 2,
|
||||
FixPattern::PureDeleteOneLine => 3,
|
||||
FixPattern::PureInsertOneLine => 4,
|
||||
FixPattern::SmallOther => 5,
|
||||
}
|
||||
}
|
||||
|
||||
static CLOSER_LINE_RE: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"^\s*[\])}]+[;,]?\s*$").expect("CLOSER_LINE_RE"));
|
||||
|
||||
fn classify_fix(section: &EditSection) -> FixPattern {
|
||||
let payload = §ion.payload_lines;
|
||||
let deleted = section.deleted_lines;
|
||||
let payload_is_one_closer = payload.len() == 1 && CLOSER_LINE_RE.is_match(&payload[0]);
|
||||
|
||||
if deleted == 0 && payload_is_one_closer {
|
||||
return FixPattern::AddSingleCloser;
|
||||
}
|
||||
if deleted == 1 && payload.is_empty() {
|
||||
// We don't have the deleted line text, but a single-line delete on a
|
||||
// sub-1KB follow-up is overwhelmingly the "drop duplicate brace" case
|
||||
// when paired with the immediately-prior big edit. Mark it so;
|
||||
// false-positives are easy to triage by hand.
|
||||
return FixPattern::RemoveSingleCloser;
|
||||
}
|
||||
if deleted == 0 && payload.len() == 1 {
|
||||
return FixPattern::PureInsertOneLine;
|
||||
}
|
||||
if deleted == 1 && payload.len() == 1 {
|
||||
return FixPattern::OneLineModify;
|
||||
}
|
||||
if deleted >= 1 && payload.is_empty() {
|
||||
return FixPattern::PureDeleteOneLine;
|
||||
}
|
||||
FixPattern::SmallOther
|
||||
}
|
||||
|
||||
// ---- per-file scanning ---------------------------------------------------
|
||||
|
||||
#[derive(Clone)]
|
||||
struct Hit {
|
||||
session: String,
|
||||
file: String,
|
||||
first_call_id: String,
|
||||
second_call_id: String,
|
||||
first_size: i64,
|
||||
first_input_len: usize,
|
||||
second_size: i64,
|
||||
pattern: FixPattern,
|
||||
second_summary: String,
|
||||
gap_secs: i64,
|
||||
}
|
||||
|
||||
/// A successful edit whose result text contained a tool self-correction.
|
||||
#[derive(Clone)]
|
||||
struct WarningHit {
|
||||
session: String,
|
||||
call_id: String,
|
||||
kind: &'static str, // auto-rebased / auto-absorbed / auto-dropped
|
||||
files: String, // comma-joined section files
|
||||
input_len: usize,
|
||||
}
|
||||
|
||||
/// Two consecutive successful edits on the same file whose anchor line
|
||||
/// ranges overlap (or one is contained in the other). Catches "agent
|
||||
/// re-edited the same locus" cases that the small-fix detector misses.
|
||||
#[derive(Clone)]
|
||||
struct LocusHit {
|
||||
session: String,
|
||||
file: String,
|
||||
first_call_id: String,
|
||||
second_call_id: String,
|
||||
first_range: (i64, i64),
|
||||
second_range: (i64, i64),
|
||||
first_size: i64,
|
||||
second_size: i64,
|
||||
gap_secs: i64,
|
||||
}
|
||||
|
||||
/// A single hashline edit whose payload contains a contiguous N-line sequence
|
||||
/// that repeats inside the same payload block. Strong signal of "model pasted
|
||||
/// the same chunk twice" when N is large.
|
||||
#[derive(Clone)]
|
||||
struct PayloadDupHit {
|
||||
session: String,
|
||||
call_id: String,
|
||||
file: String,
|
||||
block_len: usize,
|
||||
repeat_len: usize,
|
||||
sample: String,
|
||||
}
|
||||
|
||||
/// A single hashline edit that referenced the same anchor (e.g. `123ab`) from
|
||||
/// two or more distinct ops. Often benign (insert before + insert after at
|
||||
/// the same line), occasionally indicates a duplicated op in the input.
|
||||
#[derive(Clone)]
|
||||
struct AnchorDupHit {
|
||||
session: String,
|
||||
call_id: String,
|
||||
files: String,
|
||||
anchor: String,
|
||||
count: usize,
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
struct FileReport {
|
||||
fix_hits: Vec<Hit>,
|
||||
warning_hits: Vec<WarningHit>,
|
||||
locus_hits: Vec<LocusHit>,
|
||||
payload_dups: Vec<PayloadDupHit>,
|
||||
anchor_dups: Vec<AnchorDupHit>,
|
||||
/// Total number of successful hashline edits scanned.
|
||||
total_successful_edits: i64,
|
||||
}
|
||||
|
||||
/// Find the longest contiguous N-line sequence in `block` that occurs at two
|
||||
/// distinct positions. Returns `(first_index, repeat_len)` if a match of at
|
||||
/// least `min_len` lines exists with at least half its lines being
|
||||
/// non-trivial (>= 4 chars after trim) — this filters out e.g. five repeated
|
||||
/// `}` lines as boilerplate.
|
||||
fn find_longest_repeat(block: &[String], min_len: usize) -> Option<(usize, usize)> {
|
||||
let n = block.len();
|
||||
if n < 2 * min_len {
|
||||
return None;
|
||||
}
|
||||
let mut best: Option<(usize, usize)> = None;
|
||||
for i in 0..n.saturating_sub(min_len) {
|
||||
for j in (i + min_len)..=n.saturating_sub(min_len) {
|
||||
let mut k = 0;
|
||||
while i + k < j && j + k < n && block[i + k] == block[j + k] {
|
||||
k += 1;
|
||||
}
|
||||
if k < min_len {
|
||||
continue;
|
||||
}
|
||||
let meaningful = block[i..i + k]
|
||||
.iter()
|
||||
.filter(|s| s.trim().len() >= 4)
|
||||
.count();
|
||||
if meaningful < (k.div_ceil(2)).max(2) {
|
||||
continue;
|
||||
}
|
||||
if best.is_none_or(|(_, bk)| k > bk) {
|
||||
best = Some((i, k));
|
||||
}
|
||||
}
|
||||
}
|
||||
best
|
||||
}
|
||||
|
||||
/// Returns the set of `(anchor, count, file)` pairs where the same raw
|
||||
/// anchor was referenced by `count >= 2` distinct ops within ONE section
|
||||
/// (i.e. on one file). Two files happening to have the same `LINE+hash`
|
||||
/// anchor is a coincidence (2-char hashes collide), not a duplicate op.
|
||||
/// Skips `BOF` / `EOF` and any anchor with a `*` interior-hash.
|
||||
fn duplicated_anchors(sections: &[EditSection]) -> Vec<(String, usize, String)> {
|
||||
let mut out: Vec<(String, usize, String)> = Vec::new();
|
||||
for sec in sections {
|
||||
let mut counts: HashMap<&str, usize> = HashMap::new();
|
||||
for a in &sec.op_anchors {
|
||||
if a == "BOF" || a == "EOF" || a.contains('*') {
|
||||
continue;
|
||||
}
|
||||
*counts.entry(a.as_str()).or_default() += 1;
|
||||
}
|
||||
for (anchor, c) in counts {
|
||||
if c >= 2 {
|
||||
out.push((anchor.to_string(), c, sec.file.clone()));
|
||||
}
|
||||
}
|
||||
}
|
||||
out.sort_by(|a, b| b.1.cmp(&a.1));
|
||||
out
|
||||
}
|
||||
|
||||
fn process_file(path: &Path, max_fix: i64) -> FileReport {
|
||||
let f = match File::open(path) {
|
||||
Ok(f) => f,
|
||||
Err(e) => {
|
||||
eprintln!("open {}: {e}", path.display());
|
||||
return FileReport::default();
|
||||
}
|
||||
};
|
||||
let reader = BufReader::with_capacity(64 * 1024, f);
|
||||
let session = path
|
||||
.file_stem()
|
||||
.and_then(|s| s.to_str())
|
||||
.unwrap_or("")
|
||||
.to_string();
|
||||
|
||||
// Walk message events in order. For each toolCall name=="edit", parse the
|
||||
// input. Pair with its toolResult by callId. Keep the chronological list
|
||||
// of EditCall records.
|
||||
let mut pending: HashMap<String, EditCall> = HashMap::new();
|
||||
let mut order: Vec<String> = Vec::new();
|
||||
let mut calls: HashMap<String, EditCall> = HashMap::new();
|
||||
|
||||
for line in reader.lines() {
|
||||
let Ok(line) = line else { continue };
|
||||
if line.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let Ok(ev) = serde_json::from_str::<RawEvent>(&line) else {
|
||||
continue;
|
||||
};
|
||||
if ev.kind != "message" {
|
||||
continue;
|
||||
}
|
||||
let Some(msg_raw) = ev.message else { continue };
|
||||
let Ok(m) = serde_json::from_str::<Message>(msg_raw.get()) else {
|
||||
continue;
|
||||
};
|
||||
let Some(content_raw) = m.content else { continue };
|
||||
let items = parse_content(&content_raw);
|
||||
|
||||
match m.role.as_str() {
|
||||
"assistant" => {
|
||||
for it in items {
|
||||
if it.kind != "toolCall" || it.name != "edit" {
|
||||
continue;
|
||||
}
|
||||
let raw = it.arguments.as_deref();
|
||||
let Some(raw) = raw else { continue };
|
||||
let Ok(args) = serde_json::from_str::<EditArgs>(raw.get()) else {
|
||||
continue;
|
||||
};
|
||||
let Some(input) = args.input else { continue };
|
||||
if !input.trim_start().starts_with('@') {
|
||||
// Vim-mode edit or other shape — skip.
|
||||
continue;
|
||||
}
|
||||
let sections = parse_hashline_input(&input);
|
||||
if sections.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let call = EditCall {
|
||||
ts: parse_ts(&ev.timestamp),
|
||||
call_id: it.id.clone(),
|
||||
sections,
|
||||
success: false, // filled in by toolResult
|
||||
raw_input_len: input.len(),
|
||||
warnings: Vec::new(),
|
||||
};
|
||||
pending.insert(it.id.clone(), call);
|
||||
}
|
||||
}
|
||||
"toolResult" => {
|
||||
if m.tool_name != "edit" {
|
||||
continue;
|
||||
}
|
||||
let Some(mut call) = pending.remove(&m.tool_call_id) else {
|
||||
continue;
|
||||
};
|
||||
let text = join_text(&items);
|
||||
call.success = looks_successful(&text);
|
||||
if call.success {
|
||||
call.warnings = extract_warnings(&text);
|
||||
}
|
||||
let id = call.call_id.clone();
|
||||
calls.insert(id.clone(), call);
|
||||
order.push(id);
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
// Build per-file edit sequences. A single call can touch multiple files
|
||||
// (multi-section input); we record an entry per (call, section).
|
||||
struct Entry<'a> {
|
||||
section: &'a EditSection,
|
||||
call: &'a EditCall,
|
||||
}
|
||||
let mut by_file: HashMap<&str, Vec<Entry>> = HashMap::new();
|
||||
for id in &order {
|
||||
let Some(call) = calls.get(id) else { continue };
|
||||
for sec in &call.sections {
|
||||
by_file
|
||||
.entry(sec.file.as_str())
|
||||
.or_default()
|
||||
.push(Entry { section: sec, call });
|
||||
}
|
||||
}
|
||||
|
||||
let mut report = FileReport::default();
|
||||
for call in calls.values() {
|
||||
if call.success {
|
||||
report.total_successful_edits += 1;
|
||||
}
|
||||
if !call.warnings.is_empty() {
|
||||
// Dedup per kind so a call with two `Auto-rebased` lines counts once.
|
||||
let mut seen: Vec<&'static str> = Vec::new();
|
||||
for w in &call.warnings {
|
||||
if !seen.contains(w) {
|
||||
seen.push(*w);
|
||||
}
|
||||
}
|
||||
let files: Vec<String> = call.sections.iter().map(|s| s.file.clone()).collect();
|
||||
for kind in seen {
|
||||
report.warning_hits.push(WarningHit {
|
||||
session: session.clone(),
|
||||
call_id: call.call_id.clone(),
|
||||
kind,
|
||||
files: files.join(","),
|
||||
input_len: call.raw_input_len,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Only flag detectors on edits that actually applied. Failed edits
|
||||
// are handled separately and their input is moot.
|
||||
if call.success {
|
||||
// Payload self-dup: per section, per block. Threshold is
|
||||
// intentionally permissive (>=4) — caller filters by repeat_len.
|
||||
for sec in &call.sections {
|
||||
for block in &sec.payload_blocks {
|
||||
if let Some((start, len)) = find_longest_repeat(block, 4) {
|
||||
let sample = truncate_line(block.get(start).map(String::as_str).unwrap_or(""), 80);
|
||||
report.payload_dups.push(PayloadDupHit {
|
||||
session: session.clone(),
|
||||
call_id: call.call_id.clone(),
|
||||
file: sec.file.clone(),
|
||||
block_len: block.len(),
|
||||
repeat_len: len,
|
||||
sample,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Anchor dup within a single section's ops.
|
||||
for (anchor, count, file) in duplicated_anchors(&call.sections) {
|
||||
report.anchor_dups.push(AnchorDupHit {
|
||||
session: session.clone(),
|
||||
call_id: call.call_id.clone(),
|
||||
files: file,
|
||||
anchor,
|
||||
count,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (file, entries) in by_file {
|
||||
for window in entries.windows(2) {
|
||||
let a = &window[0];
|
||||
let b = &window[1];
|
||||
if !a.call.success || !b.call.success {
|
||||
continue;
|
||||
}
|
||||
if a.call.call_id == b.call.call_id {
|
||||
continue;
|
||||
}
|
||||
let first_size = a.section.change_size();
|
||||
let second_size = b.section.change_size();
|
||||
let gap = (b.call.ts - a.call.ts).max(0);
|
||||
|
||||
// (1) Small-fix follow-up.
|
||||
if second_size > 0 && second_size <= max_fix && first_size > 2 {
|
||||
let pattern = classify_fix(b.section);
|
||||
let summary = render_section_summary(b.section);
|
||||
report.fix_hits.push(Hit {
|
||||
session: session.clone(),
|
||||
file: file.to_string(),
|
||||
first_call_id: a.call.call_id.clone(),
|
||||
second_call_id: b.call.call_id.clone(),
|
||||
first_size,
|
||||
first_input_len: a.call.raw_input_len,
|
||||
second_size,
|
||||
pattern,
|
||||
second_summary: summary,
|
||||
gap_secs: gap,
|
||||
});
|
||||
}
|
||||
|
||||
// (2) Same-locus re-edit. Skip when both are tiny (those are
|
||||
// already noisy) and skip when the small-fix branch already
|
||||
// fired (we don't want to double-count).
|
||||
if first_size > 2 && second_size > max_fix {
|
||||
if let (Some(a_lo), Some(a_hi), Some(b_lo), Some(b_hi)) = (
|
||||
a.section.min_line,
|
||||
a.section.max_line,
|
||||
b.section.min_line,
|
||||
b.section.max_line,
|
||||
) {
|
||||
let overlap_lo = a_lo.max(b_lo);
|
||||
let overlap_hi = a_hi.min(b_hi);
|
||||
if overlap_lo <= overlap_hi {
|
||||
report.locus_hits.push(LocusHit {
|
||||
session: session.clone(),
|
||||
file: file.to_string(),
|
||||
first_call_id: a.call.call_id.clone(),
|
||||
second_call_id: b.call.call_id.clone(),
|
||||
first_range: (a_lo, a_hi),
|
||||
second_range: (b_lo, b_hi),
|
||||
first_size,
|
||||
second_size,
|
||||
gap_secs: gap,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
report
|
||||
}
|
||||
|
||||
fn render_section_summary(section: &EditSection) -> String {
|
||||
let mut bits: Vec<String> = Vec::new();
|
||||
if section.deleted_lines > 0 {
|
||||
bits.push(format!("-{}", section.deleted_lines));
|
||||
}
|
||||
if !section.payload_lines.is_empty() {
|
||||
bits.push(format!("+{}", section.payload_lines.len()));
|
||||
}
|
||||
let mut out = bits.join(" / ");
|
||||
if !section.payload_lines.is_empty() {
|
||||
let preview = truncate_line(§ion.payload_lines[0], 80);
|
||||
out.push_str(&format!(" | {preview}"));
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
// ---- entry point ---------------------------------------------------------
|
||||
|
||||
pub fn run(args: Vec<String>) -> Result<()> {
|
||||
let mut limit: usize = 200;
|
||||
let mut workers: usize = 0;
|
||||
let mut max_fix: i64 = 2;
|
||||
let mut show: usize = 60;
|
||||
let mut max_gap: i64 = 0;
|
||||
let mut pattern_filter: Option<String> = None;
|
||||
let mut min_dup: usize = 8;
|
||||
|
||||
let mut iter = args.into_iter();
|
||||
while let Some(a) = iter.next() {
|
||||
match a.as_str() {
|
||||
"-n" => {
|
||||
limit = iter
|
||||
.next()
|
||||
.context("-n requires a value")?
|
||||
.parse()
|
||||
.context("-n value")?;
|
||||
}
|
||||
"-j" => {
|
||||
workers = iter
|
||||
.next()
|
||||
.context("-j requires a value")?
|
||||
.parse()
|
||||
.context("-j value")?;
|
||||
}
|
||||
"--max-fix" => {
|
||||
max_fix = iter
|
||||
.next()
|
||||
.context("--max-fix requires a value")?
|
||||
.parse()
|
||||
.context("--max-fix value")?;
|
||||
}
|
||||
"--show" => {
|
||||
show = iter
|
||||
.next()
|
||||
.context("--show requires a value")?
|
||||
.parse()
|
||||
.context("--show value")?;
|
||||
}
|
||||
"--max-gap" => {
|
||||
max_gap = iter
|
||||
.next()
|
||||
.context("--max-gap requires seconds")?
|
||||
.parse()
|
||||
.context("--max-gap value")?;
|
||||
}
|
||||
"--pattern" => {
|
||||
pattern_filter = Some(iter.next().context("--pattern requires a name")?);
|
||||
}
|
||||
"--min-dup" => {
|
||||
min_dup = iter
|
||||
.next()
|
||||
.context("--min-dup requires a value")?
|
||||
.parse()
|
||||
.context("--min-dup value")?;
|
||||
}
|
||||
"-h" | "--help" => {
|
||||
eprintln!(
|
||||
"usage: session-stats followups [-n N] [-j workers] [--max-fix N] [--max-gap S]
|
||||
[--min-dup K] [--pattern NAME] [--show N]
|
||||
|
||||
Five detectors over hashline `edit` calls in the most-recent N sessions:
|
||||
|
||||
1. small-fix follow-ups
|
||||
consecutive successful edits on the same file where the follow-up
|
||||
changes <= --max-fix lines (default 2). The first edit must be > 2
|
||||
lines. Brace-related patterns surface first (remove-single-closer,
|
||||
add-single-closer). --max-gap caps elapsed seconds between the pair.
|
||||
|
||||
2. tool self-corrections
|
||||
warning lines emitted by the tool on otherwise-successful edits:
|
||||
auto-rebased, auto-absorbed, auto-dropped. These are direct evidence
|
||||
of the model writing stale anchors or duplicating adjacent context.
|
||||
|
||||
3. same-locus re-edits
|
||||
two consecutive edits on the same file whose anchor line ranges
|
||||
overlap, where both edits are > --max-fix lines (so they're not
|
||||
already in (1)).
|
||||
|
||||
4. payload self-duplication
|
||||
within one payload block, a contiguous K-line sequence appears at two
|
||||
positions. K threshold via --min-dup (default 8).
|
||||
|
||||
5. same-anchor reuse
|
||||
two or more ops in the same section reference the same `LINE+hash`
|
||||
anchor.
|
||||
|
||||
--pattern NAME filters (1) to a single FixPattern label."
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
other => bail!("unknown flag: {other}"),
|
||||
}
|
||||
}
|
||||
|
||||
let files = collect_sessions(&WalkOpts {
|
||||
date_filters: Vec::new(),
|
||||
limit_most_recent: limit,
|
||||
})?;
|
||||
eprintln!("scanning {} session files", files.len());
|
||||
|
||||
let max_fix_local = max_fix;
|
||||
let reports: Vec<FileReport> = parallel_collect(&files, workers, 5_000, |p| {
|
||||
let r = process_file(p, max_fix_local);
|
||||
let empty = r.fix_hits.is_empty() && r.warning_hits.is_empty() && r.locus_hits.is_empty()
|
||||
&& r.total_successful_edits == 0;
|
||||
if empty { None } else { Some(r) }
|
||||
});
|
||||
|
||||
let mut hits: Vec<Hit> = Vec::new();
|
||||
let mut warning_hits: Vec<WarningHit> = Vec::new();
|
||||
let mut locus_hits: Vec<LocusHit> = Vec::new();
|
||||
let mut payload_dups: Vec<PayloadDupHit> = Vec::new();
|
||||
let mut anchor_dups: Vec<AnchorDupHit> = Vec::new();
|
||||
let mut total_successful_edits: i64 = 0;
|
||||
for r in reports {
|
||||
hits.extend(r.fix_hits);
|
||||
warning_hits.extend(r.warning_hits);
|
||||
locus_hits.extend(r.locus_hits);
|
||||
payload_dups.extend(r.payload_dups);
|
||||
anchor_dups.extend(r.anchor_dups);
|
||||
total_successful_edits += r.total_successful_edits;
|
||||
}
|
||||
|
||||
if max_gap > 0 {
|
||||
hits.retain(|h| h.gap_secs <= max_gap);
|
||||
locus_hits.retain(|h| h.gap_secs <= max_gap);
|
||||
}
|
||||
if let Some(ref name) = pattern_filter {
|
||||
hits.retain(|h| h.pattern.label() == name);
|
||||
}
|
||||
hits.sort_by(|a, b| {
|
||||
pattern_priority(a.pattern)
|
||||
.cmp(&pattern_priority(b.pattern))
|
||||
.then_with(|| b.first_input_len.cmp(&a.first_input_len))
|
||||
});
|
||||
locus_hits.sort_by(|a, b| b.first_size.cmp(&a.first_size));
|
||||
|
||||
let mut by_pattern: HashMap<&'static str, i64> = HashMap::new();
|
||||
for h in &hits {
|
||||
*by_pattern.entry(h.pattern.label()).or_default() += 1;
|
||||
}
|
||||
|
||||
println!("=== heuristic followup hits ===");
|
||||
println!("total hits: {}", commas(hits.len() as i64));
|
||||
println!();
|
||||
println!("by pattern:");
|
||||
let mut pats: Vec<(&&str, &i64)> = by_pattern.iter().collect();
|
||||
pats.sort_by(|a, b| b.1.cmp(a.1));
|
||||
for (label, n) in pats {
|
||||
println!(" {:<22} {:>6}", label, commas(*n));
|
||||
}
|
||||
println!();
|
||||
|
||||
let shown = show.min(hits.len());
|
||||
println!("=== top {shown} hits (by first-edit input size) ===");
|
||||
for h in hits.iter().take(shown) {
|
||||
println!(
|
||||
"[{pat}] {file} first={first_size}L ({first_len}B) → second={second_size}L gap={gap}s",
|
||||
pat = h.pattern.label(),
|
||||
file = h.file,
|
||||
first_size = h.first_size,
|
||||
first_len = h.first_input_len,
|
||||
second_size = h.second_size,
|
||||
gap = h.gap_secs,
|
||||
);
|
||||
println!(" session={}", h.session);
|
||||
println!(" first_call={} second_call={}", h.first_call_id, h.second_call_id);
|
||||
println!(" fix: {}", h.second_summary);
|
||||
}
|
||||
println!();
|
||||
println!("=== tool self-corrections ===");
|
||||
println!("(emitted as warnings on otherwise-successful edits — the tool caught what the model wrote)");
|
||||
let mut warn_by_kind: HashMap<&'static str, i64> = HashMap::new();
|
||||
for w in &warning_hits {
|
||||
*warn_by_kind.entry(w.kind).or_default() += 1;
|
||||
}
|
||||
let mut warn_kinds: Vec<(&&str, &i64)> = warn_by_kind.iter().collect();
|
||||
warn_kinds.sort_by(|a, b| b.1.cmp(a.1));
|
||||
let pct_of = |n: i64| -> String {
|
||||
if total_successful_edits == 0 {
|
||||
String::new()
|
||||
} else {
|
||||
format!(" ({:.2}% of {} successful edits)", pct(n, total_successful_edits), commas(total_successful_edits))
|
||||
}
|
||||
};
|
||||
for (kind, n) in warn_kinds {
|
||||
println!(" {:<16} {:>6}{}", kind, commas(*n), pct_of(*n));
|
||||
}
|
||||
let warn_show = show.min(warning_hits.len());
|
||||
if warn_show > 0 {
|
||||
println!();
|
||||
println!("--- top {warn_show} self-correction examples (by input size) ---");
|
||||
let mut sorted_warns = warning_hits.clone();
|
||||
sorted_warns.sort_by(|a, b| b.input_len.cmp(&a.input_len));
|
||||
for w in sorted_warns.iter().take(warn_show) {
|
||||
println!(
|
||||
"[{kind}] {files} ({len}B)",
|
||||
kind = w.kind,
|
||||
files = truncate_line(&w.files, 80),
|
||||
len = w.input_len,
|
||||
);
|
||||
println!(" session={} call={}", w.session, w.call_id);
|
||||
}
|
||||
}
|
||||
|
||||
println!();
|
||||
println!("=== same-locus re-edits (overlapping anchor ranges, both > max-fix) ===");
|
||||
println!("hits: {}", commas(locus_hits.len() as i64));
|
||||
let locus_show = show.min(locus_hits.len());
|
||||
for h in locus_hits.iter().take(locus_show) {
|
||||
println!(
|
||||
"{file} first={a0}..{a1} ({fs}L) → second={b0}..{b1} ({ss}L) gap={gap}s",
|
||||
file = h.file,
|
||||
a0 = h.first_range.0,
|
||||
a1 = h.first_range.1,
|
||||
fs = h.first_size,
|
||||
b0 = h.second_range.0,
|
||||
b1 = h.second_range.1,
|
||||
ss = h.second_size,
|
||||
gap = h.gap_secs,
|
||||
);
|
||||
println!(" session={}", h.session);
|
||||
println!(" first_call={} second_call={}", h.first_call_id, h.second_call_id);
|
||||
}
|
||||
|
||||
println!();
|
||||
println!("=== payload self-duplication (model pasted same N-line chunk twice in one payload) ===");
|
||||
payload_dups.retain(|p| p.repeat_len >= min_dup);
|
||||
payload_dups.sort_by(|a, b| b.repeat_len.cmp(&a.repeat_len));
|
||||
println!(
|
||||
"hits with repeat_len >= {}: {} ({:.2}% of {} successful edits)",
|
||||
min_dup,
|
||||
commas(payload_dups.len() as i64),
|
||||
pct(payload_dups.len() as i64, total_successful_edits),
|
||||
commas(total_successful_edits)
|
||||
);
|
||||
let dup_show = show.min(payload_dups.len());
|
||||
for p in payload_dups.iter().take(dup_show) {
|
||||
println!(
|
||||
"k={k} block={blk}L {file}",
|
||||
k = p.repeat_len,
|
||||
blk = p.block_len,
|
||||
file = p.file,
|
||||
);
|
||||
println!(" session={} call={}", p.session, p.call_id);
|
||||
println!(" sample: {}", p.sample);
|
||||
}
|
||||
|
||||
println!();
|
||||
println!("=== same-anchor reused by multiple ops in one input ===");
|
||||
anchor_dups.sort_by(|a, b| b.count.cmp(&a.count));
|
||||
println!("hits: {}", commas(anchor_dups.len() as i64));
|
||||
let anchor_show = show.min(anchor_dups.len());
|
||||
for a in anchor_dups.iter().take(anchor_show) {
|
||||
println!(
|
||||
"anchor {anchor} referenced {n}x files={files}",
|
||||
anchor = a.anchor,
|
||||
n = a.count,
|
||||
files = truncate_line(&a.files, 80),
|
||||
);
|
||||
println!(" session={} call={}", a.session, a.call_id);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
@@ -1,702 +0,0 @@
|
||||
//! `tools` subcommand — per-tool token totals across the most-recent N session
|
||||
//! jsonl files.
|
||||
//!
|
||||
//! Token counting uses o200k_base via tiktoken-rs (the GPT-4o / GPT-5 family
|
||||
//! tokenizer). It is not Claude's own BPE, but it is well-defined offline and
|
||||
//! within ~5-10% across English/code in aggregate.
|
||||
//!
|
||||
//! Buckets:
|
||||
//! tool ARGS — assistant tool-call argument JSON
|
||||
//! tool RESULTS — tool result content text
|
||||
//! assistant THINKING — assistant `thinking` blocks
|
||||
//! assistant TEXT — assistant prose
|
||||
//! user TEXT — user-authored text content
|
||||
//!
|
||||
//! Output: grand totals + per-tool breakdown sorted by total (arg+res) tokens.
|
||||
//! Optional CSV at `$TOOL_USAGE_CSV` (per-tool totals) or
|
||||
//! `--calls-csv PATH` / `$TOOL_CALLS_CSV` (one row per tool call).
|
||||
//!
|
||||
//! Pass `--by <h|d|w|m|Nh|Nd|Nw>` to bucket per-call data into rolling windows
|
||||
//! and surface per-tool tokens/call (avg + p50 + p95) over time, so you can
|
||||
//! spot regressions in tool efficiency. The buckets do not align to calendar
|
||||
//! boundaries; they are pure `floor(unix_secs / N)` slices.
|
||||
|
||||
use crate::common::*;
|
||||
use anyhow::{Context, Result, bail};
|
||||
use std::collections::HashMap;
|
||||
use std::fs::File;
|
||||
use std::io::{BufRead, BufReader};
|
||||
use std::path::Path;
|
||||
|
||||
#[derive(Default, Clone)]
|
||||
struct ToolAgg {
|
||||
calls: i64,
|
||||
results: i64,
|
||||
arg_tok: i64,
|
||||
res_tok: i64,
|
||||
}
|
||||
|
||||
#[derive(Default, Clone)]
|
||||
struct SessionTotals {
|
||||
arg_tok: i64,
|
||||
res_tok: i64,
|
||||
thinking_tok: i64,
|
||||
text_tok: i64,
|
||||
user_tok: i64,
|
||||
n_calls: i64,
|
||||
n_results: i64,
|
||||
}
|
||||
|
||||
struct FileResult {
|
||||
totals: SessionTotals,
|
||||
tools: HashMap<String, ToolAgg>,
|
||||
calls: Vec<CallRecord>,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
struct CallRecord {
|
||||
ts: i64,
|
||||
tool: String,
|
||||
session: String,
|
||||
model: String,
|
||||
arg_tok: i32,
|
||||
res_tok: i32,
|
||||
}
|
||||
|
||||
struct PendingCall {
|
||||
tool: String,
|
||||
ts: i64,
|
||||
arg_tok: i32,
|
||||
model: String,
|
||||
}
|
||||
|
||||
pub fn run(args: Vec<String>) -> Result<()> {
|
||||
let mut limit: usize = 1_000;
|
||||
let mut workers: usize = 0;
|
||||
let mut by: Option<i64> = None;
|
||||
let mut top: usize = 12;
|
||||
let mut tool_filter: Option<String> = None;
|
||||
let mut calls_csv: Option<String> = std::env::var("TOOL_CALLS_CSV")
|
||||
.ok()
|
||||
.filter(|s| !s.is_empty());
|
||||
|
||||
let mut iter = args.into_iter();
|
||||
while let Some(a) = iter.next() {
|
||||
match a.as_str() {
|
||||
"-n" => {
|
||||
limit = iter
|
||||
.next()
|
||||
.context("-n requires a value")?
|
||||
.parse()
|
||||
.context("-n value")?;
|
||||
}
|
||||
"-j" => {
|
||||
workers = iter
|
||||
.next()
|
||||
.context("-j requires a value")?
|
||||
.parse()
|
||||
.context("-j value")?;
|
||||
}
|
||||
"--by" => {
|
||||
let spec = iter.next().context("--by requires a bucket spec")?;
|
||||
by = Some(parse_bucket(&spec)?);
|
||||
}
|
||||
"--top" => {
|
||||
top = iter
|
||||
.next()
|
||||
.context("--top requires a value")?
|
||||
.parse()
|
||||
.context("--top value")?;
|
||||
}
|
||||
"--tool" => {
|
||||
tool_filter = Some(iter.next().context("--tool requires a name")?);
|
||||
}
|
||||
"--calls-csv" => {
|
||||
calls_csv = Some(iter.next().context("--calls-csv requires a path")?);
|
||||
}
|
||||
"-h" | "--help" => {
|
||||
eprintln!(
|
||||
"usage: session-stats tools [-n N] [-j workers] [--by SPEC] [--top N]
|
||||
[--tool NAME] [--calls-csv PATH]
|
||||
|
||||
Aggregates per-tool token usage across the most-recent N session
|
||||
jsonl files (default 1000). Tokenizer: o200k_base.
|
||||
|
||||
--by SPEC bucket per-call data into rolling windows. SPEC is one of:
|
||||
hour, day, week, month, or <N>{{h,d,w}} (e.g. 7d, 12h, 2w).
|
||||
Buckets are pure floor(unix_secs / N); they do not align to
|
||||
calendar boundaries.
|
||||
--top N limit per-tool series to the N most-called tools (default 12).
|
||||
--tool NAME show only this tool in the bucketed series.
|
||||
--calls-csv PATH
|
||||
emit one CSV row per tool call (ts, session, tool, model,
|
||||
arg_tok, res_tok). Env fallback: TOOL_CALLS_CSV.
|
||||
|
||||
TOOL_USAGE_CSV env still emits the per-tool grand-totals CSV."
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
other => bail!("unknown flag: {other}"),
|
||||
}
|
||||
}
|
||||
|
||||
let files = collect_sessions(&WalkOpts {
|
||||
date_filters: Vec::new(),
|
||||
limit_most_recent: limit,
|
||||
})?;
|
||||
eprintln!(
|
||||
"scanning {} session files (tokenizer: o200k_base)",
|
||||
files.len()
|
||||
);
|
||||
|
||||
let results = parallel_collect(&files, workers, 5_000, process_file);
|
||||
|
||||
let sessions = results.len();
|
||||
let mut grand = SessionTotals::default();
|
||||
let mut tools: HashMap<String, ToolAgg> = HashMap::new();
|
||||
let mut all_calls: Vec<CallRecord> = Vec::new();
|
||||
for r in results {
|
||||
grand.arg_tok += r.totals.arg_tok;
|
||||
grand.res_tok += r.totals.res_tok;
|
||||
grand.thinking_tok += r.totals.thinking_tok;
|
||||
grand.text_tok += r.totals.text_tok;
|
||||
grand.user_tok += r.totals.user_tok;
|
||||
grand.n_calls += r.totals.n_calls;
|
||||
grand.n_results += r.totals.n_results;
|
||||
for (name, t) in r.tools {
|
||||
let dst = tools.entry(name).or_default();
|
||||
dst.calls += t.calls;
|
||||
dst.results += t.results;
|
||||
dst.arg_tok += t.arg_tok;
|
||||
dst.res_tok += t.res_tok;
|
||||
}
|
||||
all_calls.extend(r.calls);
|
||||
}
|
||||
|
||||
if let Some(ref name) = tool_filter {
|
||||
all_calls.retain(|c| &c.tool == name);
|
||||
}
|
||||
|
||||
print_grand(&grand, sessions);
|
||||
println!();
|
||||
print_table(&tools);
|
||||
write_csv(&tools)?;
|
||||
|
||||
if let Some(bucket_secs) = by {
|
||||
println!();
|
||||
print_buckets(&all_calls, bucket_secs, top, tool_filter.as_deref());
|
||||
}
|
||||
if let Some(path) = calls_csv {
|
||||
write_calls_csv(&all_calls, &path)?;
|
||||
eprintln!("wrote {} calls to {path}", commas(all_calls.len() as i64));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn process_file(path: &Path) -> Option<FileResult> {
|
||||
let f = match File::open(path) {
|
||||
Ok(f) => f,
|
||||
Err(e) => {
|
||||
eprintln!("open {}: {e}", path.display());
|
||||
return None;
|
||||
}
|
||||
};
|
||||
let reader = BufReader::with_capacity(64 * 1024, f);
|
||||
|
||||
let mut totals = SessionTotals::default();
|
||||
let mut tools: HashMap<String, ToolAgg> = HashMap::new();
|
||||
let mut calls: Vec<CallRecord> = Vec::new();
|
||||
let session = session_id_from_path(path);
|
||||
// Pending call records: keyed by toolCallId so the matching toolResult
|
||||
// can finalize a CallRecord with both arg+res tokens. Falls back to
|
||||
// message.toolName when the id is absent (legacy sessions).
|
||||
let mut pending: HashMap<String, PendingCall> = HashMap::new();
|
||||
|
||||
for line in reader.lines() {
|
||||
let Ok(line) = line else { continue };
|
||||
if line.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let Ok(ev) = serde_json::from_str::<RawEvent>(&line) else {
|
||||
continue;
|
||||
};
|
||||
if ev.kind != "message" {
|
||||
continue;
|
||||
}
|
||||
let Some(msg_raw) = ev.message else { continue };
|
||||
let Ok(m) = serde_json::from_str::<Message>(msg_raw.get()) else {
|
||||
continue;
|
||||
};
|
||||
let Some(content_raw) = m.content else { continue };
|
||||
let items = parse_content(&content_raw);
|
||||
|
||||
match m.role.as_str() {
|
||||
"assistant" => {
|
||||
for it in items {
|
||||
match it.kind.as_str() {
|
||||
"toolCall" => {
|
||||
let name = normalize_tool(&it.name);
|
||||
let args_str = it.arguments.as_deref().map(RawValue::get).unwrap_or("");
|
||||
let tok = count_tokens(args_str) as i64;
|
||||
totals.arg_tok += tok;
|
||||
totals.n_calls += 1;
|
||||
let t = tools.entry(name.clone()).or_default();
|
||||
t.calls += 1;
|
||||
t.arg_tok += tok;
|
||||
pending.insert(
|
||||
it.id,
|
||||
PendingCall {
|
||||
tool: name,
|
||||
ts: parse_ts(&ev.timestamp),
|
||||
arg_tok: clamp_i32(tok),
|
||||
model: m.model.clone(),
|
||||
},
|
||||
);
|
||||
}
|
||||
"thinking" => {
|
||||
totals.thinking_tok += count_tokens(&it.thinking) as i64;
|
||||
}
|
||||
"text" => {
|
||||
totals.text_tok += count_tokens(&it.text) as i64;
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
"toolResult" => {
|
||||
let text = join_text(&items);
|
||||
let tok = count_tokens(&text) as i64;
|
||||
totals.res_tok += tok;
|
||||
totals.n_results += 1;
|
||||
let pc = pending.remove(&m.tool_call_id);
|
||||
let name = pc
|
||||
.as_ref()
|
||||
.map(|p| p.tool.clone())
|
||||
.unwrap_or_else(|| normalize_tool(&m.tool_name));
|
||||
let t = tools.entry(name.clone()).or_default();
|
||||
t.results += 1;
|
||||
t.res_tok += tok;
|
||||
if let Some(p) = pc {
|
||||
calls.push(CallRecord {
|
||||
ts: p.ts,
|
||||
tool: name,
|
||||
session: session.clone(),
|
||||
model: p.model,
|
||||
arg_tok: p.arg_tok,
|
||||
res_tok: clamp_i32(tok),
|
||||
});
|
||||
}
|
||||
}
|
||||
"user" => {
|
||||
for it in items {
|
||||
if it.kind == "text" {
|
||||
totals.user_tok += count_tokens(&it.text) as i64;
|
||||
}
|
||||
}
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
Some(FileResult { totals, tools, calls })
|
||||
}
|
||||
|
||||
use serde_json::value::RawValue;
|
||||
|
||||
fn normalize_tool(name: &str) -> String {
|
||||
if name.is_empty() {
|
||||
"<unknown>".to_string()
|
||||
} else {
|
||||
name.to_string()
|
||||
}
|
||||
}
|
||||
|
||||
// ---- reporting ----
|
||||
|
||||
fn print_grand(g: &SessionTotals, sessions: usize) {
|
||||
let total = g.arg_tok + g.res_tok + g.thinking_tok + g.text_tok + g.user_tok;
|
||||
let rows: [(&str, i64); 5] = [
|
||||
("tool call ARGS", g.arg_tok),
|
||||
("tool RESULTS", g.res_tok),
|
||||
("assistant THINKING", g.thinking_tok),
|
||||
("assistant TEXT", g.text_tok),
|
||||
("user TEXT", g.user_tok),
|
||||
];
|
||||
let label_w = rows.iter().map(|(l, _)| l.len()).max().unwrap_or(0);
|
||||
let val_w = rows
|
||||
.iter()
|
||||
.map(|(_, n)| commas(*n).len())
|
||||
.chain(std::iter::once(commas(total).len()))
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
|
||||
println!("=== Grand totals across {} sessions ===", commas(sessions as i64));
|
||||
for (label, n) in rows {
|
||||
println!(
|
||||
"{label:<label_w$} {:>val_w$} tok ({:>5.1}%)",
|
||||
commas(n),
|
||||
pct(n, total),
|
||||
);
|
||||
}
|
||||
println!("{:<label_w$} {}", "", "-".repeat(val_w));
|
||||
println!("{:<label_w$} {:>val_w$} tok", "TOTAL", commas(total));
|
||||
println!();
|
||||
println!(
|
||||
"tool calls: {}, tool results: {}",
|
||||
commas(g.n_calls),
|
||||
commas(g.n_results)
|
||||
);
|
||||
if g.n_calls > 0 {
|
||||
println!(
|
||||
"avg arg tokens / call: {:.1}",
|
||||
g.arg_tok as f64 / g.n_calls as f64
|
||||
);
|
||||
}
|
||||
if g.n_results > 0 {
|
||||
println!(
|
||||
"avg result tokens / call: {:.1}",
|
||||
g.res_tok as f64 / g.n_results as f64
|
||||
);
|
||||
}
|
||||
if g.arg_tok > 0 {
|
||||
println!(
|
||||
"ratio result / arg: {:.2}x",
|
||||
g.res_tok as f64 / g.arg_tok as f64
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
struct ToolRow {
|
||||
name: String,
|
||||
calls: i64,
|
||||
arg_tok: i64,
|
||||
res_tok: i64,
|
||||
total: i64,
|
||||
avg_arg: f64,
|
||||
avg_res: f64,
|
||||
res_o_arg: f64,
|
||||
}
|
||||
|
||||
fn print_table(tools: &HashMap<String, ToolAgg>) {
|
||||
let mut rows: Vec<ToolRow> = tools
|
||||
.iter()
|
||||
.filter_map(|(name, t)| {
|
||||
if t.calls == 0 && t.results == 0 {
|
||||
return None;
|
||||
}
|
||||
let mut r = ToolRow {
|
||||
name: name.clone(),
|
||||
calls: t.calls,
|
||||
arg_tok: t.arg_tok,
|
||||
res_tok: t.res_tok,
|
||||
total: t.arg_tok + t.res_tok,
|
||||
avg_arg: 0.0,
|
||||
avg_res: 0.0,
|
||||
res_o_arg: 0.0,
|
||||
};
|
||||
if t.calls > 0 {
|
||||
r.avg_arg = t.arg_tok as f64 / t.calls as f64;
|
||||
r.avg_res = t.res_tok as f64 / t.calls as f64;
|
||||
}
|
||||
if t.arg_tok > 0 {
|
||||
r.res_o_arg = t.res_tok as f64 / t.arg_tok as f64;
|
||||
}
|
||||
Some(r)
|
||||
})
|
||||
.collect();
|
||||
rows.sort_by(|a, b| b.total.cmp(&a.total));
|
||||
|
||||
const TOP: usize = 25;
|
||||
let shown = TOP.min(rows.len());
|
||||
let head_rows = &rows[..shown];
|
||||
|
||||
// "(N others)" trailing summary, computed before width measurement so its
|
||||
// string contents participate in column sizing.
|
||||
let others = (rows.len() > TOP).then(|| {
|
||||
let mut sc = 0i64;
|
||||
let mut sa = 0i64;
|
||||
let mut sr = 0i64;
|
||||
for r in &rows[TOP..] {
|
||||
sc += r.calls;
|
||||
sa += r.arg_tok;
|
||||
sr += r.res_tok;
|
||||
}
|
||||
OthersRow {
|
||||
label: format!("({} others)", rows.len() - TOP),
|
||||
calls: sc,
|
||||
arg_tok: sa,
|
||||
res_tok: sr,
|
||||
total: sa + sr,
|
||||
}
|
||||
});
|
||||
|
||||
// Compute column widths from header label and every value that will
|
||||
// appear under that header (including the "others" summary row, if any).
|
||||
let max_str = |header: &str, vals: &[&str]| -> usize {
|
||||
vals.iter().map(|s| s.len()).chain(std::iter::once(header.len())).max().unwrap_or(0)
|
||||
};
|
||||
|
||||
let names: Vec<&str> = head_rows
|
||||
.iter()
|
||||
.map(|r| r.name.as_str())
|
||||
.chain(others.as_ref().map(|o| o.label.as_str()))
|
||||
.collect();
|
||||
let calls: Vec<String> = head_rows
|
||||
.iter()
|
||||
.map(|r| commas(r.calls))
|
||||
.chain(others.as_ref().map(|o| commas(o.calls)))
|
||||
.collect();
|
||||
let arg_toks: Vec<String> = head_rows
|
||||
.iter()
|
||||
.map(|r| commas(r.arg_tok))
|
||||
.chain(others.as_ref().map(|o| commas(o.arg_tok)))
|
||||
.collect();
|
||||
let res_toks: Vec<String> = head_rows
|
||||
.iter()
|
||||
.map(|r| commas(r.res_tok))
|
||||
.chain(others.as_ref().map(|o| commas(o.res_tok)))
|
||||
.collect();
|
||||
let totals: Vec<String> = head_rows
|
||||
.iter()
|
||||
.map(|r| commas(r.total))
|
||||
.chain(others.as_ref().map(|o| commas(o.total)))
|
||||
.collect();
|
||||
let avg_args: Vec<String> = head_rows.iter().map(|r| format!("{:.1}", r.avg_arg)).collect();
|
||||
let avg_ress: Vec<String> = head_rows.iter().map(|r| format!("{:.1}", r.avg_res)).collect();
|
||||
let res_o_args: Vec<String> =
|
||||
head_rows.iter().map(|r| format!("{:.2}", r.res_o_arg)).collect();
|
||||
|
||||
let name_w = max_str("tool", &names);
|
||||
let calls_w = max_str("calls", &calls.iter().map(String::as_str).collect::<Vec<_>>());
|
||||
let arg_w = max_str("arg_tok", &arg_toks.iter().map(String::as_str).collect::<Vec<_>>());
|
||||
let res_w = max_str("res_tok", &res_toks.iter().map(String::as_str).collect::<Vec<_>>());
|
||||
let tot_w = max_str("total", &totals.iter().map(String::as_str).collect::<Vec<_>>());
|
||||
let avga_w = max_str("avg_arg", &avg_args.iter().map(String::as_str).collect::<Vec<_>>());
|
||||
let avgr_w = max_str("avg_res", &avg_ress.iter().map(String::as_str).collect::<Vec<_>>());
|
||||
let ratio_w = max_str("res/arg", &res_o_args.iter().map(String::as_str).collect::<Vec<_>>());
|
||||
|
||||
let total_width = name_w + 1 + calls_w + 1 + arg_w + 1 + res_w + 1 + tot_w + 1 + avga_w + 1 + avgr_w + 1 + ratio_w;
|
||||
|
||||
println!(
|
||||
"{:<name_w$} {:>calls_w$} {:>arg_w$} {:>res_w$} {:>tot_w$} {:>avga_w$} {:>avgr_w$} {:>ratio_w$}",
|
||||
"tool", "calls", "arg_tok", "res_tok", "total", "avg_arg", "avg_res", "res/arg"
|
||||
);
|
||||
println!("{}", "-".repeat(total_width));
|
||||
|
||||
for (i, r) in head_rows.iter().enumerate() {
|
||||
println!(
|
||||
"{:<name_w$} {:>calls_w$} {:>arg_w$} {:>res_w$} {:>tot_w$} {:>avga_w$} {:>avgr_w$} {:>ratio_w$}",
|
||||
r.name, calls[i], arg_toks[i], res_toks[i], totals[i], avg_args[i], avg_ress[i], res_o_args[i],
|
||||
);
|
||||
}
|
||||
if let Some(o) = others {
|
||||
let i = head_rows.len();
|
||||
println!(
|
||||
"{:<name_w$} {:>calls_w$} {:>arg_w$} {:>res_w$} {:>tot_w$}",
|
||||
o.label, calls[i], arg_toks[i], res_toks[i], totals[i],
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
struct OthersRow {
|
||||
label: String,
|
||||
calls: i64,
|
||||
arg_tok: i64,
|
||||
res_tok: i64,
|
||||
total: i64,
|
||||
}
|
||||
|
||||
fn write_csv(tools: &HashMap<String, ToolAgg>) -> Result<()> {
|
||||
let path = std::env::var("TOOL_USAGE_CSV").unwrap_or_default();
|
||||
if path.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
let f = File::create(&path).with_context(|| format!("create {path}"))?;
|
||||
let mut w = csv::Writer::from_writer(f);
|
||||
w.write_record(["tool", "calls", "results", "arg_tok", "res_tok", "total"])?;
|
||||
let mut names: Vec<&String> = tools.keys().collect();
|
||||
names.sort_by(|a, b| {
|
||||
let ai = {
|
||||
let t = &tools[a.as_str()];
|
||||
t.arg_tok + t.res_tok
|
||||
};
|
||||
let aj = {
|
||||
let t = &tools[b.as_str()];
|
||||
t.arg_tok + t.res_tok
|
||||
};
|
||||
aj.cmp(&ai)
|
||||
});
|
||||
for n in names {
|
||||
let t = &tools[n.as_str()];
|
||||
w.write_record([
|
||||
n.as_str(),
|
||||
&t.calls.to_string(),
|
||||
&t.results.to_string(),
|
||||
&t.arg_tok.to_string(),
|
||||
&t.res_tok.to_string(),
|
||||
&(t.arg_tok + t.res_tok).to_string(),
|
||||
])?;
|
||||
}
|
||||
w.flush()?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---- per-call helpers ----
|
||||
|
||||
fn session_id_from_path(path: &Path) -> String {
|
||||
path.file_stem()
|
||||
.and_then(|s| s.to_str())
|
||||
.unwrap_or("")
|
||||
.to_string()
|
||||
}
|
||||
|
||||
fn clamp_i32(n: i64) -> i32 {
|
||||
n.clamp(0, i32::MAX as i64) as i32
|
||||
}
|
||||
|
||||
fn print_buckets(
|
||||
calls: &[CallRecord],
|
||||
bucket_secs: i64,
|
||||
top: usize,
|
||||
tool_filter: Option<&str>,
|
||||
) {
|
||||
if calls.is_empty() {
|
||||
println!("(no per-call records — no toolCall/toolResult pairs found)");
|
||||
return;
|
||||
}
|
||||
|
||||
// Pick the tools to show: top-N by call count, ignoring records with ts==0
|
||||
// (events that lacked a parseable timestamp).
|
||||
let mut per_tool: HashMap<&str, i64> = HashMap::new();
|
||||
for c in calls {
|
||||
if c.ts == 0 {
|
||||
continue;
|
||||
}
|
||||
*per_tool.entry(c.tool.as_str()).or_default() += 1;
|
||||
}
|
||||
let mut ranked: Vec<(&str, i64)> = per_tool.into_iter().collect();
|
||||
ranked.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(b.0)));
|
||||
if tool_filter.is_none() {
|
||||
ranked.truncate(top);
|
||||
}
|
||||
|
||||
let bucket_human = humanize_bucket(bucket_secs);
|
||||
println!(
|
||||
"=== per-call tokens by {} bucket (rolling, no calendar alignment) ===",
|
||||
bucket_human
|
||||
);
|
||||
println!(
|
||||
"showing {} tool{} (sorted by call count). totals/calls match per-call records only;",
|
||||
ranked.len(),
|
||||
if ranked.len() == 1 { "" } else { "s" }
|
||||
);
|
||||
println!(
|
||||
"calls without a matching toolResult or without a parseable timestamp are skipped here."
|
||||
);
|
||||
|
||||
// Group calls per (tool, bucket_id).
|
||||
let mut by_bucket: HashMap<(&str, i64), Vec<&CallRecord>> = HashMap::new();
|
||||
for c in calls {
|
||||
if c.ts == 0 {
|
||||
continue;
|
||||
}
|
||||
if !ranked.iter().any(|(t, _)| *t == c.tool.as_str()) {
|
||||
continue;
|
||||
}
|
||||
let bid = c.ts.div_euclid(bucket_secs);
|
||||
by_bucket.entry((c.tool.as_str(), bid)).or_default().push(c);
|
||||
}
|
||||
|
||||
for (tool, total_calls) in ranked {
|
||||
let mut bids: Vec<i64> = by_bucket
|
||||
.keys()
|
||||
.filter(|(t, _)| *t == tool)
|
||||
.map(|(_, b)| *b)
|
||||
.collect();
|
||||
if bids.is_empty() {
|
||||
continue;
|
||||
}
|
||||
bids.sort();
|
||||
|
||||
println!();
|
||||
println!("=== {tool} ({} calls) ===", commas(total_calls));
|
||||
println!(
|
||||
"{:<22} {:>7} {:>9} {:>9} {:>9} {:>9} {:>9}",
|
||||
"window", "calls", "avg_arg", "avg_res", "avg_tot", "p50_tot", "p95_tot"
|
||||
);
|
||||
let dashes = 22 + 1 + 7 + 5 * (1 + 9);
|
||||
println!("{}", "-".repeat(dashes));
|
||||
|
||||
for bid in bids {
|
||||
let records = by_bucket.get(&(tool, bid)).expect("present");
|
||||
let n = records.len() as i64;
|
||||
let mut sum_arg = 0i64;
|
||||
let mut sum_res = 0i64;
|
||||
let mut totals: Vec<i64> = Vec::with_capacity(records.len());
|
||||
for r in records {
|
||||
sum_arg += r.arg_tok as i64;
|
||||
sum_res += r.res_tok as i64;
|
||||
totals.push(r.arg_tok as i64 + r.res_tok as i64);
|
||||
}
|
||||
let avg_arg = sum_arg as f64 / n as f64;
|
||||
let avg_res = sum_res as f64 / n as f64;
|
||||
let avg_tot = avg_arg + avg_res;
|
||||
let p50 = percentile(&mut totals.clone(), 50.0);
|
||||
let p95 = percentile(&mut totals, 95.0);
|
||||
let label = bucket_label(bid * bucket_secs, bucket_secs);
|
||||
println!(
|
||||
"{:<22} {:>7} {:>9} {:>9} {:>9} {:>9} {:>9}",
|
||||
label,
|
||||
commas(n),
|
||||
commas(avg_arg.round() as i64),
|
||||
commas(avg_res.round() as i64),
|
||||
commas(avg_tot.round() as i64),
|
||||
commas(p50.round() as i64),
|
||||
commas(p95.round() as i64),
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn humanize_bucket(bucket_secs: i64) -> String {
|
||||
if bucket_secs % (7 * 86_400) == 0 {
|
||||
let n = bucket_secs / (7 * 86_400);
|
||||
if n == 1 { "7d".into() } else { format!("{n}w") }
|
||||
} else if bucket_secs % 86_400 == 0 {
|
||||
format!("{}d", bucket_secs / 86_400)
|
||||
} else if bucket_secs % 3600 == 0 {
|
||||
format!("{}h", bucket_secs / 3600)
|
||||
} else {
|
||||
format!("{}s", bucket_secs)
|
||||
}
|
||||
}
|
||||
|
||||
fn write_calls_csv(calls: &[CallRecord], path: &str) -> Result<()> {
|
||||
let f = File::create(path).with_context(|| format!("create {path}"))?;
|
||||
let mut w = csv::Writer::from_writer(f);
|
||||
w.write_record([
|
||||
"ts_unix",
|
||||
"ts_iso",
|
||||
"session",
|
||||
"tool",
|
||||
"model",
|
||||
"arg_tok",
|
||||
"res_tok",
|
||||
"total_tok",
|
||||
])?;
|
||||
for c in calls {
|
||||
let total = c.arg_tok as i64 + c.res_tok as i64;
|
||||
w.write_record([
|
||||
&c.ts.to_string(),
|
||||
&format_iso(c.ts),
|
||||
&c.session,
|
||||
&c.tool,
|
||||
&c.model,
|
||||
&c.arg_tok.to_string(),
|
||||
&c.res_tok.to_string(),
|
||||
&total.to_string(),
|
||||
])?;
|
||||
}
|
||||
w.flush()?;
|
||||
Ok(())
|
||||
}
|
||||
@@ -1,344 +0,0 @@
|
||||
//! Shared JSONL shapes, walk helpers, tokenizer, and formatting helpers.
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use rayon::prelude::*;
|
||||
use serde::Deserialize;
|
||||
use serde_json::value::RawValue;
|
||||
use std::collections::HashMap;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::LazyLock;
|
||||
use std::sync::atomic::{AtomicU64, Ordering};
|
||||
use std::time::SystemTime;
|
||||
use tiktoken_rs::CoreBPE;
|
||||
use walkdir::WalkDir;
|
||||
|
||||
// ---- jsonl shapes ----
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub struct RawEvent {
|
||||
#[serde(rename = "type", default)]
|
||||
pub kind: String,
|
||||
#[serde(default)]
|
||||
pub message: Option<Box<RawValue>>,
|
||||
#[serde(default)]
|
||||
pub timestamp: String,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub struct Message {
|
||||
#[serde(default)]
|
||||
pub role: String,
|
||||
#[serde(default)]
|
||||
pub content: Option<Box<RawValue>>,
|
||||
#[serde(default, rename = "toolName")]
|
||||
pub tool_name: String,
|
||||
#[serde(default, rename = "toolCallId")]
|
||||
pub tool_call_id: String,
|
||||
#[serde(default)]
|
||||
pub model: String,
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
pub struct ContentItem {
|
||||
#[serde(rename = "type", default)]
|
||||
pub kind: String,
|
||||
#[serde(default)]
|
||||
pub text: String,
|
||||
#[serde(default)]
|
||||
pub thinking: String,
|
||||
#[serde(default)]
|
||||
pub name: String,
|
||||
#[serde(default)]
|
||||
pub id: String,
|
||||
#[serde(default)]
|
||||
pub arguments: Option<Box<RawValue>>,
|
||||
}
|
||||
|
||||
// ---- session walking ----
|
||||
|
||||
pub fn sessions_root() -> Result<PathBuf> {
|
||||
let home = dirs::home_dir().context("could not resolve home directory")?;
|
||||
Ok(home.join(".omp").join("agent").join("sessions"))
|
||||
}
|
||||
|
||||
pub struct WalkOpts {
|
||||
/// Keeps only paths containing any of these substrings (e.g. "2026-04-28").
|
||||
/// Empty means accept all.
|
||||
pub date_filters: Vec<String>,
|
||||
/// Keeps only the N most-recently-modified files (after the date filter).
|
||||
/// 0 means no limit.
|
||||
pub limit_most_recent: usize,
|
||||
}
|
||||
|
||||
/// Walks the sessions root and returns the matching `.jsonl` paths.
|
||||
/// With `limit_most_recent > 0` the result is sorted by mtime descending and
|
||||
/// truncated to N entries; otherwise it's lexically sorted.
|
||||
pub fn collect_sessions(opts: &WalkOpts) -> Result<Vec<PathBuf>> {
|
||||
let base = sessions_root()?;
|
||||
let need_mtime = opts.limit_most_recent > 0;
|
||||
|
||||
let mut all: Vec<(PathBuf, SystemTime)> = Vec::new();
|
||||
for entry in WalkDir::new(&base).into_iter().filter_map(Result::ok) {
|
||||
if !entry.file_type().is_file() {
|
||||
continue;
|
||||
}
|
||||
let p = entry.path();
|
||||
if p.extension().and_then(|e| e.to_str()) != Some("jsonl") {
|
||||
continue;
|
||||
}
|
||||
let path_str = p.to_string_lossy();
|
||||
if !match_date(&path_str, &opts.date_filters) {
|
||||
continue;
|
||||
}
|
||||
let mt = if need_mtime {
|
||||
entry
|
||||
.metadata()
|
||||
.ok()
|
||||
.and_then(|m| m.modified().ok())
|
||||
.unwrap_or(SystemTime::UNIX_EPOCH)
|
||||
} else {
|
||||
SystemTime::UNIX_EPOCH
|
||||
};
|
||||
all.push((p.to_path_buf(), mt));
|
||||
}
|
||||
|
||||
if need_mtime {
|
||||
all.sort_by(|a, b| b.1.cmp(&a.1));
|
||||
all.truncate(opts.limit_most_recent);
|
||||
} else {
|
||||
all.sort_by(|a, b| a.0.cmp(&b.0));
|
||||
}
|
||||
Ok(all.into_iter().map(|(p, _)| p).collect())
|
||||
}
|
||||
|
||||
fn match_date(p: &str, filters: &[String]) -> bool {
|
||||
filters.is_empty() || filters.iter().any(|d| p.contains(d))
|
||||
}
|
||||
|
||||
// ---- content helpers ----
|
||||
|
||||
pub fn parse_content(raw: &RawValue) -> Vec<ContentItem> {
|
||||
serde_json::from_str(raw.get()).unwrap_or_default()
|
||||
}
|
||||
|
||||
/// Concatenates all `text` items in a content array.
|
||||
pub fn join_text(items: &[ContentItem]) -> String {
|
||||
let mut out = String::new();
|
||||
for it in items {
|
||||
if it.kind == "text" {
|
||||
out.push_str(&it.text);
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
// ---- tokenizer (o200k_base) ----
|
||||
|
||||
static BPE: LazyLock<CoreBPE> =
|
||||
LazyLock::new(|| tiktoken_rs::o200k_base().expect("load o200k_base BPE"));
|
||||
|
||||
/// Counts tokens for `s` using the o200k_base BPE (GPT-4o / GPT-5 family).
|
||||
/// Uses the ordinary encoder so embedded `<|...|>` sequences in tool args do
|
||||
/// not trigger special-token handling.
|
||||
pub fn count_tokens(s: &str) -> usize {
|
||||
if s.is_empty() {
|
||||
return 0;
|
||||
}
|
||||
BPE.encode_ordinary(s).len()
|
||||
}
|
||||
|
||||
// ---- formatting helpers ----
|
||||
|
||||
/// Formats an integer with thousand separators.
|
||||
pub fn commas(n: i64) -> String {
|
||||
let neg = n < 0;
|
||||
let mag = if neg { (n as i128).unsigned_abs() } else { n as u128 };
|
||||
let s = mag.to_string();
|
||||
let bytes = s.as_bytes();
|
||||
let mut out = String::with_capacity(bytes.len() + bytes.len() / 3 + 1);
|
||||
if neg {
|
||||
out.push('-');
|
||||
}
|
||||
let pre = bytes.len() % 3;
|
||||
if pre > 0 {
|
||||
out.push_str(&s[..pre]);
|
||||
if bytes.len() > pre {
|
||||
out.push(',');
|
||||
}
|
||||
}
|
||||
let mut i = pre;
|
||||
while i + 3 <= bytes.len() {
|
||||
out.push_str(&s[i..i + 3]);
|
||||
if i + 3 < bytes.len() {
|
||||
out.push(',');
|
||||
}
|
||||
i += 3;
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
pub fn pct(a: i64, b: i64) -> f64 {
|
||||
if b == 0 {
|
||||
0.0
|
||||
} else {
|
||||
100.0 * a as f64 / b as f64
|
||||
}
|
||||
}
|
||||
|
||||
/// Truncates a string to at most `n` chars, replacing newlines with " | ".
|
||||
/// Adds an ellipsis when truncation occurs.
|
||||
pub fn truncate_line(s: &str, n: usize) -> String {
|
||||
let s = s.replace('\n', " | ");
|
||||
if s.chars().count() <= n {
|
||||
return s;
|
||||
}
|
||||
let mut out: String = s.chars().take(n).collect();
|
||||
out.push('…');
|
||||
out
|
||||
}
|
||||
|
||||
/// Returns map keys sorted by descending value, ties broken alphabetically.
|
||||
pub fn sorted_by_count(m: &HashMap<String, i64>) -> Vec<&String> {
|
||||
let mut keys: Vec<&String> = m.keys().collect();
|
||||
keys.sort_by(|a, b| {
|
||||
let av = m.get(a.as_str()).copied().unwrap_or(0);
|
||||
let bv = m.get(b.as_str()).copied().unwrap_or(0);
|
||||
bv.cmp(&av).then_with(|| a.cmp(b))
|
||||
});
|
||||
keys
|
||||
}
|
||||
|
||||
pub fn print_sorted(m: &HashMap<String, i64>) {
|
||||
print_sorted_indent(m, " ");
|
||||
}
|
||||
|
||||
pub fn print_sorted_indent(m: &HashMap<String, i64>, indent: &str) {
|
||||
for k in sorted_by_count(m) {
|
||||
println!("{indent}{k:<25} {}", m[k.as_str()]);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- parallel processing ----
|
||||
|
||||
/// Runs `handle(path)` in parallel across rayon workers and collects the
|
||||
/// non-`None` results into a Vec. Logs progress every `progress_every` files
|
||||
/// (set 0 to silence).
|
||||
pub fn parallel_collect<R, H>(
|
||||
paths: &[PathBuf],
|
||||
workers: usize,
|
||||
progress_every: u64,
|
||||
handle: H,
|
||||
) -> Vec<R>
|
||||
where
|
||||
R: Send,
|
||||
H: Fn(&Path) -> Option<R> + Sync,
|
||||
{
|
||||
let total = paths.len();
|
||||
let done = AtomicU64::new(0);
|
||||
|
||||
let pool = {
|
||||
let mut b = rayon::ThreadPoolBuilder::new();
|
||||
if workers > 0 {
|
||||
b = b.num_threads(workers);
|
||||
}
|
||||
b.build().expect("rayon thread pool")
|
||||
};
|
||||
|
||||
pool.install(|| {
|
||||
paths
|
||||
.par_iter()
|
||||
.filter_map(|p| {
|
||||
let r = handle(p);
|
||||
let n = done.fetch_add(1, Ordering::Relaxed) + 1;
|
||||
if progress_every > 0 && n % progress_every == 0 {
|
||||
eprintln!(" processed {n}/{total}");
|
||||
}
|
||||
r
|
||||
})
|
||||
.collect()
|
||||
})
|
||||
}
|
||||
|
||||
// ---- timestamp / bucket helpers ----
|
||||
|
||||
/// Parses an RFC3339 timestamp into unix seconds. Returns 0 on failure.
|
||||
pub fn parse_ts(s: &str) -> i64 {
|
||||
if s.is_empty() {
|
||||
return 0;
|
||||
}
|
||||
chrono::DateTime::parse_from_rfc3339(s)
|
||||
.map(|dt| dt.timestamp())
|
||||
.unwrap_or(0)
|
||||
}
|
||||
|
||||
/// Parses a bucket spec like "h", "day", "week", "month", "1h", "12h", "7d", "2w".
|
||||
/// Returns the bucket size in seconds.
|
||||
pub fn parse_bucket(spec: &str) -> anyhow::Result<i64> {
|
||||
use anyhow::bail;
|
||||
let s = spec.trim();
|
||||
let secs: i64 = match s {
|
||||
"" => bail!("empty bucket spec"),
|
||||
"hour" | "h" | "1h" => 3600,
|
||||
"day" | "d" | "1d" => 86_400,
|
||||
"week" | "w" | "1w" | "7d" => 7 * 86_400,
|
||||
"month" | "mo" | "1mo" | "30d" => 30 * 86_400,
|
||||
other => {
|
||||
let bytes = other.as_bytes();
|
||||
let last = *bytes.last().unwrap();
|
||||
let unit_secs: i64 = match last {
|
||||
b'h' => 3600,
|
||||
b'd' => 86_400,
|
||||
b'w' => 7 * 86_400,
|
||||
_ => bail!("bad bucket spec {spec:?} (use h/d/w or e.g. 7d, 12h)"),
|
||||
};
|
||||
let n: i64 = other[..other.len() - 1]
|
||||
.parse()
|
||||
.map_err(|_| anyhow::anyhow!("bad bucket count in {spec:?}"))?;
|
||||
if n <= 0 {
|
||||
bail!("bucket count must be > 0");
|
||||
}
|
||||
n * unit_secs
|
||||
}
|
||||
};
|
||||
Ok(secs)
|
||||
}
|
||||
|
||||
/// Returns a label for the bucket starting at `start_secs`.
|
||||
/// Buckets shorter than a day include the hour; multi-day buckets show start..end (exclusive).
|
||||
pub fn bucket_label(start_secs: i64, bucket_secs: i64) -> String {
|
||||
use chrono::TimeZone;
|
||||
let start = chrono::Utc.timestamp_opt(start_secs, 0).single();
|
||||
let end = chrono::Utc.timestamp_opt(start_secs + bucket_secs - 1, 0).single();
|
||||
match (start, end) {
|
||||
(Some(s), _) if bucket_secs < 86_400 => s.format("%Y-%m-%d %H:00").to_string(),
|
||||
(Some(s), _) if bucket_secs == 86_400 => s.format("%Y-%m-%d").to_string(),
|
||||
(Some(s), Some(e)) => format!(
|
||||
"{}..{}",
|
||||
s.format("%Y-%m-%d"),
|
||||
e.format("%Y-%m-%d")
|
||||
),
|
||||
_ => start_secs.to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Formats a unix timestamp as an RFC3339 string (UTC).
|
||||
pub fn format_iso(unix_secs: i64) -> String {
|
||||
use chrono::TimeZone;
|
||||
chrono::Utc
|
||||
.timestamp_opt(unix_secs, 0)
|
||||
.single()
|
||||
.map(|dt| dt.format("%Y-%m-%dT%H:%M:%SZ").to_string())
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// Computes a percentile over an integer slice. The slice is sorted in place.
|
||||
/// `p` is 0..=100.
|
||||
pub fn percentile(v: &mut [i64], p: f64) -> f64 {
|
||||
if v.is_empty() {
|
||||
return 0.0;
|
||||
}
|
||||
v.sort_unstable();
|
||||
let p = p.clamp(0.0, 100.0);
|
||||
let idx = ((p / 100.0) * (v.len() as f64 - 1.0)).round() as usize;
|
||||
v[idx.min(v.len() - 1)] as f64
|
||||
}
|
||||
@@ -1,63 +0,0 @@
|
||||
//! session-stats: ad-hoc analyses over the local agent session corpus
|
||||
//! (`~/.omp/agent/sessions/`).
|
||||
//!
|
||||
//! Subcommands:
|
||||
//!
|
||||
//! edits [-n N] [date ...] audit edit/ast_edit/write tool usage by argument schema
|
||||
//! tools [-n N] per-tool token totals across the most-recent N sessions
|
||||
//!
|
||||
//! Run with no subcommand for help.
|
||||
|
||||
mod cmd_edits;
|
||||
mod cmd_followups;
|
||||
mod cmd_tools;
|
||||
mod common;
|
||||
|
||||
use std::process::ExitCode;
|
||||
|
||||
fn usage() {
|
||||
eprintln!(
|
||||
"usage: session-stats <subcommand> [args...]
|
||||
|
||||
edits [-n N] [date prefix ...]
|
||||
audit edit-tool usage across N most-recent
|
||||
sessions (default 1000). Optional date filters
|
||||
(e.g. 2026-04-28) further narrow the set.
|
||||
tools [-n N] [--by SPEC] [--top N] [--tool NAME] [--calls-csv PATH]
|
||||
per-tool token totals across the N most-recent
|
||||
session jsonl files (default 1000). With --by,
|
||||
also bucket per-call tokens (avg + p50/p95) into
|
||||
rolling windows so you can spot regressions.
|
||||
|
||||
Token counting uses the o200k_base tokenizer (the GPT-4o / Claude-adjacent BPE).
|
||||
Walk root: ~/.omp/agent/sessions/"
|
||||
);
|
||||
}
|
||||
|
||||
fn main() -> ExitCode {
|
||||
let mut args = std::env::args().skip(1);
|
||||
let Some(cmd) = args.next() else {
|
||||
usage();
|
||||
return ExitCode::from(2);
|
||||
};
|
||||
let rest: Vec<String> = args.collect();
|
||||
let result = match cmd.as_str() {
|
||||
"edits" => cmd_edits::run(rest),
|
||||
"followups" => cmd_followups::run(rest),
|
||||
"tools" => cmd_tools::run(rest),
|
||||
"-h" | "--help" | "help" => {
|
||||
usage();
|
||||
return ExitCode::SUCCESS;
|
||||
}
|
||||
other => {
|
||||
eprintln!("unknown subcommand {other:?}\n");
|
||||
usage();
|
||||
return ExitCode::from(2);
|
||||
}
|
||||
};
|
||||
if let Err(err) = result {
|
||||
eprintln!("fatal: {err:#}");
|
||||
return ExitCode::FAILURE;
|
||||
}
|
||||
ExitCode::SUCCESS
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user