feat(session-stats): added Python session-stats analyze/sync commands

- Added a Python `analyze.py` CLI with `tools`, `edits`, and `followups` session-stats subcommands.
- Added `sync.py` ingestion with `~/.omp/stats.db`, migration SQL, and incremental JSONL resume logic.
- Replaced `stats:run` in `package.json` with `stats:sync` and new `stats:edits`, `stats:tools`, and `stats:followups` scripts.
- Removed the Rust session-stats crate files (`Cargo.toml`, `main.rs`, `common.rs`, `cmd_*.rs`) and their old command logic.
- Removed `.gitignore` ignores for `scripts/session-stats/Cargo.lock` and `scripts/session-stats/edit-analysis.csv`.
- Updated eval tool guidance to show Python `asyncio.run(...)` usage.
This commit is contained in:
can1357
2026-05-09 04:40:45 +02:00
parent 4e8d69076d
commit 78b795ad80
14 changed files with 1950 additions and 2836 deletions
-5
View File
@@ -58,11 +58,6 @@ pi-*.html
packages/coding-agent/src/internal-urls/docs-index.generated.ts
/runs/
python/omp-rpc/src/omp_rpc.egg-info/
scripts/session-stats/Cargo.lock
scripts/session-stats/edit-analysis.csv
# parallel-agent worktrees
.wt/
CPU*.md
+4 -3
View File
@@ -115,9 +115,10 @@
"ci:release:publish": "bun scripts/ci-release-publish.ts",
"bench:gen-fixtures": "bun --cwd=packages/typescript-edit-benchmark run src/generate.ts --typescript-dir /tmp/typescript-source --count-per-type 8",
"bench:edit": "bun --cwd=packages/typescript-edit-benchmark run start",
"stats:run": "cargo run --release --manifest-path scripts/session-stats/Cargo.toml --",
"stats:edits": "cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits",
"stats:tools": "cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools",
"stats:sync": "python3 scripts/session-stats/sync.py",
"stats:tools": "python3 scripts/session-stats/analyze.py tools",
"stats:edits": "python3 scripts/session-stats/analyze.py edits",
"stats:followups": "python3 scripts/session-stats/analyze.py followups",
"prepublishOnly": "bun run check",
"prepare": "bun --cwd=packages/coding-agent run generate-docs-index",
"publish": "bun run prepublishOnly && npm publish -ws --access public",
+4 -1
View File
@@ -314,7 +314,10 @@ function applyGeneratedModelPolicy(model: ApiModel<Api>): void {
model.maxTokens = copilotLimits.maxTokens;
}
if (model.api === "openai-completions" && (model.provider === "minimax-code" || model.provider === "minimax-code-cn")) {
if (
model.api === "openai-completions" &&
(model.provider === "minimax-code" || model.provider === "minimax-code-cn")
) {
model.compat = {
...model.compat,
supportsStore: false,
@@ -20,8 +20,7 @@ At least 5 equal signs on each side. Content between one header and the next (or
- Pass multiple small cells in one call.
- Define small reusable functions for individual debugging.
- Put workflow explanations in the assistant message or cell title — never inside cell code.
{{#if py}}- Python cells run inside an IPython kernel with a live event loop. Use top-level `await` directly (e.g. `await main()`); `asyncio.run(...)` raises "cannot be called from a running event loop".{{/if}}
{{#if py}}- Python cells run inside an IPython kernel with a live event loop. Use top-level `await` directly (e.g. `await main()`); `asyncio.run(…)` raises "cannot be called from a running event loop".{{/if}}
**On failure:** errors identify the failing cell (e.g., "Cell 3 failed"). Resubmit only the fixed cell (or fixed cell + remaining cells).
</instruction>
@@ -1,6 +1,6 @@
import { beforeAll, describe, expect, it } from "bun:test";
import { renderSegment } from "../src/modes/components/status-line/segments";
import type { SegmentContext } from "../src/modes/components/status-line/segments";
import { renderSegment } from "../src/modes/components/status-line/segments";
import { initTheme, theme } from "../src/modes/theme/theme";
beforeAll(async () => {
-32
View File
@@ -1,32 +0,0 @@
[package]
name = "session-stats"
version = "0.1.0"
edition = "2024"
publish = false
# Standalone crate: do not inherit from the parent workspace.
[workspace]
[[bin]]
name = "session-stats"
path = "src/main.rs"
[dependencies]
chrono = { version = "0.4", default-features = false, features = ["std", "clock"] }
anyhow = "1"
csv = "1"
dirs = "6"
rayon = "1.12"
regex = "1"
serde = { version = "1", features = ["derive"] }
serde_json = { version = "1", features = ["raw_value"] }
tiktoken-rs = "0.11"
walkdir = "2"
[profile.release]
opt-level = 3
lto = "thin"
codegen-units = 1
debug = "line-tables-only"
split-debuginfo = "off"
strip = "none"
+60 -72
View File
@@ -1,82 +1,70 @@
# session-stats
Ad-hoc analyses over the local agent session corpus
(`~/.omp/agent/sessions/`). Single Rust binary with subcommands.
## Subcommands
### `edits` — edit-tool reliability audit
Audits how agents have used the `edit` / `ast_edit` / `write` tools.
For each call we:
- detect the **argument-schema family** in use (the edit tool has shipped many
shapes over time: `oldText/newText`, `op+pos+end+lines`, `loc+content`,
`loc+splice/pre/post/sed`, etc.);
- record the locator shape and verb combination (for the current schema);
- pair the call with its `toolResult` and classify the outcome
(`success` / `truncated` / `aborted` / `fail:anchor-stale` /
`fail:no-match` / `fail:parse` / `fail:no-enclosing-block` / …).
Output: markdown-ish report on stdout plus per-call CSV at `$EDIT_ANALYSIS_CSV`
(default `./edit-analysis.csv`).
### `tools` — per-tool token budget
Aggregates token usage across the most-recent N sessions. Buckets:
- `tool ARGS` — assistant tool-call argument JSON
- `tool RESULTS` — tool result content text
- `assistant THINKING` — assistant `thinking` blocks
- `assistant TEXT` — assistant prose
- `user TEXT` — user-authored text content
Token counting uses **`o200k_base`** via `tiktoken-rs` (the GPT-4o / GPT-5
family BPE — well-defined offline and within ~5-10% of Claude's own counts in
aggregate across English/code).
Output: grand totals + per-tool breakdown sorted by total (arg+res) tokens.
Optional CSV at `$TOOL_USAGE_CSV`.
## Usage
```sh
# Edit audit on the most-recent sessions.
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits
# Edit audit on the 200 most-recent sessions.
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits -n 200
# Edit audit on a specific date.
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits 2026-04-28
# Tool token budget on the 1000 most-recent sessions.
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 1000
# Tool token budget on every jsonl on disk.
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 0
# Dump per-tool CSV alongside the report.
TOOL_USAGE_CSV=tools.csv \
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 200
```
The walk root is `~/.omp/agent/sessions/`. Subagent jsonls
(`<session-id>/<n>-<name>.jsonl`) count as their own session and are included
in the recency window independently.
(`~/.omp/agent/sessions/`). SQLite-backed; data is synced once into the same
`~/.omp/stats.db` that `packages/stats` uses, then queried by short Python
scripts.
## Layout
```
scripts/session-stats/
Cargo.toml
src/
main.rs # subcommand dispatch
common.rs # shared JSONL shapes, walk, tokenizer, formatting helpers
cmd_edits.rs # edits subcommand
cmd_tools.rs # tools subcommand
sync.py # walks ~/.omp/agent/sessions/ and populates ss_* tables
analyze.py # tools | edits | followups subcommands over the synced db
```
The crate is a standalone Cargo project (it carries its own `[workspace]`
declaration) so it does not perturb the main workspace's lockfile.
## One-time prep
```sh
pip install tiktoken
```
## Sync
```sh
bun run stats:sync # incremental
python3 scripts/session-stats/sync.py --workers 16 --full # rebuild all
python3 scripts/session-stats/sync.py --limit 200 # newest 200 only
```
The sync is incremental: per-file `mtime`, `size`, `byte_offset`, and
`parser_version` are tracked in `ss_sessions`. Re-runs only parse new bytes
and only re-tokenize / re-classify what changed. A bump of `EDIT_PARSER_VERSION`
in `sync.py` invalidates `ss_edit_*` rows on next sync.
Tokenization is `o200k_base` (GPT-4o / GPT-5 family) via tiktoken — well
within ~5–10% of Claude's BPE in aggregate.
## Schema
All tables are prefixed `ss_` to avoid collision with `packages/stats`.
|Table|Granularity|
|---|---|
|`ss_sessions`|one row per `.jsonl`; carries sync state + session metadata|
|`ss_tool_calls`|one row per `toolCall` content block (`arg_json`, `arg_tokens`)|
|`ss_tool_results`|one row per `toolResult` message (`result_text`, `result_tokens`, `is_error`)|
|`ss_assistant_msgs`|per assistant message text + thinking blobs and token counts|
|`ss_user_msgs`|per user message text and token count|
|`ss_edit_calls`|per `edit` call: `success`, `warnings`, `raw_input_len`|
|`ss_edit_sections`|per `@PATH` section in an edit; precomputed `longest_repeat_*`, `dup_anchors`|
Indexes on `(tool_name, timestamp)` and `(session_file, seq)` make per-tool
aggregations and ordered session walks cheap.
## Analyses
```sh
bun run stats:tools # per-tool token totals
bun run stats:tools -- --by d --top 8 # bucket by day, top 8 tools each
bun run stats:edits # edit-tool reliability audit
bun run stats:followups # five hashline-edit detectors
bun run stats:followups -- --max-fix 2 --min-dup 8 --show 20
```
All three accept `-n N` / `--folder SUBSTR` to scope the query.
The Rust crate that previously lived here was retired in favor of this
SQLite-backed flow. The schema persists everything the analyses used to
recompute on every run (token counts, hashline parse output, success flags),
so subsequent invocations are sub-second over the full corpus.
+861
View File
@@ -0,0 +1,861 @@
#!/usr/bin/env python3
"""
Analyses over the session-stats sqlite tables (`ss_*`) populated by sync.py.
Subcommands:
tools — per-tool token totals (port of cmd_tools.rs)
edits — edit-tool reliability audit (port of cmd_edits.rs)
followups — five hashline-edit detectors (port of cmd_followups.rs)
Each subcommand reads from ~/.omp/stats.db. Run sync.py first.
"""
from __future__ import annotations
import argparse
import json
import re
import sqlite3
import sys
from collections import Counter, defaultdict
from pathlib import Path
DB_PATH = Path.home() / ".omp" / "stats.db"
# --------------------------------------------------------------------------- #
# Shared helpers
def open_ro() -> sqlite3.Connection:
if not DB_PATH.exists():
sys.exit(f"db not found: {DB_PATH}. Run sync.py first.")
conn = sqlite3.connect(f"file:{DB_PATH}?mode=ro", uri=True)
conn.row_factory = sqlite3.Row
return conn
def commas(n: int) -> str:
return f"{n:,}"
def pct(part: int, total: int) -> float:
return 0.0 if total == 0 else (100.0 * part / total)
def truncate_line(s: str, n: int) -> str:
s = s.replace("\n", " | ")
if len(s) <= n:
return s
return s[: n - 1] + "…"
def parse_bucket(spec: str) -> int:
"""`h`,`d`,`w`,`m`,`<N>h`,`<N>d`,`<N>w` -> seconds."""
units = {"h": 3600, "d": 86400, "w": 604800, "m": 2592000}
if spec in units:
return units[spec]
if spec[-1] in units and spec[:-1].isdigit():
return int(spec[:-1]) * units[spec[-1]]
if spec == "hour":
return 3600
if spec == "day":
return 86400
if spec == "week":
return 604800
raise ValueError(f"bad --by spec: {spec}")
def percentile(values: list[int], p: float) -> float:
if not values:
return 0.0
s = sorted(values)
k = (len(s) - 1) * (p / 100.0)
lo, hi = int(k), min(int(k) + 1, len(s) - 1)
if lo == hi:
return float(s[lo])
return s[lo] + (s[hi] - s[lo]) * (k - lo)
# --------------------------------------------------------------------------- #
# `tools` — per-tool token totals (cmd_tools.rs port)
TOOLS_AGGREGATE_SQL = """
WITH per_tool AS (
SELECT
c.tool_name,
COUNT(*) AS calls,
IFNULL(SUM(c.arg_tokens), 0) AS arg_tok
FROM ss_tool_calls c
GROUP BY c.tool_name
),
per_tool_res AS (
SELECT
r.tool_name,
COUNT(*) AS results,
IFNULL(SUM(r.result_tokens),0) AS res_tok
FROM ss_tool_results r
GROUP BY r.tool_name
)
SELECT
COALESCE(p.tool_name, q.tool_name) AS tool_name,
IFNULL(p.calls, 0) AS calls,
IFNULL(q.results, 0) AS results,
IFNULL(p.arg_tok, 0) AS arg_tok,
IFNULL(q.res_tok, 0) AS res_tok
FROM per_tool p FULL OUTER JOIN per_tool_res q USING (tool_name)
ORDER BY (IFNULL(p.arg_tok, 0) + IFNULL(q.res_tok, 0)) DESC
"""
def cmd_tools(args: argparse.Namespace) -> int:
conn = open_ro()
where_session, where_args = _session_filter_clause(conn, args)
def with_session(table_alias: str) -> tuple[str, tuple]:
"""Returns ('AND <alias>.session_file IN (...)', params) or ('', ())."""
if not where_args:
return "", ()
ph = ",".join("?" * len(where_args))
return f"AND {table_alias}.session_file IN ({ph})", where_args
sf_clause_c, sf_params_c = with_session("c")
sf_clause_r, sf_params_r = with_session("r")
sf_clause_a, sf_params_a = with_session("a")
sf_clause_u, sf_params_u = with_session("u")
# Grand totals (each subquery applies its own session filter).
grand = conn.execute(
f"""
SELECT
(SELECT IFNULL(SUM(c.arg_tokens),0) FROM ss_tool_calls c WHERE 1=1 {sf_clause_c}) AS tool_args,
(SELECT IFNULL(SUM(r.result_tokens),0) FROM ss_tool_results r WHERE 1=1 {sf_clause_r}) AS tool_res,
(SELECT IFNULL(SUM(a.thinking_tokens),0) FROM ss_assistant_msgs a WHERE 1=1 {sf_clause_a}) AS thinking,
(SELECT IFNULL(SUM(a.text_tokens),0) FROM ss_assistant_msgs a WHERE 1=1 {sf_clause_a}) AS asst_text,
(SELECT IFNULL(SUM(u.text_tokens),0) FROM ss_user_msgs u WHERE 1=1 {sf_clause_u}) AS user_text,
(SELECT COUNT(*) FROM ss_tool_calls c WHERE 1=1 {sf_clause_c}) AS n_calls,
(SELECT COUNT(*) FROM ss_tool_results r WHERE 1=1 {sf_clause_r}) AS n_results
""",
sf_params_c + sf_params_r + sf_params_a + sf_params_a + sf_params_u + sf_params_c + sf_params_r,
).fetchone()
n_sessions = conn.execute(
f"SELECT COUNT(*) FROM ss_sessions {where_session}", where_args
).fetchone()[0]
g = grand
grand_total = g["tool_args"] + g["tool_res"] + g["thinking"] + g["asst_text"] + g["user_text"]
print("=== grand totals ===")
print(f"sessions: {commas(n_sessions)}")
print(f"tool calls / results: {commas(g['n_calls'])} / {commas(g['n_results'])}")
print(f"tool ARGS tokens: {commas(g['tool_args']):>14} ({pct(g['tool_args'], grand_total):5.1f}%)")
print(f"tool RESULTS tokens: {commas(g['tool_res']):>14} ({pct(g['tool_res'], grand_total):5.1f}%)")
print(f"assistant THINKING: {commas(g['thinking']):>14} ({pct(g['thinking'], grand_total):5.1f}%)")
print(f"assistant TEXT: {commas(g['asst_text']):>14} ({pct(g['asst_text'], grand_total):5.1f}%)")
print(f"user TEXT: {commas(g['user_text']):>14} ({pct(g['user_text'], grand_total):5.1f}%)")
print(f"total: {commas(grand_total):>14}")
# Per-tool table.
rows = conn.execute(
f"""
WITH per_tool AS (
SELECT c.tool_name, COUNT(*) AS calls, IFNULL(SUM(c.arg_tokens),0) AS arg_tok
FROM ss_tool_calls c WHERE 1=1 {sf_clause_c}
GROUP BY c.tool_name
),
per_tool_res AS (
SELECT r.tool_name, COUNT(*) AS results, IFNULL(SUM(r.result_tokens),0) AS res_tok
FROM ss_tool_results r WHERE 1=1 {sf_clause_r}
GROUP BY r.tool_name
)
SELECT
COALESCE(p.tool_name, q.tool_name) AS tool_name,
IFNULL(p.calls,0) AS calls, IFNULL(q.results,0) AS results,
IFNULL(p.arg_tok,0) AS arg_tok, IFNULL(q.res_tok,0) AS res_tok
FROM per_tool p FULL OUTER JOIN per_tool_res q USING (tool_name)
ORDER BY (IFNULL(p.arg_tok,0) + IFNULL(q.res_tok,0)) DESC
""",
sf_params_c + sf_params_r,
).fetchall()
print("\n=== per-tool tokens ===")
print(f"{'tool':<24} {'calls':>7} {'args':>14} {'results':>14} {'total':>14}")
print("-" * 78)
for r in rows:
total = r["arg_tok"] + r["res_tok"]
print(
f"{r['tool_name']:<24} {r['calls']:>7} "
f"{commas(r['arg_tok']):>14} {commas(r['res_tok']):>14} {commas(total):>14}"
)
if args.by:
bucket = parse_bucket(args.by)
_print_buckets(conn, bucket, args.top, args.tool)
return 0
def _session_filter_clause(conn, args) -> tuple[str, tuple]:
"""Builds an optional WHERE clause for session_file filtering by --limit / --folder."""
clauses, params = [], []
if args.folder:
clauses.append("folder LIKE ?")
params.append(f"%{args.folder}%")
if args.limit > 0:
# Resolve to a concrete session_file IN (...) so other tables can reuse it.
rows = conn.execute(
f"""
SELECT session_file FROM ss_sessions
{('WHERE ' + ' AND '.join(clauses)) if clauses else ''}
ORDER BY mtime DESC LIMIT ?
""",
(*params, args.limit),
).fetchall()
files = [r[0] for r in rows]
if not files:
return ("WHERE 0", ())
placeholders = ",".join("?" * len(files))
return (f"WHERE session_file IN ({placeholders})", tuple(files))
if clauses:
return ("WHERE " + " AND ".join(clauses), tuple(params))
return ("", ())
def _print_buckets(conn, bucket_secs: int, top: int, tool_filter: str | None) -> None:
where = "WHERE c.tool_name = ?" if tool_filter else ""
params = (tool_filter,) if tool_filter else ()
rows = conn.execute(
f"""
SELECT
(c.timestamp / 1000 / ?) * ? AS bucket,
c.tool_name,
COUNT(*) AS calls,
IFNULL(SUM(c.arg_tokens), 0) AS arg_tok,
IFNULL(SUM(r.result_tokens), 0) AS res_tok
FROM ss_tool_calls c
LEFT JOIN ss_tool_results r
ON r.session_file = c.session_file AND r.call_id = c.call_id
{where}
GROUP BY bucket, c.tool_name
ORDER BY bucket DESC
""",
(bucket_secs, bucket_secs) + params,
).fetchall()
by_bucket: dict[int, list[sqlite3.Row]] = defaultdict(list)
for r in rows:
by_bucket[r["bucket"]].append(r)
print(f"\n=== per-tool tokens, bucketed by {bucket_secs}s "
f"({'all tools' if not tool_filter else tool_filter}) ===")
for bucket in sorted(by_bucket.keys(), reverse=True)[:20]:
from datetime import datetime, timezone
label = datetime.fromtimestamp(bucket, tz=timezone.utc).strftime("%Y-%m-%d %H:%MZ")
print(f"\n[{label}]")
ranked = sorted(by_bucket[bucket], key=lambda r: -(r["arg_tok"] + r["res_tok"]))
for r in ranked[:top]:
tot = r["arg_tok"] + r["res_tok"]
print(f" {r['tool_name']:<22} {r['calls']:>5}c "
f"args={commas(r['arg_tok']):>12} res={commas(r['res_tok']):>12} "
f"tot={commas(tot):>12}")
# --------------------------------------------------------------------------- #
# `edits` — edit-tool reliability audit (cmd_edits.rs port)
_RE_TRUNCATED = re.compile(r"\[Output truncated", re.I)
_RE_ABORTED = re.compile(
r"Tool execution was aborted|Request was aborted|cancelled|canceled by user", re.I
)
_RE_SUCCESS = re.compile(
r"^(Updated|Successfully (wrote|replaced|edited|deleted|inserted)|Replaced|"
r"Applied|Deleted|Created|Wrote|edit applied|Edited|Inserted|OK\b)",
re.I,
)
_RE_ANCHOR_STALE = re.compile(
r"(Edit rejected:.*line[s]? .* changed since the last read|"
r"line[s]? ha(s|ve) changed since last read)",
re.I,
)
_RE_ANCHOR_MISSING = re.compile(
r"anchor .* (not found|unknown|missing)|loc requires the full anchor", re.I
)
_RE_NO_ENCLOSING = re.compile(r"No enclosing .* block", re.I)
_RE_PARSE_ERROR = re.compile(r"parse|syntax error|unbalanced|unexpected token", re.I)
_RE_SSR_NO_MATCH = re.compile(
r"0 matches|no replacements|no match found|No replacements made|Failed to find expected lines",
re.I,
)
_RE_FILE_NOT_READ = re.compile(r"must be read first|has not been read|not yet read", re.I)
_RE_FILE_CHANGED = re.compile(r"file has been (modified|changed) externally", re.I)
_RE_PERM_DENIED = re.compile(r"permission denied|not allowed", re.I)
_RE_GENERIC_REJECTED = re.compile(r"\b(rejected|failed|error|invalid)\b", re.I)
def classify_edit_result(text: str) -> str:
t = (text or "").strip()
if not t:
return "empty"
first = t.split("\n", 1)[0]
if _RE_TRUNCATED.search(first):
return "truncated"
if _RE_ABORTED.search(t):
return "aborted"
if _RE_SUCCESS.match(first):
return "success"
if _RE_ANCHOR_STALE.search(t):
return "fail:anchor-stale"
if _RE_NO_ENCLOSING.search(t):
return "fail:no-enclosing-block"
if _RE_ANCHOR_MISSING.search(t):
return "fail:anchor-missing"
if _RE_PARSE_ERROR.search(t):
return "fail:parse"
if _RE_SSR_NO_MATCH.search(t):
return "fail:no-match"
if _RE_FILE_NOT_READ.search(t):
return "fail:file-not-read"
if _RE_FILE_CHANGED.search(t):
return "fail:file-changed"
if _RE_PERM_DENIED.search(t):
return "fail:perm"
if _RE_GENERIC_REJECTED.search(first):
return "fail:other"
return "unknown"
_ANCHOR_BARE = re.compile(r"^[a-zA-Z]?[0-9]+[a-z]{2}$")
def _detect_edit_format(tool_name: str, args_obj: dict | None) -> str:
if tool_name == "write":
return "write"
if tool_name == "ast_edit":
return "ast_edit"
if not isinstance(args_obj, dict):
return "unknown"
has = lambda k: k in args_obj # noqa: E731
if has("oldText") and has("newText"):
return "oldText/newText"
if has("old_text") and has("new_text"):
return "old_text/new_text"
if has("diff") and has("op"):
return "diff+op"
if has("diff") and has("operation"):
return "diff+operation"
if has("diff"):
return "diff"
if has("replace") or has("insert"):
return "replace/insert"
if has("input") and isinstance(args_obj.get("input"), str):
return "hashline"
edits = args_obj.get("edits")
if isinstance(edits, list) and edits and isinstance(edits[0], dict):
first = edits[0]
fh = lambda k: k in first # noqa: E731
if fh("loc") and (fh("splice") or fh("pre") or fh("post") or fh("sed")):
return "loc+splice/pre/post/sed"
if fh("loc") and fh("content"):
return "loc+content"
if fh("set_line"):
return "set_line"
if fh("insert_after"):
return "insert_after"
if fh("op") and fh("pos") and fh("end") and fh("lines"):
return "op+pos+end+lines"
if fh("op") and fh("pos") and fh("lines"):
return "op+pos+lines"
if fh("op") and fh("sel") and fh("content"):
return "op+sel+content"
if fh("all") and (fh("new_text") or fh("old_text")):
return "per-edit:old_text/new_text"
return "edits[" + ",".join(sorted(first.keys())) + "]"
return ",".join(sorted(args_obj.keys()))
def _loc_shape(loc: str) -> str:
if not loc:
return "empty"
if loc == "$":
return "$file"
if ":" in loc and not loc.startswith("$"):
rest = loc.rsplit(":", 1)[1]
else:
rest = loc
if rest.startswith("(") and rest.endswith(")"):
return "bracket-(body)"
if rest.startswith("[") and rest.endswith("]"):
return "bracket-[block]"
if rest.startswith("(") or rest.startswith("["):
return "bracket-tail"
if rest.endswith(")") or rest.endswith("]"):
return "bracket-head"
if _ANCHOR_BARE.match(rest):
return "bare-anchor"
return "other"
def _classify_edit_args(tool_name: str, args_obj: dict | None) -> tuple[str, list[str], list[str]]:
"""Returns (format, verbs, loc_shapes)."""
fmt = _detect_edit_format(tool_name, args_obj)
verbs: list[str] = []
loc_shapes: list[str] = []
if tool_name == "write":
verbs.append("write")
elif tool_name == "edit" and isinstance(args_obj, dict):
edits = args_obj.get("edits")
if isinstance(edits, list):
for op in edits:
if not isinstance(op, dict):
continue
loc_val = op.get("loc")
loc_shapes.append(_loc_shape(loc_val if isinstance(loc_val, str) else ""))
v: list[str] = []
if op.get("splice"):
v.append("splice")
if op.get("pre"):
v.append("pre")
if op.get("post"):
v.append("post")
if op.get("sed"):
v.append("sed")
if not v:
v.append("none")
verbs.append("+".join(v))
return fmt, verbs, loc_shapes
def cmd_edits(args: argparse.Namespace) -> int:
conn = open_ro()
where_session, where_args = _session_filter_clause(conn, args)
sf_clause = "AND c.session_file IN (" + ",".join("?" * len(where_args)) + ")" if where_args else ""
rows = conn.execute(
f"""
SELECT
c.session_file, c.call_id, c.tool_name, c.arg_json, c.timestamp,
r.result_text, r.is_error
FROM ss_tool_calls c
LEFT JOIN ss_tool_results r
ON r.session_file = c.session_file AND r.call_id = c.call_id
WHERE c.tool_name IN ('edit','ast_edit','write') {sf_clause}
ORDER BY c.timestamp
""",
where_args,
).fetchall()
if not rows:
print("no edit-family tool calls found")
return 0
by_tool: Counter = Counter()
by_format: Counter = Counter()
status_by_tool: dict[str, Counter] = defaultdict(Counter)
status_by_format: dict[str, Counter] = defaultdict(Counter)
verb_count: Counter = Counter()
loc_count: Counter = Counter()
fails_by_verb: dict[str, Counter] = defaultdict(Counter)
fails_by_loc: dict[str, Counter] = defaultdict(Counter)
sessions = set()
failed_samples: list[tuple[str, str, list[str], list[str], str]] = []
for r in rows:
sessions.add(r["session_file"])
tool = r["tool_name"]
try:
args_obj = json.loads(r["arg_json"]) if r["arg_json"] else None
except Exception:
args_obj = None
fmt, verbs, locs = _classify_edit_args(tool, args_obj)
status = classify_edit_result(r["result_text"] or "")
by_tool[tool] += 1
by_format[fmt] += 1
status_by_tool[tool][status] += 1
status_by_format[fmt][status] += 1
for v in verbs:
verb_count[v] += 1
fails_by_verb[v][status] += 1
for l in locs:
loc_count[l] += 1
fails_by_loc[l][status] += 1
if status.startswith("fail") and len(failed_samples) < 8:
text = r["result_text"] or ""
first = text.split("\n\n", 1)[0]
failed_samples.append((tool, status, verbs, locs, truncate_line(first, 220)))
print("# Edit-tool usage")
print(f"\nTotal tool calls: {len(rows)} (across {len(sessions)} sessions)")
_print_counter("\n## By tool", by_tool)
print("\n## Outcome by tool")
for tool in sorted(by_tool):
print(f"\n {tool} ({by_tool[tool]} calls):")
for st, n in sorted(status_by_tool[tool].items(), key=lambda kv: -kv[1]):
print(f" {st:<28} {n}")
_print_counter("\n## edit verb distribution (per sub-edit)", verb_count)
_print_counter("\n## edit locator shape distribution", loc_count)
print("\n## Failure rate per verb shape")
for v, _ in verb_count.most_common():
total, failed = _fail_totals(fails_by_verb[v])
print(f" {v:<20} {failed}/{total} failed ({pct(failed, total):.0f}%)")
print("\n## Failure rate per locator shape")
for l, _ in loc_count.most_common():
total, failed = _fail_totals(fails_by_loc[l])
print(f" {l:<20} {failed}/{total} failed ({pct(failed, total):.0f}%)")
_print_counter("\n## edit-tool argument-format usage", by_format)
print("\n## Failure rate per argument format")
for f, _ in by_format.most_common():
total, failed = _fail_totals(status_by_format[f])
print(f" {f:<32} {failed:>6}/{total:<6} failed ({pct(failed, total):.0f}%)")
print("\n## Failure breakdown per top format")
for f, _ in by_format.most_common(8):
print(f"\n {f} ({by_format[f]} total)")
for st, n in sorted(status_by_format[f].items(), key=lambda kv: -kv[1]):
print(f" {st:<28} {n}")
print("\n## Sample failed edits")
for tool, status, verbs, locs, snippet in failed_samples:
print(f"\n— {tool} [{status}] verbs={verbs} loc={locs}\n result: {snippet}")
return 0
def _print_counter(header: str, c: Counter) -> None:
print(header)
for k, v in c.most_common():
print(f" {k:<32} {v}")
def _fail_totals(c: Counter) -> tuple[int, int]:
total = sum(c.values())
failed = sum(v for k, v in c.items() if k.startswith("fail"))
return total, failed
# --------------------------------------------------------------------------- #
# `followups` — five hashline-edit detectors (cmd_followups.rs port)
_CLOSER_LINE_RE = re.compile(r"^\s*[\])}]+[;,]?\s*$")
_FIX_PATTERNS = [
"remove-single-closer",
"add-single-closer",
"one-line-modify",
"pure-delete-1",
"pure-insert-1",
"small-other",
]
_FIX_PRIORITY = {p: i for i, p in enumerate(_FIX_PATTERNS)}
def _classify_fix(deleted: int, payload_lines: list[str]) -> str:
closer = len(payload_lines) == 1 and bool(_CLOSER_LINE_RE.match(payload_lines[0]))
if deleted == 0 and closer:
return "add-single-closer"
if deleted == 1 and not payload_lines:
return "remove-single-closer"
if deleted == 0 and len(payload_lines) == 1:
return "pure-insert-1"
if deleted == 1 and len(payload_lines) == 1:
return "one-line-modify"
if deleted >= 1 and not payload_lines:
return "pure-delete-1"
return "small-other"
def cmd_followups(args: argparse.Namespace) -> int:
conn = open_ro()
where_session, where_args = _session_filter_clause(conn, args)
sf_clause = "AND c.session_file IN (" + ",".join("?" * len(where_args)) + ")" if where_args else ""
# All edit calls + their sections, ordered per session.
call_rows = conn.execute(
f"""
SELECT c.session_file, c.call_id, c.seq, c.timestamp, c.raw_input_len,
c.success, c.warnings
FROM ss_edit_calls c
WHERE 1=1 {sf_clause}
ORDER BY c.session_file, c.seq
""",
where_args,
).fetchall()
section_rows = conn.execute(
f"""
SELECT s.session_file, s.call_id, s.seq, s.section_idx, s.target_file,
s.op_count, s.deleted_lines, s.payload_count, s.change_size,
s.min_line, s.max_line, s.payload_blocks,
s.longest_repeat_len, s.longest_repeat_block_idx,
s.longest_repeat_sample, s.dup_anchors
FROM ss_edit_sections s
JOIN ss_edit_calls c USING (session_file, call_id)
WHERE 1=1 {sf_clause.replace('c.session_file', 's.session_file')}
ORDER BY s.session_file, s.seq, s.section_idx
""",
where_args,
).fetchall()
# Index sections by (session_file, call_id).
sec_by_call: dict[tuple[str, str], list[sqlite3.Row]] = defaultdict(list)
for s in section_rows:
sec_by_call[(s["session_file"], s["call_id"])].append(s)
# Build per-(session, target_file) ordered list of (call_meta, section).
by_session_file: dict[tuple[str, str], list[tuple[sqlite3.Row, sqlite3.Row]]] = defaultdict(list)
total_successful_edits = 0
warning_hits: list[dict] = []
payload_dups: list[dict] = []
anchor_dups: list[dict] = []
for c in call_rows:
if c["success"] == 1:
total_successful_edits += 1
warns = json.loads(c["warnings"] or "[]")
seen: list[str] = []
for w in warns:
if w not in seen:
seen.append(w)
if seen:
files_csv = ",".join(
s["target_file"] for s in sec_by_call.get((c["session_file"], c["call_id"]), [])
)
for kind in seen:
warning_hits.append({
"session": c["session_file"],
"call_id": c["call_id"],
"kind": kind,
"files": files_csv,
"input_len": c["raw_input_len"],
})
if c["success"] != 1:
continue
for s in sec_by_call.get((c["session_file"], c["call_id"]), []):
by_session_file[(c["session_file"], s["target_file"])].append((c, s))
# Payload self-dup
if s["longest_repeat_len"] >= 4:
payload_dups.append({
"session": c["session_file"],
"call_id": c["call_id"],
"file": s["target_file"],
"block_len": _block_len(s, s["longest_repeat_block_idx"]),
"repeat_len": s["longest_repeat_len"],
"sample": s["longest_repeat_sample"] or "",
})
# Anchor reuse
try:
dups = json.loads(s["dup_anchors"] or "[]")
except Exception:
dups = []
for d in dups:
anchor_dups.append({
"session": c["session_file"],
"call_id": c["call_id"],
"files": d[2] if len(d) > 2 else s["target_file"],
"anchor": d[0],
"count": d[1],
})
# (1) small-fix follow-ups + (3) same-locus re-edits.
fix_hits: list[dict] = []
locus_hits: list[dict] = []
for (session, target), entries in by_session_file.items():
for i in range(len(entries) - 1):
ac, asec = entries[i]
bc, bsec = entries[i + 1]
if ac["call_id"] == bc["call_id"]:
continue
first_size = asec["change_size"]
second_size = bsec["change_size"]
gap = max(0, (bc["timestamp"] - ac["timestamp"]) // 1000)
# (1) small fix on big edit
if 0 < second_size <= args.max_fix and first_size > 2:
pl = _flatten_payload(bsec)
pattern = _classify_fix(bsec["deleted_lines"], pl)
summary = _render_section_summary(bsec, pl)
fix_hits.append({
"session": session, "file": target,
"first_call_id": ac["call_id"], "second_call_id": bc["call_id"],
"first_size": first_size, "first_input_len": ac["raw_input_len"],
"second_size": second_size,
"pattern": pattern, "second_summary": summary, "gap_secs": gap,
})
# (3) same-locus re-edit (both > max-fix)
if (
first_size > 2 and second_size > args.max_fix
and asec["min_line"] is not None and asec["max_line"] is not None
and bsec["min_line"] is not None and bsec["max_line"] is not None
):
a_lo, a_hi = asec["min_line"], asec["max_line"]
b_lo, b_hi = bsec["min_line"], bsec["max_line"]
if max(a_lo, b_lo) <= min(a_hi, b_hi):
locus_hits.append({
"session": session, "file": target,
"first_call_id": ac["call_id"], "second_call_id": bc["call_id"],
"first_range": (a_lo, a_hi), "second_range": (b_lo, b_hi),
"first_size": first_size, "second_size": second_size, "gap_secs": gap,
})
if args.max_gap > 0:
fix_hits = [h for h in fix_hits if h["gap_secs"] <= args.max_gap]
locus_hits = [h for h in locus_hits if h["gap_secs"] <= args.max_gap]
if args.pattern:
fix_hits = [h for h in fix_hits if h["pattern"] == args.pattern]
fix_hits.sort(key=lambda h: (_FIX_PRIORITY.get(h["pattern"], 99), -h["first_input_len"]))
locus_hits.sort(key=lambda h: -h["first_size"])
payload_dups = [p for p in payload_dups if p["repeat_len"] >= args.min_dup]
payload_dups.sort(key=lambda p: -p["repeat_len"])
anchor_dups.sort(key=lambda a: -a["count"])
# ---- print ----
by_pattern = Counter(h["pattern"] for h in fix_hits)
print("=== heuristic followup hits ===")
print(f"total hits: {commas(len(fix_hits))}")
print("\nby pattern:")
for label, n in by_pattern.most_common():
print(f" {label:<22} {n:>6}")
shown = min(args.show, len(fix_hits))
print(f"\n=== top {shown} hits (by first-edit input size) ===")
for h in fix_hits[:shown]:
print(
f"[{h['pattern']}] {h['file']} first={h['first_size']}L "
f"({h['first_input_len']}B) → second={h['second_size']}L gap={h['gap_secs']}s"
)
print(f" session={h['session']}")
print(f" first_call={h['first_call_id']} second_call={h['second_call_id']}")
print(f" fix: {h['second_summary']}")
print("\n=== tool self-corrections ===")
print("(emitted as warnings on otherwise-successful edits — the tool caught what the model wrote)")
by_kind = Counter(w["kind"] for w in warning_hits)
for kind, n in by_kind.most_common():
suffix = (f" ({pct(n, total_successful_edits):.2f}% of "
f"{commas(total_successful_edits)} successful edits)") if total_successful_edits else ""
print(f" {kind:<16} {n:>6}{suffix}")
warn_show = min(args.show, len(warning_hits))
if warn_show:
print(f"\n--- top {warn_show} self-correction examples (by input size) ---")
sorted_warns = sorted(warning_hits, key=lambda w: -w["input_len"])
for w in sorted_warns[:warn_show]:
print(f"[{w['kind']}] {truncate_line(w['files'], 80)} ({w['input_len']}B)")
print(f" session={w['session']} call={w['call_id']}")
print("\n=== same-locus re-edits (overlapping anchor ranges, both > max-fix) ===")
print(f"hits: {commas(len(locus_hits))}")
locus_show = min(args.show, len(locus_hits))
for h in locus_hits[:locus_show]:
print(
f"{h['file']} first={h['first_range'][0]}..{h['first_range'][1]} "
f"({h['first_size']}L) → second={h['second_range'][0]}..{h['second_range'][1]} "
f"({h['second_size']}L) gap={h['gap_secs']}s"
)
print(f" session={h['session']}")
print(f" first_call={h['first_call_id']} second_call={h['second_call_id']}")
print("\n=== payload self-duplication (model pasted same N-line chunk twice in one payload) ===")
print(
f"hits with repeat_len >= {args.min_dup}: {commas(len(payload_dups))} "
f"({pct(len(payload_dups), total_successful_edits):.2f}% of "
f"{commas(total_successful_edits)} successful edits)"
)
for p in payload_dups[: args.show]:
print(f"k={p['repeat_len']} block={p['block_len']}L {p['file']}")
print(f" session={p['session']} call={p['call_id']}")
print(f" sample: {p['sample']}")
print("\n=== same-anchor reused by multiple ops in one input ===")
print(f"hits: {commas(len(anchor_dups))}")
for a in anchor_dups[: args.show]:
print(
f"anchor {a['anchor']} referenced {a['count']}x "
f"files={truncate_line(a['files'], 80)}"
)
print(f" session={a['session']} call={a['call_id']}")
return 0
def _flatten_payload(section_row: sqlite3.Row) -> list[str]:
try:
blocks = json.loads(section_row["payload_blocks"] or "[]")
except Exception:
return []
out: list[str] = []
for b in blocks:
out.extend(b)
return out
def _block_len(section_row: sqlite3.Row, idx: int | None) -> int:
if idx is None:
return 0
try:
blocks = json.loads(section_row["payload_blocks"] or "[]")
except Exception:
return 0
if 0 <= idx < len(blocks):
return len(blocks[idx])
return 0
def _render_section_summary(section_row: sqlite3.Row, payload_lines: list[str]) -> str:
bits: list[str] = []
deleted = section_row["deleted_lines"]
if deleted > 0:
bits.append(f"-{deleted}")
if payload_lines:
bits.append(f"+{len(payload_lines)}")
out = " / ".join(bits)
if payload_lines:
out += f" | {truncate_line(payload_lines[0], 80)}"
return out
# --------------------------------------------------------------------------- #
# Entry point
def main() -> int:
ap = argparse.ArgumentParser(description="session-stats analyses (sqlite-backed)")
sub = ap.add_subparsers(dest="cmd", required=True)
common = argparse.ArgumentParser(add_help=False)
common.add_argument("-n", "--limit", type=int, default=0,
help="restrict to N most-recent sessions (0 = all)")
common.add_argument("--folder", default=None,
help="filter sessions whose folder contains this substring")
ap_tools = sub.add_parser("tools", parents=[common], help="per-tool token totals")
ap_tools.add_argument("--by", default=None,
help="bucket per-call data: h, d, w, m, or <N>{h,d,w}")
ap_tools.add_argument("--top", type=int, default=10, help="top tools per bucket")
ap_tools.add_argument("--tool", default=None, help="restrict bucket view to one tool")
ap_tools.set_defaults(func=cmd_tools)
ap_edits = sub.add_parser("edits", parents=[common], help="edit reliability audit")
ap_edits.set_defaults(func=cmd_edits)
ap_fu = sub.add_parser("followups", parents=[common], help="hashline edit followup detectors")
ap_fu.add_argument("--max-fix", type=int, default=2)
ap_fu.add_argument("--max-gap", type=int, default=0, help="cap seconds between paired edits")
ap_fu.add_argument("--min-dup", type=int, default=8, help="min payload-dup repeat length")
ap_fu.add_argument("--pattern", default=None, help="filter (1) to a single FixPattern")
ap_fu.add_argument("--show", type=int, default=60)
ap_fu.set_defaults(func=cmd_followups)
args = ap.parse_args()
return args.func(args)
if __name__ == "__main__":
sys.exit(main())
-639
View File
@@ -1,639 +0,0 @@
//! `edits` subcommand — audits how agents have used the edit / ast_edit /
//! write tools across session jsonl files.
//!
//! For every edit-family toolCall we record:
//! - which argument-schema family is in use (the edit tool has shipped many
//! shapes over time: oldText/newText, op+pos+end+lines, loc+content,
//! loc+splice/pre/post/sed, etc.);
//! - the locator shape and verb combination (for the current
//! loc+splice/pre/post/sed schema);
//! then pair the call with its toolResult and classify success / failure
//! category (anchor-stale, no-match, parse, etc.).
//!
//! Output: markdown-ish report on stdout plus a per-call CSV at
//! `$EDIT_ANALYSIS_CSV` (default `./edit-analysis.csv`).
use crate::common::*;
use anyhow::{Context, Result, bail};
use regex::Regex;
use serde::Deserialize;
use serde_json::Value;
use serde_json::value::RawValue;
use std::collections::HashMap;
use std::fs::File;
use std::io::{BufRead, BufReader};
use std::path::Path;
use std::sync::LazyLock;
#[derive(Default, Clone)]
struct EditEntry {
file: String,
call_id: String,
tool_name: String,
num_edits: i64,
/// splice/pre/post/sed per sub-edit
verbs: Vec<String>,
/// bare-anchor / bracket-(body) / ...
loc_shapes: Vec<String>,
/// edit-tool argument schema family
format: String,
result_raw: String,
/// "success" / "fail:..." / etc.
status: String,
}
#[derive(Deserialize, Default)]
struct EditOp {
#[serde(default)]
loc: String,
#[serde(default)]
splice: Option<Box<RawValue>>,
#[serde(default)]
pre: Option<Box<RawValue>>,
#[serde(default)]
post: Option<Box<RawValue>>,
#[serde(default)]
sed: Option<Box<RawValue>>,
}
#[derive(Deserialize, Default)]
struct EditArgs {
#[serde(default)]
edits: Vec<EditOp>,
#[serde(default)]
ops: Vec<Box<RawValue>>,
}
pub fn run(args: Vec<String>) -> Result<()> {
let mut limit: usize = 1_000;
let mut workers: usize = 0;
let mut date_filters: Vec<String> = Vec::new();
let mut iter = args.into_iter();
while let Some(a) = iter.next() {
match a.as_str() {
"-n" => {
limit = iter
.next()
.context("-n requires a value")?
.parse()
.context("-n value")?;
}
"-j" => {
workers = iter
.next()
.context("-j requires a value")?
.parse()
.context("-j value")?;
}
"-h" | "--help" => {
eprintln!(
"usage: session-stats edits [-n N] [-j workers] [date prefix ...]"
);
return Ok(());
}
other if other.starts_with('-') => bail!("unknown flag: {other}"),
other => date_filters.push(other.to_string()),
}
}
let files = collect_sessions(&WalkOpts {
date_filters,
limit_most_recent: limit,
})?;
eprintln!("scanning {} session files", files.len());
let mut entries: Vec<EditEntry> = parallel_collect(&files, workers, 5_000, |p| {
Some(process_file(p))
})
.into_iter()
.flatten()
.collect();
// Stable ordering for sample output.
entries.sort_by(|a, b| a.file.cmp(&b.file));
report_edits(&entries);
write_csv(&entries)?;
Ok(())
}
fn process_file(path: &Path) -> Vec<EditEntry> {
let f = match File::open(path) {
Ok(f) => f,
Err(e) => {
eprintln!("open {}: {e}", path.display());
return Vec::new();
}
};
let reader = BufReader::with_capacity(64 * 1024, f);
let path_str = path.to_string_lossy().into_owned();
let mut calls: HashMap<String, EditEntry> = HashMap::new();
let mut order: Vec<String> = Vec::new();
for line in reader.lines() {
let Ok(line) = line else { continue };
if line.is_empty() {
continue;
}
let Ok(ev) = serde_json::from_str::<RawEvent>(&line) else {
continue;
};
if ev.kind != "message" {
continue;
}
let Some(msg_raw) = ev.message else { continue };
let Ok(m) = serde_json::from_str::<Message>(msg_raw.get()) else {
continue;
};
let Some(content_raw) = m.content else { continue };
let items = parse_content(&content_raw);
match m.role.as_str() {
"assistant" => {
for it in items {
if it.kind != "toolCall" || !is_edit_tool(&it.name) {
continue;
}
let raw = it.arguments.as_deref();
let mut e = classify_edit_args(&it.name, raw);
e.file.clone_from(&path_str);
e.call_id.clone_from(&it.id);
e.tool_name.clone_from(&it.name);
let id = it.id.clone();
calls.insert(id.clone(), e);
order.push(id);
}
}
"toolResult" => {
if !is_edit_tool(&m.tool_name) {
continue;
}
let Some(e) = calls.get_mut(&m.tool_call_id) else {
continue;
};
let text = join_text(&items);
e.status = classify_edit_result(&text);
e.result_raw = text;
}
_ => {}
}
}
let mut out: Vec<EditEntry> = Vec::with_capacity(order.len());
for id in order {
if let Some(e) = calls.remove(&id) {
out.push(e);
}
}
out
}
fn is_edit_tool(name: &str) -> bool {
matches!(
name.to_ascii_lowercase().as_str(),
"edit" | "ast_edit" | "write"
)
}
// ---- argument classification ----
static ANCHOR_BARE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"^[a-zA-Z]?[0-9]+[a-z]{2}$").expect("anchor_bare"));
fn classify_edit_args(name: &str, raw: Option<&RawValue>) -> EditEntry {
let mut e = EditEntry {
format: detect_edit_format(name, raw),
..EditEntry::default()
};
let lname = name.to_ascii_lowercase();
match lname.as_str() {
"edit" => {
let a: EditArgs = raw
.and_then(|r| serde_json::from_str(r.get()).ok())
.unwrap_or_default();
e.num_edits = a.edits.len() as i64;
for op in &a.edits {
e.loc_shapes.push(loc_shape(&op.loc));
let mut verbs: Vec<&str> = Vec::new();
if !is_null_or_empty(op.splice.as_deref()) {
verbs.push("splice");
}
if !is_null_or_empty(op.pre.as_deref()) {
verbs.push("pre");
}
if !is_null_or_empty(op.post.as_deref()) {
verbs.push("post");
}
if !is_null_or_empty(op.sed.as_deref()) {
verbs.push("sed");
}
if verbs.is_empty() {
verbs.push("none");
}
e.verbs.push(verbs.join("+"));
}
}
"ast_edit" => {
let a: EditArgs = raw
.and_then(|r| serde_json::from_str(r.get()).ok())
.unwrap_or_default();
e.num_edits = a.ops.len() as i64;
}
"write" => {
e.num_edits = 1;
e.verbs.push("write".to_string());
}
_ => {}
}
e
}
/// Looks at top-level argument keys (and the first sub-edit for the `edit`
/// tool) to identify which schema is in use. Older sessions used many
/// incompatible schemas.
fn detect_edit_format(name: &str, raw: Option<&RawValue>) -> String {
match name.to_ascii_lowercase().as_str() {
"write" => return "write".to_string(),
"ast_edit" => return "ast_edit".to_string(),
_ => {}
}
let Some(raw) = raw else {
return "unknown".to_string();
};
let top: HashMap<String, Value> = match serde_json::from_str(raw.get()) {
Ok(v) => v,
Err(_) => return "unknown".to_string(),
};
let has = |k: &str| top.contains_key(k);
if has("oldText") && has("newText") {
return "oldText/newText".to_string();
}
if has("old_text") && has("new_text") {
return "old_text/new_text".to_string();
}
if has("diff") && has("op") {
return "diff+op".to_string();
}
if has("diff") && has("operation") {
return "diff+operation".to_string();
}
if has("diff") {
return "diff".to_string();
}
if has("replace") || has("insert") {
return "replace/insert".to_string();
}
if let Some(edits_val) = top.get("edits")
&& let Some(arr) = edits_val.as_array()
&& let Some(first) = arr.first().and_then(Value::as_object)
{
let fh = |k: &str| first.contains_key(k);
if fh("loc") && (fh("splice") || fh("pre") || fh("post") || fh("sed")) {
return "loc+splice/pre/post/sed".to_string();
}
if fh("loc") && fh("content") {
return "loc+content".to_string();
}
if fh("set_line") {
return "set_line".to_string();
}
if fh("insert_after") {
return "insert_after".to_string();
}
if fh("op") && fh("pos") && fh("end") && fh("lines") {
return "op+pos+end+lines".to_string();
}
if fh("op") && fh("pos") && fh("lines") {
return "op+pos+lines".to_string();
}
if fh("op") && fh("sel") && fh("content") {
return "op+sel+content".to_string();
}
if fh("all") && (fh("new_text") || fh("old_text")) {
return "per-edit:old_text/new_text".to_string();
}
let mut keys: Vec<&str> = first.keys().map(String::as_str).collect();
keys.sort_unstable();
return format!("edits[{}]", keys.join(","));
}
let mut keys: Vec<&str> = top.keys().map(String::as_str).collect();
keys.sort_unstable();
keys.join(",")
}
fn is_null_or_empty(b: Option<&RawValue>) -> bool {
let Some(b) = b else { return true };
let s = b.get().trim();
s.is_empty() || s == "null"
}
fn loc_shape(loc: &str) -> String {
if loc.is_empty() {
return "empty".to_string();
}
if loc == "$" {
return "$file".to_string();
}
let rest = if let Some(i) = loc.rfind(':')
&& !loc.starts_with('$')
{
&loc[i + 1..]
} else {
loc
};
if rest.starts_with('(') && rest.ends_with(')') {
return "bracket-(body)".to_string();
}
if rest.starts_with('[') && rest.ends_with(']') {
return "bracket-[block]".to_string();
}
if rest.starts_with('(') || rest.starts_with('[') {
return "bracket-tail".to_string();
}
if rest.ends_with(')') || rest.ends_with(']') {
return "bracket-head".to_string();
}
if ANCHOR_BARE.is_match(rest) {
return "bare-anchor".to_string();
}
"other".to_string()
}
// ---- result classification ----
macro_rules! re {
($pat:expr) => {
LazyLock::new(|| Regex::new($pat).expect("compile result regex"))
};
}
static RE_ANCHOR_STALE: LazyLock<Regex> = re!(
r"(?i)(Edit rejected:.*line[s]? .* changed since the last read|line[s]? ha(s|ve) changed since last read)"
);
static RE_ANCHOR_MISSING: LazyLock<Regex> =
re!(r"(?i)anchor .* (not found|unknown|missing)|loc requires the full anchor");
static RE_NO_ENCLOSING: LazyLock<Regex> = re!(r"(?i)No enclosing .* block");
static RE_PARSE_ERROR: LazyLock<Regex> =
re!(r"(?i)parse|syntax error|unbalanced|unexpected token");
static RE_SSR_NO_MATCH: LazyLock<Regex> = re!(
r"(?i)0 matches|no replacements|no match found|No replacements made|Failed to find expected lines"
);
static RE_FILE_NOT_READ: LazyLock<Regex> =
re!(r"(?i)must be read first|has not been read|not yet read");
static RE_FILE_CHANGED: LazyLock<Regex> =
re!(r"(?i)file has been (modified|changed) externally");
static RE_PERM_DENIED: LazyLock<Regex> = re!(r"(?i)permission denied|not allowed");
static RE_GENERIC_REJECTED: LazyLock<Regex> =
re!(r"(?i)\b(rejected|failed|error|invalid)\b");
static RE_TRUNCATED: LazyLock<Regex> = re!(r"(?i)\[Output truncated");
static RE_ABORTED: LazyLock<Regex> = re!(
r"(?i)Tool execution was aborted|Request was aborted|cancelled|canceled by user"
);
static RE_SUCCESS: LazyLock<Regex> = re!(
r"(?i)^(Updated|Successfully (wrote|replaced|edited|deleted|inserted)|Replaced|Applied|Deleted|Created|Wrote|edit applied|Edited|Inserted|OK\b)"
);
fn classify_edit_result(text: &str) -> String {
let t = text.trim();
if t.is_empty() {
return "empty".to_string();
}
let first = t.split_once('\n').map_or(t, |(a, _)| a);
if RE_TRUNCATED.is_match(first) {
return "truncated".to_string();
}
if RE_ABORTED.is_match(t) {
return "aborted".to_string();
}
if RE_SUCCESS.is_match(first) {
return "success".to_string();
}
if RE_ANCHOR_STALE.is_match(t) {
return "fail:anchor-stale".to_string();
}
if RE_NO_ENCLOSING.is_match(t) {
return "fail:no-enclosing-block".to_string();
}
if RE_ANCHOR_MISSING.is_match(t) {
return "fail:anchor-missing".to_string();
}
if RE_PARSE_ERROR.is_match(t) {
return "fail:parse".to_string();
}
if RE_SSR_NO_MATCH.is_match(t) {
return "fail:no-match".to_string();
}
if RE_FILE_NOT_READ.is_match(t) {
return "fail:file-not-read".to_string();
}
if RE_FILE_CHANGED.is_match(t) {
return "fail:file-changed".to_string();
}
if RE_PERM_DENIED.is_match(t) {
return "fail:perm".to_string();
}
if RE_GENERIC_REJECTED.is_match(first) {
return "fail:other".to_string();
}
"unknown".to_string()
}
// ---- reporting ----
fn report_edits(entries: &[EditEntry]) {
if entries.is_empty() {
println!("no edit-family tool calls found in matched sessions");
return;
}
let mut by_tool: HashMap<String, i64> = HashMap::new();
let mut by_format: HashMap<String, i64> = HashMap::new();
let mut status_by_format: HashMap<String, HashMap<String, i64>> = HashMap::new();
let mut status_by_tool: HashMap<String, HashMap<String, i64>> = HashMap::new();
let mut verb_count: HashMap<String, i64> = HashMap::new();
let mut loc_count: HashMap<String, i64> = HashMap::new();
let mut fails_by_verb: HashMap<String, HashMap<String, i64>> = HashMap::new();
let mut fails_by_loc: HashMap<String, HashMap<String, i64>> = HashMap::new();
for e in entries {
*by_tool.entry(e.tool_name.clone()).or_insert(0) += 1;
*status_by_tool
.entry(e.tool_name.clone())
.or_default()
.entry(e.status.clone())
.or_insert(0) += 1;
*by_format.entry(e.format.clone()).or_insert(0) += 1;
*status_by_format
.entry(e.format.clone())
.or_default()
.entry(e.status.clone())
.or_insert(0) += 1;
for v in &e.verbs {
*verb_count.entry(v.clone()).or_insert(0) += 1;
*fails_by_verb
.entry(v.clone())
.or_default()
.entry(e.status.clone())
.or_insert(0) += 1;
}
for l in &e.loc_shapes {
*loc_count.entry(l.clone()).or_insert(0) += 1;
*fails_by_loc
.entry(l.clone())
.or_default()
.entry(e.status.clone())
.or_insert(0) += 1;
}
}
println!("# Edit-tool usage");
println!(
"\nTotal tool calls: {} (across {} sessions)",
entries.len(),
count_edit_sessions(entries)
);
println!("\n## By tool");
print_sorted(&by_tool);
println!("\n## Outcome by tool");
let mut tools: Vec<&String> = by_tool.keys().collect();
tools.sort();
for t in tools {
println!("\n {t} ({} calls):", by_tool[t.as_str()]);
if let Some(m) = status_by_tool.get(t.as_str()) {
print_sorted_indent(m, " ");
}
}
println!("\n## edit verb distribution (per sub-edit)");
print_sorted(&verb_count);
println!("\n## edit locator shape distribution");
print_sorted(&loc_count);
println!("\n## Failure rate per verb shape");
for v in sorted_by_count(&verb_count) {
let (total, failed) = fail_totals(fails_by_verb.get(v.as_str()));
println!(
" {v:<20} {failed}/{total} failed ({:.0}%)",
pct(failed, total)
);
}
println!("\n## Failure rate per locator shape");
for l in sorted_by_count(&loc_count) {
let (total, failed) = fail_totals(fails_by_loc.get(l.as_str()));
println!(
" {l:<20} {failed}/{total} failed ({:.0}%)",
pct(failed, total)
);
}
println!("\n## edit-tool argument-format usage");
print_sorted(&by_format);
println!("\n## Failure rate per argument format");
for fname in sorted_by_count(&by_format) {
let (total, failed) = fail_totals(status_by_format.get(fname.as_str()));
println!(
" {fname:<32} {failed:>6}/{total:<6} failed ({:.0}%)",
pct(failed, total)
);
}
println!("\n## Failure breakdown per top format");
for fname in sorted_by_count(&by_format).into_iter().take(8) {
println!("\n {fname} ({} total)", by_format[fname.as_str()]);
if let Some(m) = status_by_format.get(fname.as_str()) {
print_sorted_indent(m, " ");
}
}
println!("\n## Sample failed edits");
let mut shown = 0;
for e in entries {
if !e.status.starts_with("fail") {
continue;
}
let first = e
.result_raw
.split_once("\n\n")
.map_or(e.result_raw.as_str(), |(a, _)| a);
println!(
"\n— {} [{}] verbs={:?} loc={:?}\n result: {}",
e.tool_name,
e.status,
e.verbs,
e.loc_shapes,
truncate_line(first, 220)
);
shown += 1;
if shown >= 8 {
break;
}
}
}
fn fail_totals(m: Option<&HashMap<String, i64>>) -> (i64, i64) {
let Some(m) = m else { return (0, 0) };
let mut total = 0i64;
let mut failed = 0i64;
for (status, n) in m {
total += n;
if status.starts_with("fail") {
failed += n;
}
}
(total, failed)
}
fn count_edit_sessions(entries: &[EditEntry]) -> usize {
let mut s: std::collections::HashSet<&str> = std::collections::HashSet::new();
for e in entries {
s.insert(&e.file);
}
s.len()
}
fn write_csv(entries: &[EditEntry]) -> Result<()> {
let path = std::env::var("EDIT_ANALYSIS_CSV").unwrap_or_else(|_| "edit-analysis.csv".to_string());
let f = File::create(&path).with_context(|| format!("create {path}"))?;
let mut w = csv::Writer::from_writer(f);
w.write_record([
"session",
"tool",
"status",
"num_edits",
"verbs",
"loc_shapes",
"result_first_line",
])?;
for e in entries {
let first = e
.result_raw
.split_once('\n')
.map_or(e.result_raw.as_str(), |(a, _)| a);
let session = Path::new(&e.file)
.file_name()
.and_then(|s| s.to_str())
.unwrap_or(&e.file);
w.write_record([
session,
&e.tool_name,
&e.status,
&e.num_edits.to_string(),
&e.verbs.join(","),
&e.loc_shapes.join(","),
&truncate_line(first, 200),
])?;
}
w.flush()?;
Ok(())
}
-972
View File
@@ -1,972 +0,0 @@
//! `followups` subcommand — heuristic for buggy hashline edits.
//!
//! Premise: when a hashline `edit` introduces a duplicate `}`, drops a line,
//! or otherwise breaks adjacent context, the next thing the agent does is
//! usually a tiny corrective edit on the same file. So we flag, per session,
//! pairs of edits where:
//!
//! - both are hashline `edit` toolCalls (input begins with a `@PATH`),
//! - both target the same file,
//! - the FIRST edit succeeded,
//! - the SECOND edit is small (<= --max-fix lines changed, default 2),
//! - the second edit is the next edit (in the same session) on that file,
//! and there is no other edit in between (so retries-after-failure are
//! not counted).
//!
//! For each hit we classify the small follow-up's pattern: adds a single
//! closing brace/paren/bracket, removes one (likely duplicate), pure single
//! insert, pure single delete, or one-line tweak.
use crate::common::*;
use anyhow::{Context, Result, bail};
use regex::Regex;
use serde::Deserialize;
use std::collections::HashMap;
use std::fs::File;
use std::io::{BufRead, BufReader};
use std::path::Path;
use std::sync::LazyLock;
// ---- parsed shapes --------------------------------------------------------
#[derive(Clone, Default)]
struct EditSection {
file: String,
payload_lines: Vec<String>,
/// Payload lines grouped by op. Each inner vec is one op's payload (`+`,
/// `<`, or `=` followed by `~TEXT` lines), in source order. Used to detect
/// intra-block self-similarity (model duplicating an N-line chunk).
payload_blocks: Vec<Vec<String>>,
/// Raw anchor refs (`123ab`) from ops in this section, in source order.
/// Used to detect the same anchor being referenced by multiple ops in the
/// same call.
op_anchors: Vec<String>,
/// Total line count covered by `- A..B` and `= A..B` ranges (range size).
deleted_lines: i64,
op_count: i64,
/// Min/max anchor line touched by ops in this section. `None` for a
/// section whose only ops target BOF/EOF (no concrete line).
min_line: Option<i64>,
max_line: Option<i64>,
}
impl EditSection {
fn change_size(&self) -> i64 {
self.payload_lines.len() as i64 + self.deleted_lines
}
fn touch(&mut self, line: i64) {
self.min_line = Some(self.min_line.map_or(line, |m| m.min(line)));
self.max_line = Some(self.max_line.map_or(line, |m| m.max(line)));
}
}
#[derive(Clone)]
struct EditCall {
ts: i64,
call_id: String,
sections: Vec<EditSection>,
success: bool,
raw_input_len: usize,
/// Tool-emitted warnings extracted from a successful result text. Each
/// entry is one of `auto-rebased` / `auto-absorbed` / `auto-dropped`.
warnings: Vec<&'static str>,
}
#[derive(Deserialize, Default)]
struct EditArgs {
#[serde(default)]
input: Option<String>,
}
// ---- per-section parsing --------------------------------------------------
// Op headers. We parse loosely: anything that doesn't match a known op or a
// `~` payload line is ignored (blank line, comment, etc.).
//
// Range sizes: `LINEhash..LINEhash` -> end_line - start_line + 1 (clamped to
// >= 1). For single-anchor `- A` we count 1.
static RANGE_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$").expect("RANGE_RE"));
static SINGLE_ANCHOR_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"^\s*(\d+)[a-z*]+\s*$").expect("SINGLE_ANCHOR_RE"));
/// Returns (range_size, optional (start_line, end_line)). Size is at least 1.
fn parse_range(raw: &str) -> (i64, Option<(i64, i64)>) {
let Some(caps) = RANGE_RE.captures(raw.trim()) else {
return (1, None);
};
let start: i64 = caps.get(1).and_then(|m| m.as_str().parse().ok()).unwrap_or(0);
let end: i64 = caps
.get(2)
.and_then(|m| m.as_str().parse().ok())
.unwrap_or(start);
let size = (end - start + 1).max(1);
let lines = if start > 0 { Some((start, end.max(start))) } else { None };
(size, lines)
}
fn parse_anchor_line(raw: &str) -> Option<i64> {
let caps = SINGLE_ANCHOR_RE.captures(raw.trim())?;
caps.get(1)?.as_str().parse().ok()
}
/// Parses a single hashline `input` arg into per-file sections. Returns an
/// empty vec if the input doesn't begin with `@PATH` (vim-mode edits etc.).
fn parse_hashline_input(input: &str) -> Vec<EditSection> {
let mut sections: Vec<EditSection> = Vec::new();
let mut cur: Option<EditSection> = None;
// Index of the currently-open payload block in cur.payload_blocks, or
// `None` if no op is awaiting payload.
let mut open_block: Option<usize> = None;
fn open_new_block(s: &mut EditSection, open_block: &mut Option<usize>) {
s.payload_blocks.push(Vec::new());
*open_block = Some(s.payload_blocks.len() - 1);
}
fn close_block(open_block: &mut Option<usize>) {
*open_block = None;
}
for raw_line in input.split('\n') {
let line = raw_line.strip_suffix('\r').unwrap_or(raw_line);
// File header: `@<path>`. Always starts a new section.
if let Some(rest) = line.strip_prefix('@') {
if let Some(s) = cur.take() {
sections.push(s);
}
cur = Some(EditSection { file: rest.trim().to_string(), ..Default::default() });
open_block = None;
continue;
}
let Some(s) = cur.as_mut() else {
// Stray content before any `@PATH` — not a hashline input.
continue;
};
// Payload lines (`~...`) belong to whatever op was last opened.
if let Some(payload) = line.strip_prefix('~') {
s.payload_lines.push(payload.to_string());
if open_block.is_none() {
open_new_block(s, &mut open_block);
}
if let Some(idx) = open_block {
s.payload_blocks[idx].push(payload.to_string());
}
continue;
}
let trimmed = line.trim_start();
let op_byte = trimmed.as_bytes().first().copied();
match op_byte {
Some(b'+') | Some(b'<') => {
let body = trimmed[1..].trim_start();
let (anchor_part, tail) = match body.split_once('~') {
Some((a, t)) => (a, Some(t)),
None => (body, None),
};
let anchor_trimmed = anchor_part.trim();
if !anchor_trimmed.is_empty() && anchor_trimmed != "BOF" && anchor_trimmed != "EOF" {
s.op_anchors.push(anchor_trimmed.to_string());
}
if let Some(line) = parse_anchor_line(anchor_part) {
s.touch(line);
}
if let Some(tail) = tail {
// Inline `+ ANCHOR~text`: replaces a single line; not a
// payload-collecting op.
s.payload_lines.push(tail.to_string());
s.deleted_lines += 1;
close_block(&mut open_block);
} else {
open_new_block(s, &mut open_block);
}
s.op_count += 1;
}
Some(b'-') | Some(b'=') => {
let body = trimmed[1..].trim_start();
let (size, lines) = parse_range(body);
s.deleted_lines += size;
if let Some((lo, hi)) = lines {
s.touch(lo);
s.touch(hi);
}
// Collect raw anchor refs (`A` or `A..B`) for dup detection.
for part in body.trim().split("..") {
let t = part.trim();
if !t.is_empty() {
s.op_anchors.push(t.to_string());
}
}
s.op_count += 1;
if op_byte == Some(b'=') {
open_new_block(s, &mut open_block);
} else {
close_block(&mut open_block);
}
}
_ => {
// blank / unrecognized — leave payload state alone.
}
}
}
if let Some(s) = cur.take() {
sections.push(s);
}
sections
}
// ---- success classification ----------------------------------------------
static RE_FAILURE_HEAD: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(
r"(?i)^(edit rejected|error\b|failed\b|invalid\b|unrecognized\b|cannot\b|no enclosing|file has been (modified|changed)|file has not been read|permission denied|tool execution was aborted|request was aborted|cancelled|canceled|line \d+:|expected|unexpected|patch failed|no replacements|0 matches)",
)
.expect("RE_FAILURE_HEAD")
});
fn looks_successful(text: &str) -> bool {
let head = text.lines().find(|l| !l.trim().is_empty()).unwrap_or("");
if head.is_empty() {
return false;
}
!RE_FAILURE_HEAD.is_match(head.trim_start())
}
fn extract_warnings(text: &str) -> Vec<&'static str> {
let mut out: Vec<&'static str> = Vec::new();
for line in text.lines() {
let t = line.trim_start();
if t.starts_with("Auto-rebased anchor") {
out.push("auto-rebased");
} else if t.starts_with("Auto-absorbed") {
out.push("auto-absorbed");
} else if t.starts_with("Auto-dropped") {
out.push("auto-dropped");
}
}
out
}
// ---- followup pattern classification -------------------------------------
#[derive(Clone, Copy, PartialEq, Eq, Hash)]
enum FixPattern {
AddSingleCloser, // payload contains exactly one line that's pure }/]/)
RemoveSingleCloser, // single delete of a pure }/]/) line
PureInsertOneLine,
PureDeleteOneLine,
OneLineModify,
SmallOther,
}
impl FixPattern {
fn label(&self) -> &'static str {
match self {
FixPattern::AddSingleCloser => "add-single-closer",
FixPattern::RemoveSingleCloser => "remove-single-closer",
FixPattern::PureInsertOneLine => "pure-insert-1",
FixPattern::PureDeleteOneLine => "pure-delete-1",
FixPattern::OneLineModify => "one-line-modify",
FixPattern::SmallOther => "small-other",
}
}
}
fn pattern_priority(p: FixPattern) -> u8 {
match p {
FixPattern::RemoveSingleCloser => 0,
FixPattern::AddSingleCloser => 1,
FixPattern::OneLineModify => 2,
FixPattern::PureDeleteOneLine => 3,
FixPattern::PureInsertOneLine => 4,
FixPattern::SmallOther => 5,
}
}
static CLOSER_LINE_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"^\s*[\])}]+[;,]?\s*$").expect("CLOSER_LINE_RE"));
fn classify_fix(section: &EditSection) -> FixPattern {
let payload = &section.payload_lines;
let deleted = section.deleted_lines;
let payload_is_one_closer = payload.len() == 1 && CLOSER_LINE_RE.is_match(&payload[0]);
if deleted == 0 && payload_is_one_closer {
return FixPattern::AddSingleCloser;
}
if deleted == 1 && payload.is_empty() {
// We don't have the deleted line text, but a single-line delete on a
// sub-1KB follow-up is overwhelmingly the "drop duplicate brace" case
// when paired with the immediately-prior big edit. Mark it so;
// false-positives are easy to triage by hand.
return FixPattern::RemoveSingleCloser;
}
if deleted == 0 && payload.len() == 1 {
return FixPattern::PureInsertOneLine;
}
if deleted == 1 && payload.len() == 1 {
return FixPattern::OneLineModify;
}
if deleted >= 1 && payload.is_empty() {
return FixPattern::PureDeleteOneLine;
}
FixPattern::SmallOther
}
// ---- per-file scanning ---------------------------------------------------
#[derive(Clone)]
struct Hit {
session: String,
file: String,
first_call_id: String,
second_call_id: String,
first_size: i64,
first_input_len: usize,
second_size: i64,
pattern: FixPattern,
second_summary: String,
gap_secs: i64,
}
/// A successful edit whose result text contained a tool self-correction.
#[derive(Clone)]
struct WarningHit {
session: String,
call_id: String,
kind: &'static str, // auto-rebased / auto-absorbed / auto-dropped
files: String, // comma-joined section files
input_len: usize,
}
/// Two consecutive successful edits on the same file whose anchor line
/// ranges overlap (or one is contained in the other). Catches "agent
/// re-edited the same locus" cases that the small-fix detector misses.
#[derive(Clone)]
struct LocusHit {
session: String,
file: String,
first_call_id: String,
second_call_id: String,
first_range: (i64, i64),
second_range: (i64, i64),
first_size: i64,
second_size: i64,
gap_secs: i64,
}
/// A single hashline edit whose payload contains a contiguous N-line sequence
/// that repeats inside the same payload block. Strong signal of "model pasted
/// the same chunk twice" when N is large.
#[derive(Clone)]
struct PayloadDupHit {
session: String,
call_id: String,
file: String,
block_len: usize,
repeat_len: usize,
sample: String,
}
/// A single hashline edit that referenced the same anchor (e.g. `123ab`) from
/// two or more distinct ops. Often benign (insert before + insert after at
/// the same line), occasionally indicates a duplicated op in the input.
#[derive(Clone)]
struct AnchorDupHit {
session: String,
call_id: String,
files: String,
anchor: String,
count: usize,
}
#[derive(Default)]
struct FileReport {
fix_hits: Vec<Hit>,
warning_hits: Vec<WarningHit>,
locus_hits: Vec<LocusHit>,
payload_dups: Vec<PayloadDupHit>,
anchor_dups: Vec<AnchorDupHit>,
/// Total number of successful hashline edits scanned.
total_successful_edits: i64,
}
/// Find the longest contiguous N-line sequence in `block` that occurs at two
/// distinct positions. Returns `(first_index, repeat_len)` if a match of at
/// least `min_len` lines exists with at least half its lines being
/// non-trivial (>= 4 chars after trim) — this filters out e.g. five repeated
/// `}` lines as boilerplate.
fn find_longest_repeat(block: &[String], min_len: usize) -> Option<(usize, usize)> {
let n = block.len();
if n < 2 * min_len {
return None;
}
let mut best: Option<(usize, usize)> = None;
for i in 0..n.saturating_sub(min_len) {
for j in (i + min_len)..=n.saturating_sub(min_len) {
let mut k = 0;
while i + k < j && j + k < n && block[i + k] == block[j + k] {
k += 1;
}
if k < min_len {
continue;
}
let meaningful = block[i..i + k]
.iter()
.filter(|s| s.trim().len() >= 4)
.count();
if meaningful < (k.div_ceil(2)).max(2) {
continue;
}
if best.is_none_or(|(_, bk)| k > bk) {
best = Some((i, k));
}
}
}
best
}
/// Returns the set of `(anchor, count, file)` pairs where the same raw
/// anchor was referenced by `count >= 2` distinct ops within ONE section
/// (i.e. on one file). Two files happening to have the same `LINE+hash`
/// anchor is a coincidence (2-char hashes collide), not a duplicate op.
/// Skips `BOF` / `EOF` and any anchor with a `*` interior-hash.
fn duplicated_anchors(sections: &[EditSection]) -> Vec<(String, usize, String)> {
let mut out: Vec<(String, usize, String)> = Vec::new();
for sec in sections {
let mut counts: HashMap<&str, usize> = HashMap::new();
for a in &sec.op_anchors {
if a == "BOF" || a == "EOF" || a.contains('*') {
continue;
}
*counts.entry(a.as_str()).or_default() += 1;
}
for (anchor, c) in counts {
if c >= 2 {
out.push((anchor.to_string(), c, sec.file.clone()));
}
}
}
out.sort_by(|a, b| b.1.cmp(&a.1));
out
}
fn process_file(path: &Path, max_fix: i64) -> FileReport {
let f = match File::open(path) {
Ok(f) => f,
Err(e) => {
eprintln!("open {}: {e}", path.display());
return FileReport::default();
}
};
let reader = BufReader::with_capacity(64 * 1024, f);
let session = path
.file_stem()
.and_then(|s| s.to_str())
.unwrap_or("")
.to_string();
// Walk message events in order. For each toolCall name=="edit", parse the
// input. Pair with its toolResult by callId. Keep the chronological list
// of EditCall records.
let mut pending: HashMap<String, EditCall> = HashMap::new();
let mut order: Vec<String> = Vec::new();
let mut calls: HashMap<String, EditCall> = HashMap::new();
for line in reader.lines() {
let Ok(line) = line else { continue };
if line.is_empty() {
continue;
}
let Ok(ev) = serde_json::from_str::<RawEvent>(&line) else {
continue;
};
if ev.kind != "message" {
continue;
}
let Some(msg_raw) = ev.message else { continue };
let Ok(m) = serde_json::from_str::<Message>(msg_raw.get()) else {
continue;
};
let Some(content_raw) = m.content else { continue };
let items = parse_content(&content_raw);
match m.role.as_str() {
"assistant" => {
for it in items {
if it.kind != "toolCall" || it.name != "edit" {
continue;
}
let raw = it.arguments.as_deref();
let Some(raw) = raw else { continue };
let Ok(args) = serde_json::from_str::<EditArgs>(raw.get()) else {
continue;
};
let Some(input) = args.input else { continue };
if !input.trim_start().starts_with('@') {
// Vim-mode edit or other shape — skip.
continue;
}
let sections = parse_hashline_input(&input);
if sections.is_empty() {
continue;
}
let call = EditCall {
ts: parse_ts(&ev.timestamp),
call_id: it.id.clone(),
sections,
success: false, // filled in by toolResult
raw_input_len: input.len(),
warnings: Vec::new(),
};
pending.insert(it.id.clone(), call);
}
}
"toolResult" => {
if m.tool_name != "edit" {
continue;
}
let Some(mut call) = pending.remove(&m.tool_call_id) else {
continue;
};
let text = join_text(&items);
call.success = looks_successful(&text);
if call.success {
call.warnings = extract_warnings(&text);
}
let id = call.call_id.clone();
calls.insert(id.clone(), call);
order.push(id);
}
_ => {}
}
}
// Build per-file edit sequences. A single call can touch multiple files
// (multi-section input); we record an entry per (call, section).
struct Entry<'a> {
section: &'a EditSection,
call: &'a EditCall,
}
let mut by_file: HashMap<&str, Vec<Entry>> = HashMap::new();
for id in &order {
let Some(call) = calls.get(id) else { continue };
for sec in &call.sections {
by_file
.entry(sec.file.as_str())
.or_default()
.push(Entry { section: sec, call });
}
}
let mut report = FileReport::default();
for call in calls.values() {
if call.success {
report.total_successful_edits += 1;
}
if !call.warnings.is_empty() {
// Dedup per kind so a call with two `Auto-rebased` lines counts once.
let mut seen: Vec<&'static str> = Vec::new();
for w in &call.warnings {
if !seen.contains(w) {
seen.push(*w);
}
}
let files: Vec<String> = call.sections.iter().map(|s| s.file.clone()).collect();
for kind in seen {
report.warning_hits.push(WarningHit {
session: session.clone(),
call_id: call.call_id.clone(),
kind,
files: files.join(","),
input_len: call.raw_input_len,
});
}
}
// Only flag detectors on edits that actually applied. Failed edits
// are handled separately and their input is moot.
if call.success {
// Payload self-dup: per section, per block. Threshold is
// intentionally permissive (>=4) — caller filters by repeat_len.
for sec in &call.sections {
for block in &sec.payload_blocks {
if let Some((start, len)) = find_longest_repeat(block, 4) {
let sample = truncate_line(block.get(start).map(String::as_str).unwrap_or(""), 80);
report.payload_dups.push(PayloadDupHit {
session: session.clone(),
call_id: call.call_id.clone(),
file: sec.file.clone(),
block_len: block.len(),
repeat_len: len,
sample,
});
}
}
}
// Anchor dup within a single section's ops.
for (anchor, count, file) in duplicated_anchors(&call.sections) {
report.anchor_dups.push(AnchorDupHit {
session: session.clone(),
call_id: call.call_id.clone(),
files: file,
anchor,
count,
});
}
}
}
for (file, entries) in by_file {
for window in entries.windows(2) {
let a = &window[0];
let b = &window[1];
if !a.call.success || !b.call.success {
continue;
}
if a.call.call_id == b.call.call_id {
continue;
}
let first_size = a.section.change_size();
let second_size = b.section.change_size();
let gap = (b.call.ts - a.call.ts).max(0);
// (1) Small-fix follow-up.
if second_size > 0 && second_size <= max_fix && first_size > 2 {
let pattern = classify_fix(b.section);
let summary = render_section_summary(b.section);
report.fix_hits.push(Hit {
session: session.clone(),
file: file.to_string(),
first_call_id: a.call.call_id.clone(),
second_call_id: b.call.call_id.clone(),
first_size,
first_input_len: a.call.raw_input_len,
second_size,
pattern,
second_summary: summary,
gap_secs: gap,
});
}
// (2) Same-locus re-edit. Skip when both are tiny (those are
// already noisy) and skip when the small-fix branch already
// fired (we don't want to double-count).
if first_size > 2 && second_size > max_fix {
if let (Some(a_lo), Some(a_hi), Some(b_lo), Some(b_hi)) = (
a.section.min_line,
a.section.max_line,
b.section.min_line,
b.section.max_line,
) {
let overlap_lo = a_lo.max(b_lo);
let overlap_hi = a_hi.min(b_hi);
if overlap_lo <= overlap_hi {
report.locus_hits.push(LocusHit {
session: session.clone(),
file: file.to_string(),
first_call_id: a.call.call_id.clone(),
second_call_id: b.call.call_id.clone(),
first_range: (a_lo, a_hi),
second_range: (b_lo, b_hi),
first_size,
second_size,
gap_secs: gap,
});
}
}
}
}
}
report
}
fn render_section_summary(section: &EditSection) -> String {
let mut bits: Vec<String> = Vec::new();
if section.deleted_lines > 0 {
bits.push(format!("-{}", section.deleted_lines));
}
if !section.payload_lines.is_empty() {
bits.push(format!("+{}", section.payload_lines.len()));
}
let mut out = bits.join(" / ");
if !section.payload_lines.is_empty() {
let preview = truncate_line(&section.payload_lines[0], 80);
out.push_str(&format!(" | {preview}"));
}
out
}
// ---- entry point ---------------------------------------------------------
pub fn run(args: Vec<String>) -> Result<()> {
let mut limit: usize = 200;
let mut workers: usize = 0;
let mut max_fix: i64 = 2;
let mut show: usize = 60;
let mut max_gap: i64 = 0;
let mut pattern_filter: Option<String> = None;
let mut min_dup: usize = 8;
let mut iter = args.into_iter();
while let Some(a) = iter.next() {
match a.as_str() {
"-n" => {
limit = iter
.next()
.context("-n requires a value")?
.parse()
.context("-n value")?;
}
"-j" => {
workers = iter
.next()
.context("-j requires a value")?
.parse()
.context("-j value")?;
}
"--max-fix" => {
max_fix = iter
.next()
.context("--max-fix requires a value")?
.parse()
.context("--max-fix value")?;
}
"--show" => {
show = iter
.next()
.context("--show requires a value")?
.parse()
.context("--show value")?;
}
"--max-gap" => {
max_gap = iter
.next()
.context("--max-gap requires seconds")?
.parse()
.context("--max-gap value")?;
}
"--pattern" => {
pattern_filter = Some(iter.next().context("--pattern requires a name")?);
}
"--min-dup" => {
min_dup = iter
.next()
.context("--min-dup requires a value")?
.parse()
.context("--min-dup value")?;
}
"-h" | "--help" => {
eprintln!(
"usage: session-stats followups [-n N] [-j workers] [--max-fix N] [--max-gap S]
[--min-dup K] [--pattern NAME] [--show N]
Five detectors over hashline `edit` calls in the most-recent N sessions:
1. small-fix follow-ups
consecutive successful edits on the same file where the follow-up
changes <= --max-fix lines (default 2). The first edit must be > 2
lines. Brace-related patterns surface first (remove-single-closer,
add-single-closer). --max-gap caps elapsed seconds between the pair.
2. tool self-corrections
warning lines emitted by the tool on otherwise-successful edits:
auto-rebased, auto-absorbed, auto-dropped. These are direct evidence
of the model writing stale anchors or duplicating adjacent context.
3. same-locus re-edits
two consecutive edits on the same file whose anchor line ranges
overlap, where both edits are > --max-fix lines (so they're not
already in (1)).
4. payload self-duplication
within one payload block, a contiguous K-line sequence appears at two
positions. K threshold via --min-dup (default 8).
5. same-anchor reuse
two or more ops in the same section reference the same `LINE+hash`
anchor.
--pattern NAME filters (1) to a single FixPattern label."
);
return Ok(());
}
other => bail!("unknown flag: {other}"),
}
}
let files = collect_sessions(&WalkOpts {
date_filters: Vec::new(),
limit_most_recent: limit,
})?;
eprintln!("scanning {} session files", files.len());
let max_fix_local = max_fix;
let reports: Vec<FileReport> = parallel_collect(&files, workers, 5_000, |p| {
let r = process_file(p, max_fix_local);
let empty = r.fix_hits.is_empty() && r.warning_hits.is_empty() && r.locus_hits.is_empty()
&& r.total_successful_edits == 0;
if empty { None } else { Some(r) }
});
let mut hits: Vec<Hit> = Vec::new();
let mut warning_hits: Vec<WarningHit> = Vec::new();
let mut locus_hits: Vec<LocusHit> = Vec::new();
let mut payload_dups: Vec<PayloadDupHit> = Vec::new();
let mut anchor_dups: Vec<AnchorDupHit> = Vec::new();
let mut total_successful_edits: i64 = 0;
for r in reports {
hits.extend(r.fix_hits);
warning_hits.extend(r.warning_hits);
locus_hits.extend(r.locus_hits);
payload_dups.extend(r.payload_dups);
anchor_dups.extend(r.anchor_dups);
total_successful_edits += r.total_successful_edits;
}
if max_gap > 0 {
hits.retain(|h| h.gap_secs <= max_gap);
locus_hits.retain(|h| h.gap_secs <= max_gap);
}
if let Some(ref name) = pattern_filter {
hits.retain(|h| h.pattern.label() == name);
}
hits.sort_by(|a, b| {
pattern_priority(a.pattern)
.cmp(&pattern_priority(b.pattern))
.then_with(|| b.first_input_len.cmp(&a.first_input_len))
});
locus_hits.sort_by(|a, b| b.first_size.cmp(&a.first_size));
let mut by_pattern: HashMap<&'static str, i64> = HashMap::new();
for h in &hits {
*by_pattern.entry(h.pattern.label()).or_default() += 1;
}
println!("=== heuristic followup hits ===");
println!("total hits: {}", commas(hits.len() as i64));
println!();
println!("by pattern:");
let mut pats: Vec<(&&str, &i64)> = by_pattern.iter().collect();
pats.sort_by(|a, b| b.1.cmp(a.1));
for (label, n) in pats {
println!(" {:<22} {:>6}", label, commas(*n));
}
println!();
let shown = show.min(hits.len());
println!("=== top {shown} hits (by first-edit input size) ===");
for h in hits.iter().take(shown) {
println!(
"[{pat}] {file} first={first_size}L ({first_len}B) → second={second_size}L gap={gap}s",
pat = h.pattern.label(),
file = h.file,
first_size = h.first_size,
first_len = h.first_input_len,
second_size = h.second_size,
gap = h.gap_secs,
);
println!(" session={}", h.session);
println!(" first_call={} second_call={}", h.first_call_id, h.second_call_id);
println!(" fix: {}", h.second_summary);
}
println!();
println!("=== tool self-corrections ===");
println!("(emitted as warnings on otherwise-successful edits — the tool caught what the model wrote)");
let mut warn_by_kind: HashMap<&'static str, i64> = HashMap::new();
for w in &warning_hits {
*warn_by_kind.entry(w.kind).or_default() += 1;
}
let mut warn_kinds: Vec<(&&str, &i64)> = warn_by_kind.iter().collect();
warn_kinds.sort_by(|a, b| b.1.cmp(a.1));
let pct_of = |n: i64| -> String {
if total_successful_edits == 0 {
String::new()
} else {
format!(" ({:.2}% of {} successful edits)", pct(n, total_successful_edits), commas(total_successful_edits))
}
};
for (kind, n) in warn_kinds {
println!(" {:<16} {:>6}{}", kind, commas(*n), pct_of(*n));
}
let warn_show = show.min(warning_hits.len());
if warn_show > 0 {
println!();
println!("--- top {warn_show} self-correction examples (by input size) ---");
let mut sorted_warns = warning_hits.clone();
sorted_warns.sort_by(|a, b| b.input_len.cmp(&a.input_len));
for w in sorted_warns.iter().take(warn_show) {
println!(
"[{kind}] {files} ({len}B)",
kind = w.kind,
files = truncate_line(&w.files, 80),
len = w.input_len,
);
println!(" session={} call={}", w.session, w.call_id);
}
}
println!();
println!("=== same-locus re-edits (overlapping anchor ranges, both > max-fix) ===");
println!("hits: {}", commas(locus_hits.len() as i64));
let locus_show = show.min(locus_hits.len());
for h in locus_hits.iter().take(locus_show) {
println!(
"{file} first={a0}..{a1} ({fs}L) → second={b0}..{b1} ({ss}L) gap={gap}s",
file = h.file,
a0 = h.first_range.0,
a1 = h.first_range.1,
fs = h.first_size,
b0 = h.second_range.0,
b1 = h.second_range.1,
ss = h.second_size,
gap = h.gap_secs,
);
println!(" session={}", h.session);
println!(" first_call={} second_call={}", h.first_call_id, h.second_call_id);
}
println!();
println!("=== payload self-duplication (model pasted same N-line chunk twice in one payload) ===");
payload_dups.retain(|p| p.repeat_len >= min_dup);
payload_dups.sort_by(|a, b| b.repeat_len.cmp(&a.repeat_len));
println!(
"hits with repeat_len >= {}: {} ({:.2}% of {} successful edits)",
min_dup,
commas(payload_dups.len() as i64),
pct(payload_dups.len() as i64, total_successful_edits),
commas(total_successful_edits)
);
let dup_show = show.min(payload_dups.len());
for p in payload_dups.iter().take(dup_show) {
println!(
"k={k} block={blk}L {file}",
k = p.repeat_len,
blk = p.block_len,
file = p.file,
);
println!(" session={} call={}", p.session, p.call_id);
println!(" sample: {}", p.sample);
}
println!();
println!("=== same-anchor reused by multiple ops in one input ===");
anchor_dups.sort_by(|a, b| b.count.cmp(&a.count));
println!("hits: {}", commas(anchor_dups.len() as i64));
let anchor_show = show.min(anchor_dups.len());
for a in anchor_dups.iter().take(anchor_show) {
println!(
"anchor {anchor} referenced {n}x files={files}",
anchor = a.anchor,
n = a.count,
files = truncate_line(&a.files, 80),
);
println!(" session={} call={}", a.session, a.call_id);
}
Ok(())
}
-702
View File
@@ -1,702 +0,0 @@
//! `tools` subcommand — per-tool token totals across the most-recent N session
//! jsonl files.
//!
//! Token counting uses o200k_base via tiktoken-rs (the GPT-4o / GPT-5 family
//! tokenizer). It is not Claude's own BPE, but it is well-defined offline and
//! within ~5-10% across English/code in aggregate.
//!
//! Buckets:
//! tool ARGS — assistant tool-call argument JSON
//! tool RESULTS — tool result content text
//! assistant THINKING — assistant `thinking` blocks
//! assistant TEXT — assistant prose
//! user TEXT — user-authored text content
//!
//! Output: grand totals + per-tool breakdown sorted by total (arg+res) tokens.
//! Optional CSV at `$TOOL_USAGE_CSV` (per-tool totals) or
//! `--calls-csv PATH` / `$TOOL_CALLS_CSV` (one row per tool call).
//!
//! Pass `--by <h|d|w|m|Nh|Nd|Nw>` to bucket per-call data into rolling windows
//! and surface per-tool tokens/call (avg + p50 + p95) over time, so you can
//! spot regressions in tool efficiency. The buckets do not align to calendar
//! boundaries; they are pure `floor(unix_secs / N)` slices.
use crate::common::*;
use anyhow::{Context, Result, bail};
use std::collections::HashMap;
use std::fs::File;
use std::io::{BufRead, BufReader};
use std::path::Path;
#[derive(Default, Clone)]
struct ToolAgg {
calls: i64,
results: i64,
arg_tok: i64,
res_tok: i64,
}
#[derive(Default, Clone)]
struct SessionTotals {
arg_tok: i64,
res_tok: i64,
thinking_tok: i64,
text_tok: i64,
user_tok: i64,
n_calls: i64,
n_results: i64,
}
struct FileResult {
totals: SessionTotals,
tools: HashMap<String, ToolAgg>,
calls: Vec<CallRecord>,
}
#[derive(Clone)]
struct CallRecord {
ts: i64,
tool: String,
session: String,
model: String,
arg_tok: i32,
res_tok: i32,
}
struct PendingCall {
tool: String,
ts: i64,
arg_tok: i32,
model: String,
}
pub fn run(args: Vec<String>) -> Result<()> {
let mut limit: usize = 1_000;
let mut workers: usize = 0;
let mut by: Option<i64> = None;
let mut top: usize = 12;
let mut tool_filter: Option<String> = None;
let mut calls_csv: Option<String> = std::env::var("TOOL_CALLS_CSV")
.ok()
.filter(|s| !s.is_empty());
let mut iter = args.into_iter();
while let Some(a) = iter.next() {
match a.as_str() {
"-n" => {
limit = iter
.next()
.context("-n requires a value")?
.parse()
.context("-n value")?;
}
"-j" => {
workers = iter
.next()
.context("-j requires a value")?
.parse()
.context("-j value")?;
}
"--by" => {
let spec = iter.next().context("--by requires a bucket spec")?;
by = Some(parse_bucket(&spec)?);
}
"--top" => {
top = iter
.next()
.context("--top requires a value")?
.parse()
.context("--top value")?;
}
"--tool" => {
tool_filter = Some(iter.next().context("--tool requires a name")?);
}
"--calls-csv" => {
calls_csv = Some(iter.next().context("--calls-csv requires a path")?);
}
"-h" | "--help" => {
eprintln!(
"usage: session-stats tools [-n N] [-j workers] [--by SPEC] [--top N]
[--tool NAME] [--calls-csv PATH]
Aggregates per-tool token usage across the most-recent N session
jsonl files (default 1000). Tokenizer: o200k_base.
--by SPEC bucket per-call data into rolling windows. SPEC is one of:
hour, day, week, month, or <N>{{h,d,w}} (e.g. 7d, 12h, 2w).
Buckets are pure floor(unix_secs / N); they do not align to
calendar boundaries.
--top N limit per-tool series to the N most-called tools (default 12).
--tool NAME show only this tool in the bucketed series.
--calls-csv PATH
emit one CSV row per tool call (ts, session, tool, model,
arg_tok, res_tok). Env fallback: TOOL_CALLS_CSV.
TOOL_USAGE_CSV env still emits the per-tool grand-totals CSV."
);
return Ok(());
}
other => bail!("unknown flag: {other}"),
}
}
let files = collect_sessions(&WalkOpts {
date_filters: Vec::new(),
limit_most_recent: limit,
})?;
eprintln!(
"scanning {} session files (tokenizer: o200k_base)",
files.len()
);
let results = parallel_collect(&files, workers, 5_000, process_file);
let sessions = results.len();
let mut grand = SessionTotals::default();
let mut tools: HashMap<String, ToolAgg> = HashMap::new();
let mut all_calls: Vec<CallRecord> = Vec::new();
for r in results {
grand.arg_tok += r.totals.arg_tok;
grand.res_tok += r.totals.res_tok;
grand.thinking_tok += r.totals.thinking_tok;
grand.text_tok += r.totals.text_tok;
grand.user_tok += r.totals.user_tok;
grand.n_calls += r.totals.n_calls;
grand.n_results += r.totals.n_results;
for (name, t) in r.tools {
let dst = tools.entry(name).or_default();
dst.calls += t.calls;
dst.results += t.results;
dst.arg_tok += t.arg_tok;
dst.res_tok += t.res_tok;
}
all_calls.extend(r.calls);
}
if let Some(ref name) = tool_filter {
all_calls.retain(|c| &c.tool == name);
}
print_grand(&grand, sessions);
println!();
print_table(&tools);
write_csv(&tools)?;
if let Some(bucket_secs) = by {
println!();
print_buckets(&all_calls, bucket_secs, top, tool_filter.as_deref());
}
if let Some(path) = calls_csv {
write_calls_csv(&all_calls, &path)?;
eprintln!("wrote {} calls to {path}", commas(all_calls.len() as i64));
}
Ok(())
}
fn process_file(path: &Path) -> Option<FileResult> {
let f = match File::open(path) {
Ok(f) => f,
Err(e) => {
eprintln!("open {}: {e}", path.display());
return None;
}
};
let reader = BufReader::with_capacity(64 * 1024, f);
let mut totals = SessionTotals::default();
let mut tools: HashMap<String, ToolAgg> = HashMap::new();
let mut calls: Vec<CallRecord> = Vec::new();
let session = session_id_from_path(path);
// Pending call records: keyed by toolCallId so the matching toolResult
// can finalize a CallRecord with both arg+res tokens. Falls back to
// message.toolName when the id is absent (legacy sessions).
let mut pending: HashMap<String, PendingCall> = HashMap::new();
for line in reader.lines() {
let Ok(line) = line else { continue };
if line.is_empty() {
continue;
}
let Ok(ev) = serde_json::from_str::<RawEvent>(&line) else {
continue;
};
if ev.kind != "message" {
continue;
}
let Some(msg_raw) = ev.message else { continue };
let Ok(m) = serde_json::from_str::<Message>(msg_raw.get()) else {
continue;
};
let Some(content_raw) = m.content else { continue };
let items = parse_content(&content_raw);
match m.role.as_str() {
"assistant" => {
for it in items {
match it.kind.as_str() {
"toolCall" => {
let name = normalize_tool(&it.name);
let args_str = it.arguments.as_deref().map(RawValue::get).unwrap_or("");
let tok = count_tokens(args_str) as i64;
totals.arg_tok += tok;
totals.n_calls += 1;
let t = tools.entry(name.clone()).or_default();
t.calls += 1;
t.arg_tok += tok;
pending.insert(
it.id,
PendingCall {
tool: name,
ts: parse_ts(&ev.timestamp),
arg_tok: clamp_i32(tok),
model: m.model.clone(),
},
);
}
"thinking" => {
totals.thinking_tok += count_tokens(&it.thinking) as i64;
}
"text" => {
totals.text_tok += count_tokens(&it.text) as i64;
}
_ => {}
}
}
}
"toolResult" => {
let text = join_text(&items);
let tok = count_tokens(&text) as i64;
totals.res_tok += tok;
totals.n_results += 1;
let pc = pending.remove(&m.tool_call_id);
let name = pc
.as_ref()
.map(|p| p.tool.clone())
.unwrap_or_else(|| normalize_tool(&m.tool_name));
let t = tools.entry(name.clone()).or_default();
t.results += 1;
t.res_tok += tok;
if let Some(p) = pc {
calls.push(CallRecord {
ts: p.ts,
tool: name,
session: session.clone(),
model: p.model,
arg_tok: p.arg_tok,
res_tok: clamp_i32(tok),
});
}
}
"user" => {
for it in items {
if it.kind == "text" {
totals.user_tok += count_tokens(&it.text) as i64;
}
}
}
_ => {}
}
}
Some(FileResult { totals, tools, calls })
}
use serde_json::value::RawValue;
fn normalize_tool(name: &str) -> String {
if name.is_empty() {
"<unknown>".to_string()
} else {
name.to_string()
}
}
// ---- reporting ----
fn print_grand(g: &SessionTotals, sessions: usize) {
let total = g.arg_tok + g.res_tok + g.thinking_tok + g.text_tok + g.user_tok;
let rows: [(&str, i64); 5] = [
("tool call ARGS", g.arg_tok),
("tool RESULTS", g.res_tok),
("assistant THINKING", g.thinking_tok),
("assistant TEXT", g.text_tok),
("user TEXT", g.user_tok),
];
let label_w = rows.iter().map(|(l, _)| l.len()).max().unwrap_or(0);
let val_w = rows
.iter()
.map(|(_, n)| commas(*n).len())
.chain(std::iter::once(commas(total).len()))
.max()
.unwrap_or(0);
println!("=== Grand totals across {} sessions ===", commas(sessions as i64));
for (label, n) in rows {
println!(
"{label:<label_w$} {:>val_w$} tok ({:>5.1}%)",
commas(n),
pct(n, total),
);
}
println!("{:<label_w$} {}", "", "-".repeat(val_w));
println!("{:<label_w$} {:>val_w$} tok", "TOTAL", commas(total));
println!();
println!(
"tool calls: {}, tool results: {}",
commas(g.n_calls),
commas(g.n_results)
);
if g.n_calls > 0 {
println!(
"avg arg tokens / call: {:.1}",
g.arg_tok as f64 / g.n_calls as f64
);
}
if g.n_results > 0 {
println!(
"avg result tokens / call: {:.1}",
g.res_tok as f64 / g.n_results as f64
);
}
if g.arg_tok > 0 {
println!(
"ratio result / arg: {:.2}x",
g.res_tok as f64 / g.arg_tok as f64
);
}
}
struct ToolRow {
name: String,
calls: i64,
arg_tok: i64,
res_tok: i64,
total: i64,
avg_arg: f64,
avg_res: f64,
res_o_arg: f64,
}
fn print_table(tools: &HashMap<String, ToolAgg>) {
let mut rows: Vec<ToolRow> = tools
.iter()
.filter_map(|(name, t)| {
if t.calls == 0 && t.results == 0 {
return None;
}
let mut r = ToolRow {
name: name.clone(),
calls: t.calls,
arg_tok: t.arg_tok,
res_tok: t.res_tok,
total: t.arg_tok + t.res_tok,
avg_arg: 0.0,
avg_res: 0.0,
res_o_arg: 0.0,
};
if t.calls > 0 {
r.avg_arg = t.arg_tok as f64 / t.calls as f64;
r.avg_res = t.res_tok as f64 / t.calls as f64;
}
if t.arg_tok > 0 {
r.res_o_arg = t.res_tok as f64 / t.arg_tok as f64;
}
Some(r)
})
.collect();
rows.sort_by(|a, b| b.total.cmp(&a.total));
const TOP: usize = 25;
let shown = TOP.min(rows.len());
let head_rows = &rows[..shown];
// "(N others)" trailing summary, computed before width measurement so its
// string contents participate in column sizing.
let others = (rows.len() > TOP).then(|| {
let mut sc = 0i64;
let mut sa = 0i64;
let mut sr = 0i64;
for r in &rows[TOP..] {
sc += r.calls;
sa += r.arg_tok;
sr += r.res_tok;
}
OthersRow {
label: format!("({} others)", rows.len() - TOP),
calls: sc,
arg_tok: sa,
res_tok: sr,
total: sa + sr,
}
});
// Compute column widths from header label and every value that will
// appear under that header (including the "others" summary row, if any).
let max_str = |header: &str, vals: &[&str]| -> usize {
vals.iter().map(|s| s.len()).chain(std::iter::once(header.len())).max().unwrap_or(0)
};
let names: Vec<&str> = head_rows
.iter()
.map(|r| r.name.as_str())
.chain(others.as_ref().map(|o| o.label.as_str()))
.collect();
let calls: Vec<String> = head_rows
.iter()
.map(|r| commas(r.calls))
.chain(others.as_ref().map(|o| commas(o.calls)))
.collect();
let arg_toks: Vec<String> = head_rows
.iter()
.map(|r| commas(r.arg_tok))
.chain(others.as_ref().map(|o| commas(o.arg_tok)))
.collect();
let res_toks: Vec<String> = head_rows
.iter()
.map(|r| commas(r.res_tok))
.chain(others.as_ref().map(|o| commas(o.res_tok)))
.collect();
let totals: Vec<String> = head_rows
.iter()
.map(|r| commas(r.total))
.chain(others.as_ref().map(|o| commas(o.total)))
.collect();
let avg_args: Vec<String> = head_rows.iter().map(|r| format!("{:.1}", r.avg_arg)).collect();
let avg_ress: Vec<String> = head_rows.iter().map(|r| format!("{:.1}", r.avg_res)).collect();
let res_o_args: Vec<String> =
head_rows.iter().map(|r| format!("{:.2}", r.res_o_arg)).collect();
let name_w = max_str("tool", &names);
let calls_w = max_str("calls", &calls.iter().map(String::as_str).collect::<Vec<_>>());
let arg_w = max_str("arg_tok", &arg_toks.iter().map(String::as_str).collect::<Vec<_>>());
let res_w = max_str("res_tok", &res_toks.iter().map(String::as_str).collect::<Vec<_>>());
let tot_w = max_str("total", &totals.iter().map(String::as_str).collect::<Vec<_>>());
let avga_w = max_str("avg_arg", &avg_args.iter().map(String::as_str).collect::<Vec<_>>());
let avgr_w = max_str("avg_res", &avg_ress.iter().map(String::as_str).collect::<Vec<_>>());
let ratio_w = max_str("res/arg", &res_o_args.iter().map(String::as_str).collect::<Vec<_>>());
let total_width = name_w + 1 + calls_w + 1 + arg_w + 1 + res_w + 1 + tot_w + 1 + avga_w + 1 + avgr_w + 1 + ratio_w;
println!(
"{:<name_w$} {:>calls_w$} {:>arg_w$} {:>res_w$} {:>tot_w$} {:>avga_w$} {:>avgr_w$} {:>ratio_w$}",
"tool", "calls", "arg_tok", "res_tok", "total", "avg_arg", "avg_res", "res/arg"
);
println!("{}", "-".repeat(total_width));
for (i, r) in head_rows.iter().enumerate() {
println!(
"{:<name_w$} {:>calls_w$} {:>arg_w$} {:>res_w$} {:>tot_w$} {:>avga_w$} {:>avgr_w$} {:>ratio_w$}",
r.name, calls[i], arg_toks[i], res_toks[i], totals[i], avg_args[i], avg_ress[i], res_o_args[i],
);
}
if let Some(o) = others {
let i = head_rows.len();
println!(
"{:<name_w$} {:>calls_w$} {:>arg_w$} {:>res_w$} {:>tot_w$}",
o.label, calls[i], arg_toks[i], res_toks[i], totals[i],
);
}
}
struct OthersRow {
label: String,
calls: i64,
arg_tok: i64,
res_tok: i64,
total: i64,
}
fn write_csv(tools: &HashMap<String, ToolAgg>) -> Result<()> {
let path = std::env::var("TOOL_USAGE_CSV").unwrap_or_default();
if path.is_empty() {
return Ok(());
}
let f = File::create(&path).with_context(|| format!("create {path}"))?;
let mut w = csv::Writer::from_writer(f);
w.write_record(["tool", "calls", "results", "arg_tok", "res_tok", "total"])?;
let mut names: Vec<&String> = tools.keys().collect();
names.sort_by(|a, b| {
let ai = {
let t = &tools[a.as_str()];
t.arg_tok + t.res_tok
};
let aj = {
let t = &tools[b.as_str()];
t.arg_tok + t.res_tok
};
aj.cmp(&ai)
});
for n in names {
let t = &tools[n.as_str()];
w.write_record([
n.as_str(),
&t.calls.to_string(),
&t.results.to_string(),
&t.arg_tok.to_string(),
&t.res_tok.to_string(),
&(t.arg_tok + t.res_tok).to_string(),
])?;
}
w.flush()?;
Ok(())
}
// ---- per-call helpers ----
fn session_id_from_path(path: &Path) -> String {
path.file_stem()
.and_then(|s| s.to_str())
.unwrap_or("")
.to_string()
}
fn clamp_i32(n: i64) -> i32 {
n.clamp(0, i32::MAX as i64) as i32
}
fn print_buckets(
calls: &[CallRecord],
bucket_secs: i64,
top: usize,
tool_filter: Option<&str>,
) {
if calls.is_empty() {
println!("(no per-call records — no toolCall/toolResult pairs found)");
return;
}
// Pick the tools to show: top-N by call count, ignoring records with ts==0
// (events that lacked a parseable timestamp).
let mut per_tool: HashMap<&str, i64> = HashMap::new();
for c in calls {
if c.ts == 0 {
continue;
}
*per_tool.entry(c.tool.as_str()).or_default() += 1;
}
let mut ranked: Vec<(&str, i64)> = per_tool.into_iter().collect();
ranked.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(b.0)));
if tool_filter.is_none() {
ranked.truncate(top);
}
let bucket_human = humanize_bucket(bucket_secs);
println!(
"=== per-call tokens by {} bucket (rolling, no calendar alignment) ===",
bucket_human
);
println!(
"showing {} tool{} (sorted by call count). totals/calls match per-call records only;",
ranked.len(),
if ranked.len() == 1 { "" } else { "s" }
);
println!(
"calls without a matching toolResult or without a parseable timestamp are skipped here."
);
// Group calls per (tool, bucket_id).
let mut by_bucket: HashMap<(&str, i64), Vec<&CallRecord>> = HashMap::new();
for c in calls {
if c.ts == 0 {
continue;
}
if !ranked.iter().any(|(t, _)| *t == c.tool.as_str()) {
continue;
}
let bid = c.ts.div_euclid(bucket_secs);
by_bucket.entry((c.tool.as_str(), bid)).or_default().push(c);
}
for (tool, total_calls) in ranked {
let mut bids: Vec<i64> = by_bucket
.keys()
.filter(|(t, _)| *t == tool)
.map(|(_, b)| *b)
.collect();
if bids.is_empty() {
continue;
}
bids.sort();
println!();
println!("=== {tool} ({} calls) ===", commas(total_calls));
println!(
"{:<22} {:>7} {:>9} {:>9} {:>9} {:>9} {:>9}",
"window", "calls", "avg_arg", "avg_res", "avg_tot", "p50_tot", "p95_tot"
);
let dashes = 22 + 1 + 7 + 5 * (1 + 9);
println!("{}", "-".repeat(dashes));
for bid in bids {
let records = by_bucket.get(&(tool, bid)).expect("present");
let n = records.len() as i64;
let mut sum_arg = 0i64;
let mut sum_res = 0i64;
let mut totals: Vec<i64> = Vec::with_capacity(records.len());
for r in records {
sum_arg += r.arg_tok as i64;
sum_res += r.res_tok as i64;
totals.push(r.arg_tok as i64 + r.res_tok as i64);
}
let avg_arg = sum_arg as f64 / n as f64;
let avg_res = sum_res as f64 / n as f64;
let avg_tot = avg_arg + avg_res;
let p50 = percentile(&mut totals.clone(), 50.0);
let p95 = percentile(&mut totals, 95.0);
let label = bucket_label(bid * bucket_secs, bucket_secs);
println!(
"{:<22} {:>7} {:>9} {:>9} {:>9} {:>9} {:>9}",
label,
commas(n),
commas(avg_arg.round() as i64),
commas(avg_res.round() as i64),
commas(avg_tot.round() as i64),
commas(p50.round() as i64),
commas(p95.round() as i64),
);
}
}
}
fn humanize_bucket(bucket_secs: i64) -> String {
if bucket_secs % (7 * 86_400) == 0 {
let n = bucket_secs / (7 * 86_400);
if n == 1 { "7d".into() } else { format!("{n}w") }
} else if bucket_secs % 86_400 == 0 {
format!("{}d", bucket_secs / 86_400)
} else if bucket_secs % 3600 == 0 {
format!("{}h", bucket_secs / 3600)
} else {
format!("{}s", bucket_secs)
}
}
fn write_calls_csv(calls: &[CallRecord], path: &str) -> Result<()> {
let f = File::create(path).with_context(|| format!("create {path}"))?;
let mut w = csv::Writer::from_writer(f);
w.write_record([
"ts_unix",
"ts_iso",
"session",
"tool",
"model",
"arg_tok",
"res_tok",
"total_tok",
])?;
for c in calls {
let total = c.arg_tok as i64 + c.res_tok as i64;
w.write_record([
&c.ts.to_string(),
&format_iso(c.ts),
&c.session,
&c.tool,
&c.model,
&c.arg_tok.to_string(),
&c.res_tok.to_string(),
&total.to_string(),
])?;
}
w.flush()?;
Ok(())
}
-344
View File
@@ -1,344 +0,0 @@
//! Shared JSONL shapes, walk helpers, tokenizer, and formatting helpers.
use anyhow::{Context, Result};
use rayon::prelude::*;
use serde::Deserialize;
use serde_json::value::RawValue;
use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::sync::LazyLock;
use std::sync::atomic::{AtomicU64, Ordering};
use std::time::SystemTime;
use tiktoken_rs::CoreBPE;
use walkdir::WalkDir;
// ---- jsonl shapes ----
#[derive(Deserialize)]
pub struct RawEvent {
#[serde(rename = "type", default)]
pub kind: String,
#[serde(default)]
pub message: Option<Box<RawValue>>,
#[serde(default)]
pub timestamp: String,
}
#[derive(Deserialize)]
pub struct Message {
#[serde(default)]
pub role: String,
#[serde(default)]
pub content: Option<Box<RawValue>>,
#[serde(default, rename = "toolName")]
pub tool_name: String,
#[serde(default, rename = "toolCallId")]
pub tool_call_id: String,
#[serde(default)]
pub model: String,
}
#[derive(Deserialize)]
pub struct ContentItem {
#[serde(rename = "type", default)]
pub kind: String,
#[serde(default)]
pub text: String,
#[serde(default)]
pub thinking: String,
#[serde(default)]
pub name: String,
#[serde(default)]
pub id: String,
#[serde(default)]
pub arguments: Option<Box<RawValue>>,
}
// ---- session walking ----
pub fn sessions_root() -> Result<PathBuf> {
let home = dirs::home_dir().context("could not resolve home directory")?;
Ok(home.join(".omp").join("agent").join("sessions"))
}
pub struct WalkOpts {
/// Keeps only paths containing any of these substrings (e.g. "2026-04-28").
/// Empty means accept all.
pub date_filters: Vec<String>,
/// Keeps only the N most-recently-modified files (after the date filter).
/// 0 means no limit.
pub limit_most_recent: usize,
}
/// Walks the sessions root and returns the matching `.jsonl` paths.
/// With `limit_most_recent > 0` the result is sorted by mtime descending and
/// truncated to N entries; otherwise it's lexically sorted.
pub fn collect_sessions(opts: &WalkOpts) -> Result<Vec<PathBuf>> {
let base = sessions_root()?;
let need_mtime = opts.limit_most_recent > 0;
let mut all: Vec<(PathBuf, SystemTime)> = Vec::new();
for entry in WalkDir::new(&base).into_iter().filter_map(Result::ok) {
if !entry.file_type().is_file() {
continue;
}
let p = entry.path();
if p.extension().and_then(|e| e.to_str()) != Some("jsonl") {
continue;
}
let path_str = p.to_string_lossy();
if !match_date(&path_str, &opts.date_filters) {
continue;
}
let mt = if need_mtime {
entry
.metadata()
.ok()
.and_then(|m| m.modified().ok())
.unwrap_or(SystemTime::UNIX_EPOCH)
} else {
SystemTime::UNIX_EPOCH
};
all.push((p.to_path_buf(), mt));
}
if need_mtime {
all.sort_by(|a, b| b.1.cmp(&a.1));
all.truncate(opts.limit_most_recent);
} else {
all.sort_by(|a, b| a.0.cmp(&b.0));
}
Ok(all.into_iter().map(|(p, _)| p).collect())
}
fn match_date(p: &str, filters: &[String]) -> bool {
filters.is_empty() || filters.iter().any(|d| p.contains(d))
}
// ---- content helpers ----
pub fn parse_content(raw: &RawValue) -> Vec<ContentItem> {
serde_json::from_str(raw.get()).unwrap_or_default()
}
/// Concatenates all `text` items in a content array.
pub fn join_text(items: &[ContentItem]) -> String {
let mut out = String::new();
for it in items {
if it.kind == "text" {
out.push_str(&it.text);
}
}
out
}
// ---- tokenizer (o200k_base) ----
static BPE: LazyLock<CoreBPE> =
LazyLock::new(|| tiktoken_rs::o200k_base().expect("load o200k_base BPE"));
/// Counts tokens for `s` using the o200k_base BPE (GPT-4o / GPT-5 family).
/// Uses the ordinary encoder so embedded `<|...|>` sequences in tool args do
/// not trigger special-token handling.
pub fn count_tokens(s: &str) -> usize {
if s.is_empty() {
return 0;
}
BPE.encode_ordinary(s).len()
}
// ---- formatting helpers ----
/// Formats an integer with thousand separators.
pub fn commas(n: i64) -> String {
let neg = n < 0;
let mag = if neg { (n as i128).unsigned_abs() } else { n as u128 };
let s = mag.to_string();
let bytes = s.as_bytes();
let mut out = String::with_capacity(bytes.len() + bytes.len() / 3 + 1);
if neg {
out.push('-');
}
let pre = bytes.len() % 3;
if pre > 0 {
out.push_str(&s[..pre]);
if bytes.len() > pre {
out.push(',');
}
}
let mut i = pre;
while i + 3 <= bytes.len() {
out.push_str(&s[i..i + 3]);
if i + 3 < bytes.len() {
out.push(',');
}
i += 3;
}
out
}
pub fn pct(a: i64, b: i64) -> f64 {
if b == 0 {
0.0
} else {
100.0 * a as f64 / b as f64
}
}
/// Truncates a string to at most `n` chars, replacing newlines with " | ".
/// Adds an ellipsis when truncation occurs.
pub fn truncate_line(s: &str, n: usize) -> String {
let s = s.replace('\n', " | ");
if s.chars().count() <= n {
return s;
}
let mut out: String = s.chars().take(n).collect();
out.push('…');
out
}
/// Returns map keys sorted by descending value, ties broken alphabetically.
pub fn sorted_by_count(m: &HashMap<String, i64>) -> Vec<&String> {
let mut keys: Vec<&String> = m.keys().collect();
keys.sort_by(|a, b| {
let av = m.get(a.as_str()).copied().unwrap_or(0);
let bv = m.get(b.as_str()).copied().unwrap_or(0);
bv.cmp(&av).then_with(|| a.cmp(b))
});
keys
}
pub fn print_sorted(m: &HashMap<String, i64>) {
print_sorted_indent(m, " ");
}
pub fn print_sorted_indent(m: &HashMap<String, i64>, indent: &str) {
for k in sorted_by_count(m) {
println!("{indent}{k:<25} {}", m[k.as_str()]);
}
}
// ---- parallel processing ----
/// Runs `handle(path)` in parallel across rayon workers and collects the
/// non-`None` results into a Vec. Logs progress every `progress_every` files
/// (set 0 to silence).
pub fn parallel_collect<R, H>(
paths: &[PathBuf],
workers: usize,
progress_every: u64,
handle: H,
) -> Vec<R>
where
R: Send,
H: Fn(&Path) -> Option<R> + Sync,
{
let total = paths.len();
let done = AtomicU64::new(0);
let pool = {
let mut b = rayon::ThreadPoolBuilder::new();
if workers > 0 {
b = b.num_threads(workers);
}
b.build().expect("rayon thread pool")
};
pool.install(|| {
paths
.par_iter()
.filter_map(|p| {
let r = handle(p);
let n = done.fetch_add(1, Ordering::Relaxed) + 1;
if progress_every > 0 && n % progress_every == 0 {
eprintln!(" processed {n}/{total}");
}
r
})
.collect()
})
}
// ---- timestamp / bucket helpers ----
/// Parses an RFC3339 timestamp into unix seconds. Returns 0 on failure.
pub fn parse_ts(s: &str) -> i64 {
if s.is_empty() {
return 0;
}
chrono::DateTime::parse_from_rfc3339(s)
.map(|dt| dt.timestamp())
.unwrap_or(0)
}
/// Parses a bucket spec like "h", "day", "week", "month", "1h", "12h", "7d", "2w".
/// Returns the bucket size in seconds.
pub fn parse_bucket(spec: &str) -> anyhow::Result<i64> {
use anyhow::bail;
let s = spec.trim();
let secs: i64 = match s {
"" => bail!("empty bucket spec"),
"hour" | "h" | "1h" => 3600,
"day" | "d" | "1d" => 86_400,
"week" | "w" | "1w" | "7d" => 7 * 86_400,
"month" | "mo" | "1mo" | "30d" => 30 * 86_400,
other => {
let bytes = other.as_bytes();
let last = *bytes.last().unwrap();
let unit_secs: i64 = match last {
b'h' => 3600,
b'd' => 86_400,
b'w' => 7 * 86_400,
_ => bail!("bad bucket spec {spec:?} (use h/d/w or e.g. 7d, 12h)"),
};
let n: i64 = other[..other.len() - 1]
.parse()
.map_err(|_| anyhow::anyhow!("bad bucket count in {spec:?}"))?;
if n <= 0 {
bail!("bucket count must be > 0");
}
n * unit_secs
}
};
Ok(secs)
}
/// Returns a label for the bucket starting at `start_secs`.
/// Buckets shorter than a day include the hour; multi-day buckets show start..end (exclusive).
pub fn bucket_label(start_secs: i64, bucket_secs: i64) -> String {
use chrono::TimeZone;
let start = chrono::Utc.timestamp_opt(start_secs, 0).single();
let end = chrono::Utc.timestamp_opt(start_secs + bucket_secs - 1, 0).single();
match (start, end) {
(Some(s), _) if bucket_secs < 86_400 => s.format("%Y-%m-%d %H:00").to_string(),
(Some(s), _) if bucket_secs == 86_400 => s.format("%Y-%m-%d").to_string(),
(Some(s), Some(e)) => format!(
"{}..{}",
s.format("%Y-%m-%d"),
e.format("%Y-%m-%d")
),
_ => start_secs.to_string(),
}
}
/// Formats a unix timestamp as an RFC3339 string (UTC).
pub fn format_iso(unix_secs: i64) -> String {
use chrono::TimeZone;
chrono::Utc
.timestamp_opt(unix_secs, 0)
.single()
.map(|dt| dt.format("%Y-%m-%dT%H:%M:%SZ").to_string())
.unwrap_or_default()
}
/// Computes a percentile over an integer slice. The slice is sorted in place.
/// `p` is 0..=100.
pub fn percentile(v: &mut [i64], p: f64) -> f64 {
if v.is_empty() {
return 0.0;
}
v.sort_unstable();
let p = p.clamp(0.0, 100.0);
let idx = ((p / 100.0) * (v.len() as f64 - 1.0)).round() as usize;
v[idx.min(v.len() - 1)] as f64
}
-63
View File
@@ -1,63 +0,0 @@
//! session-stats: ad-hoc analyses over the local agent session corpus
//! (`~/.omp/agent/sessions/`).
//!
//! Subcommands:
//!
//! edits [-n N] [date ...] audit edit/ast_edit/write tool usage by argument schema
//! tools [-n N] per-tool token totals across the most-recent N sessions
//!
//! Run with no subcommand for help.
mod cmd_edits;
mod cmd_followups;
mod cmd_tools;
mod common;
use std::process::ExitCode;
fn usage() {
eprintln!(
"usage: session-stats <subcommand> [args...]
edits [-n N] [date prefix ...]
audit edit-tool usage across N most-recent
sessions (default 1000). Optional date filters
(e.g. 2026-04-28) further narrow the set.
tools [-n N] [--by SPEC] [--top N] [--tool NAME] [--calls-csv PATH]
per-tool token totals across the N most-recent
session jsonl files (default 1000). With --by,
also bucket per-call tokens (avg + p50/p95) into
rolling windows so you can spot regressions.
Token counting uses the o200k_base tokenizer (the GPT-4o / Claude-adjacent BPE).
Walk root: ~/.omp/agent/sessions/"
);
}
fn main() -> ExitCode {
let mut args = std::env::args().skip(1);
let Some(cmd) = args.next() else {
usage();
return ExitCode::from(2);
};
let rest: Vec<String> = args.collect();
let result = match cmd.as_str() {
"edits" => cmd_edits::run(rest),
"followups" => cmd_followups::run(rest),
"tools" => cmd_tools::run(rest),
"-h" | "--help" | "help" => {
usage();
return ExitCode::SUCCESS;
}
other => {
eprintln!("unknown subcommand {other:?}\n");
usage();
return ExitCode::from(2);
}
};
if let Err(err) = result {
eprintln!("fatal: {err:#}");
return ExitCode::FAILURE;
}
ExitCode::SUCCESS
}
File diff suppressed because it is too large Load Diff