diff --git a/.gitignore b/.gitignore index f61c5d5f9..b17590ad7 100644 --- a/.gitignore +++ b/.gitignore @@ -58,11 +58,6 @@ pi-*.html packages/coding-agent/src/internal-urls/docs-index.generated.ts /runs/ python/omp-rpc/src/omp_rpc.egg-info/ - -scripts/session-stats/Cargo.lock - -scripts/session-stats/edit-analysis.csv - # parallel-agent worktrees .wt/ CPU*.md diff --git a/package.json b/package.json index 0bba8f024..e2e5d3689 100644 --- a/package.json +++ b/package.json @@ -115,9 +115,10 @@ "ci:release:publish": "bun scripts/ci-release-publish.ts", "bench:gen-fixtures": "bun --cwd=packages/typescript-edit-benchmark run src/generate.ts --typescript-dir /tmp/typescript-source --count-per-type 8", "bench:edit": "bun --cwd=packages/typescript-edit-benchmark run start", - "stats:run": "cargo run --release --manifest-path scripts/session-stats/Cargo.toml --", - "stats:edits": "cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits", - "stats:tools": "cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools", + "stats:sync": "python3 scripts/session-stats/sync.py", + "stats:tools": "python3 scripts/session-stats/analyze.py tools", + "stats:edits": "python3 scripts/session-stats/analyze.py edits", + "stats:followups": "python3 scripts/session-stats/analyze.py followups", "prepublishOnly": "bun run check", "prepare": "bun --cwd=packages/coding-agent run generate-docs-index", "publish": "bun run prepublishOnly && npm publish -ws --access public", diff --git a/packages/ai/src/model-thinking.ts b/packages/ai/src/model-thinking.ts index 3264551d4..943b96e1b 100644 --- a/packages/ai/src/model-thinking.ts +++ b/packages/ai/src/model-thinking.ts @@ -314,7 +314,10 @@ function applyGeneratedModelPolicy(model: ApiModel): void { model.maxTokens = copilotLimits.maxTokens; } - if (model.api === "openai-completions" && (model.provider === "minimax-code" || model.provider === "minimax-code-cn")) { + if ( + model.api === "openai-completions" && + (model.provider === "minimax-code" || model.provider === "minimax-code-cn") + ) { model.compat = { ...model.compat, supportsStore: false, diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index 138f67b9e..d08a8bc93 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -20,8 +20,7 @@ At least 5 equal signs on each side. Content between one header and the next (or - Pass multiple small cells in one call. - Define small reusable functions for individual debugging. - Put workflow explanations in the assistant message or cell title — never inside cell code. -{{#if py}}- Python cells run inside an IPython kernel with a live event loop. Use top-level `await` directly (e.g. `await main()`); `asyncio.run(...)` raises "cannot be called from a running event loop".{{/if}} - +{{#if py}}- Python cells run inside an IPython kernel with a live event loop. Use top-level `await` directly (e.g. `await main()`); `asyncio.run(…)` raises "cannot be called from a running event loop".{{/if}} **On failure:** errors identify the failing cell (e.g., "Cell 3 failed"). Resubmit only the fixed cell (or fixed cell + remaining cells). diff --git a/packages/coding-agent/test/issue-953-repro.test.ts b/packages/coding-agent/test/issue-953-repro.test.ts index 94ed4b81b..2c4e4194f 100644 --- a/packages/coding-agent/test/issue-953-repro.test.ts +++ b/packages/coding-agent/test/issue-953-repro.test.ts @@ -1,6 +1,6 @@ import { beforeAll, describe, expect, it } from "bun:test"; -import { renderSegment } from "../src/modes/components/status-line/segments"; import type { SegmentContext } from "../src/modes/components/status-line/segments"; +import { renderSegment } from "../src/modes/components/status-line/segments"; import { initTheme, theme } from "../src/modes/theme/theme"; beforeAll(async () => { diff --git a/scripts/session-stats/Cargo.toml b/scripts/session-stats/Cargo.toml deleted file mode 100644 index 1f3f048c5..000000000 --- a/scripts/session-stats/Cargo.toml +++ /dev/null @@ -1,32 +0,0 @@ -[package] -name = "session-stats" -version = "0.1.0" -edition = "2024" -publish = false - -# Standalone crate: do not inherit from the parent workspace. -[workspace] - -[[bin]] -name = "session-stats" -path = "src/main.rs" - -[dependencies] -chrono = { version = "0.4", default-features = false, features = ["std", "clock"] } -anyhow = "1" -csv = "1" -dirs = "6" -rayon = "1.12" -regex = "1" -serde = { version = "1", features = ["derive"] } -serde_json = { version = "1", features = ["raw_value"] } -tiktoken-rs = "0.11" -walkdir = "2" - -[profile.release] -opt-level = 3 -lto = "thin" -codegen-units = 1 -debug = "line-tables-only" -split-debuginfo = "off" -strip = "none" diff --git a/scripts/session-stats/README.md b/scripts/session-stats/README.md index 255e23c1a..53021900e 100644 --- a/scripts/session-stats/README.md +++ b/scripts/session-stats/README.md @@ -1,82 +1,70 @@ # session-stats Ad-hoc analyses over the local agent session corpus -(`~/.omp/agent/sessions/`). Single Rust binary with subcommands. - -## Subcommands - -### `edits` — edit-tool reliability audit - -Audits how agents have used the `edit` / `ast_edit` / `write` tools. - -For each call we: - -- detect the **argument-schema family** in use (the edit tool has shipped many - shapes over time: `oldText/newText`, `op+pos+end+lines`, `loc+content`, - `loc+splice/pre/post/sed`, etc.); -- record the locator shape and verb combination (for the current schema); -- pair the call with its `toolResult` and classify the outcome - (`success` / `truncated` / `aborted` / `fail:anchor-stale` / - `fail:no-match` / `fail:parse` / `fail:no-enclosing-block` / …). - -Output: markdown-ish report on stdout plus per-call CSV at `$EDIT_ANALYSIS_CSV` -(default `./edit-analysis.csv`). - -### `tools` — per-tool token budget - -Aggregates token usage across the most-recent N sessions. Buckets: - -- `tool ARGS` — assistant tool-call argument JSON -- `tool RESULTS` — tool result content text -- `assistant THINKING` — assistant `thinking` blocks -- `assistant TEXT` — assistant prose -- `user TEXT` — user-authored text content - -Token counting uses **`o200k_base`** via `tiktoken-rs` (the GPT-4o / GPT-5 -family BPE — well-defined offline and within ~5-10% of Claude's own counts in -aggregate across English/code). - -Output: grand totals + per-tool breakdown sorted by total (arg+res) tokens. -Optional CSV at `$TOOL_USAGE_CSV`. - -## Usage - -```sh -# Edit audit on the most-recent sessions. -cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits - -# Edit audit on the 200 most-recent sessions. -cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits -n 200 - -# Edit audit on a specific date. -cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits 2026-04-28 - -# Tool token budget on the 1000 most-recent sessions. -cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 1000 - -# Tool token budget on every jsonl on disk. -cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 0 - -# Dump per-tool CSV alongside the report. -TOOL_USAGE_CSV=tools.csv \ - cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 200 -``` - -The walk root is `~/.omp/agent/sessions/`. Subagent jsonls -(`/-.jsonl`) count as their own session and are included -in the recency window independently. +(`~/.omp/agent/sessions/`). SQLite-backed; data is synced once into the same +`~/.omp/stats.db` that `packages/stats` uses, then queried by short Python +scripts. ## Layout ``` scripts/session-stats/ - Cargo.toml - src/ - main.rs # subcommand dispatch - common.rs # shared JSONL shapes, walk, tokenizer, formatting helpers - cmd_edits.rs # edits subcommand - cmd_tools.rs # tools subcommand + sync.py # walks ~/.omp/agent/sessions/ and populates ss_* tables + analyze.py # tools | edits | followups subcommands over the synced db ``` -The crate is a standalone Cargo project (it carries its own `[workspace]` -declaration) so it does not perturb the main workspace's lockfile. +## One-time prep + +```sh +pip install tiktoken +``` + +## Sync + +```sh +bun run stats:sync # incremental +python3 scripts/session-stats/sync.py --workers 16 --full # rebuild all +python3 scripts/session-stats/sync.py --limit 200 # newest 200 only +``` + +The sync is incremental: per-file `mtime`, `size`, `byte_offset`, and +`parser_version` are tracked in `ss_sessions`. Re-runs only parse new bytes +and only re-tokenize / re-classify what changed. A bump of `EDIT_PARSER_VERSION` +in `sync.py` invalidates `ss_edit_*` rows on next sync. + +Tokenization is `o200k_base` (GPT-4o / GPT-5 family) via tiktoken — well +within ~5–10% of Claude's BPE in aggregate. + +## Schema + +All tables are prefixed `ss_` to avoid collision with `packages/stats`. + +|Table|Granularity| +|---|---| +|`ss_sessions`|one row per `.jsonl`; carries sync state + session metadata| +|`ss_tool_calls`|one row per `toolCall` content block (`arg_json`, `arg_tokens`)| +|`ss_tool_results`|one row per `toolResult` message (`result_text`, `result_tokens`, `is_error`)| +|`ss_assistant_msgs`|per assistant message text + thinking blobs and token counts| +|`ss_user_msgs`|per user message text and token count| +|`ss_edit_calls`|per `edit` call: `success`, `warnings`, `raw_input_len`| +|`ss_edit_sections`|per `@PATH` section in an edit; precomputed `longest_repeat_*`, `dup_anchors`| + +Indexes on `(tool_name, timestamp)` and `(session_file, seq)` make per-tool +aggregations and ordered session walks cheap. + +## Analyses + +```sh +bun run stats:tools # per-tool token totals +bun run stats:tools -- --by d --top 8 # bucket by day, top 8 tools each +bun run stats:edits # edit-tool reliability audit +bun run stats:followups # five hashline-edit detectors +bun run stats:followups -- --max-fix 2 --min-dup 8 --show 20 +``` + +All three accept `-n N` / `--folder SUBSTR` to scope the query. + +The Rust crate that previously lived here was retired in favor of this +SQLite-backed flow. The schema persists everything the analyses used to +recompute on every run (token counts, hashline parse output, success flags), +so subsequent invocations are sub-second over the full corpus. diff --git a/scripts/session-stats/analyze.py b/scripts/session-stats/analyze.py new file mode 100644 index 000000000..8c60dd097 --- /dev/null +++ b/scripts/session-stats/analyze.py @@ -0,0 +1,861 @@ +#!/usr/bin/env python3 +""" +Analyses over the session-stats sqlite tables (`ss_*`) populated by sync.py. + +Subcommands: + tools — per-tool token totals (port of cmd_tools.rs) + edits — edit-tool reliability audit (port of cmd_edits.rs) + followups — five hashline-edit detectors (port of cmd_followups.rs) + +Each subcommand reads from ~/.omp/stats.db. Run sync.py first. +""" + +from __future__ import annotations + +import argparse +import json +import re +import sqlite3 +import sys +from collections import Counter, defaultdict +from pathlib import Path + +DB_PATH = Path.home() / ".omp" / "stats.db" + + +# --------------------------------------------------------------------------- # +# Shared helpers + +def open_ro() -> sqlite3.Connection: + if not DB_PATH.exists(): + sys.exit(f"db not found: {DB_PATH}. Run sync.py first.") + conn = sqlite3.connect(f"file:{DB_PATH}?mode=ro", uri=True) + conn.row_factory = sqlite3.Row + return conn + + +def commas(n: int) -> str: + return f"{n:,}" + + +def pct(part: int, total: int) -> float: + return 0.0 if total == 0 else (100.0 * part / total) + + +def truncate_line(s: str, n: int) -> str: + s = s.replace("\n", " | ") + if len(s) <= n: + return s + return s[: n - 1] + "…" + + +def parse_bucket(spec: str) -> int: + """`h`,`d`,`w`,`m`,`h`,`d`,`w` -> seconds.""" + units = {"h": 3600, "d": 86400, "w": 604800, "m": 2592000} + if spec in units: + return units[spec] + if spec[-1] in units and spec[:-1].isdigit(): + return int(spec[:-1]) * units[spec[-1]] + if spec == "hour": + return 3600 + if spec == "day": + return 86400 + if spec == "week": + return 604800 + raise ValueError(f"bad --by spec: {spec}") + + +def percentile(values: list[int], p: float) -> float: + if not values: + return 0.0 + s = sorted(values) + k = (len(s) - 1) * (p / 100.0) + lo, hi = int(k), min(int(k) + 1, len(s) - 1) + if lo == hi: + return float(s[lo]) + return s[lo] + (s[hi] - s[lo]) * (k - lo) + + +# --------------------------------------------------------------------------- # +# `tools` — per-tool token totals (cmd_tools.rs port) + +TOOLS_AGGREGATE_SQL = """ +WITH per_tool AS ( + SELECT + c.tool_name, + COUNT(*) AS calls, + IFNULL(SUM(c.arg_tokens), 0) AS arg_tok + FROM ss_tool_calls c + GROUP BY c.tool_name +), +per_tool_res AS ( + SELECT + r.tool_name, + COUNT(*) AS results, + IFNULL(SUM(r.result_tokens),0) AS res_tok + FROM ss_tool_results r + GROUP BY r.tool_name +) +SELECT + COALESCE(p.tool_name, q.tool_name) AS tool_name, + IFNULL(p.calls, 0) AS calls, + IFNULL(q.results, 0) AS results, + IFNULL(p.arg_tok, 0) AS arg_tok, + IFNULL(q.res_tok, 0) AS res_tok +FROM per_tool p FULL OUTER JOIN per_tool_res q USING (tool_name) +ORDER BY (IFNULL(p.arg_tok, 0) + IFNULL(q.res_tok, 0)) DESC +""" + + +def cmd_tools(args: argparse.Namespace) -> int: + conn = open_ro() + where_session, where_args = _session_filter_clause(conn, args) + + def with_session(table_alias: str) -> tuple[str, tuple]: + """Returns ('AND .session_file IN (...)', params) or ('', ()).""" + if not where_args: + return "", () + ph = ",".join("?" * len(where_args)) + return f"AND {table_alias}.session_file IN ({ph})", where_args + + sf_clause_c, sf_params_c = with_session("c") + sf_clause_r, sf_params_r = with_session("r") + sf_clause_a, sf_params_a = with_session("a") + sf_clause_u, sf_params_u = with_session("u") + + # Grand totals (each subquery applies its own session filter). + grand = conn.execute( + f""" + SELECT + (SELECT IFNULL(SUM(c.arg_tokens),0) FROM ss_tool_calls c WHERE 1=1 {sf_clause_c}) AS tool_args, + (SELECT IFNULL(SUM(r.result_tokens),0) FROM ss_tool_results r WHERE 1=1 {sf_clause_r}) AS tool_res, + (SELECT IFNULL(SUM(a.thinking_tokens),0) FROM ss_assistant_msgs a WHERE 1=1 {sf_clause_a}) AS thinking, + (SELECT IFNULL(SUM(a.text_tokens),0) FROM ss_assistant_msgs a WHERE 1=1 {sf_clause_a}) AS asst_text, + (SELECT IFNULL(SUM(u.text_tokens),0) FROM ss_user_msgs u WHERE 1=1 {sf_clause_u}) AS user_text, + (SELECT COUNT(*) FROM ss_tool_calls c WHERE 1=1 {sf_clause_c}) AS n_calls, + (SELECT COUNT(*) FROM ss_tool_results r WHERE 1=1 {sf_clause_r}) AS n_results + """, + sf_params_c + sf_params_r + sf_params_a + sf_params_a + sf_params_u + sf_params_c + sf_params_r, + ).fetchone() + n_sessions = conn.execute( + f"SELECT COUNT(*) FROM ss_sessions {where_session}", where_args + ).fetchone()[0] + g = grand + + grand_total = g["tool_args"] + g["tool_res"] + g["thinking"] + g["asst_text"] + g["user_text"] + print("=== grand totals ===") + print(f"sessions: {commas(n_sessions)}") + print(f"tool calls / results: {commas(g['n_calls'])} / {commas(g['n_results'])}") + print(f"tool ARGS tokens: {commas(g['tool_args']):>14} ({pct(g['tool_args'], grand_total):5.1f}%)") + print(f"tool RESULTS tokens: {commas(g['tool_res']):>14} ({pct(g['tool_res'], grand_total):5.1f}%)") + print(f"assistant THINKING: {commas(g['thinking']):>14} ({pct(g['thinking'], grand_total):5.1f}%)") + print(f"assistant TEXT: {commas(g['asst_text']):>14} ({pct(g['asst_text'], grand_total):5.1f}%)") + print(f"user TEXT: {commas(g['user_text']):>14} ({pct(g['user_text'], grand_total):5.1f}%)") + print(f"total: {commas(grand_total):>14}") + + # Per-tool table. + rows = conn.execute( + f""" + WITH per_tool AS ( + SELECT c.tool_name, COUNT(*) AS calls, IFNULL(SUM(c.arg_tokens),0) AS arg_tok + FROM ss_tool_calls c WHERE 1=1 {sf_clause_c} + GROUP BY c.tool_name + ), + per_tool_res AS ( + SELECT r.tool_name, COUNT(*) AS results, IFNULL(SUM(r.result_tokens),0) AS res_tok + FROM ss_tool_results r WHERE 1=1 {sf_clause_r} + GROUP BY r.tool_name + ) + SELECT + COALESCE(p.tool_name, q.tool_name) AS tool_name, + IFNULL(p.calls,0) AS calls, IFNULL(q.results,0) AS results, + IFNULL(p.arg_tok,0) AS arg_tok, IFNULL(q.res_tok,0) AS res_tok + FROM per_tool p FULL OUTER JOIN per_tool_res q USING (tool_name) + ORDER BY (IFNULL(p.arg_tok,0) + IFNULL(q.res_tok,0)) DESC + """, + sf_params_c + sf_params_r, + ).fetchall() + + print("\n=== per-tool tokens ===") + print(f"{'tool':<24} {'calls':>7} {'args':>14} {'results':>14} {'total':>14}") + print("-" * 78) + for r in rows: + total = r["arg_tok"] + r["res_tok"] + print( + f"{r['tool_name']:<24} {r['calls']:>7} " + f"{commas(r['arg_tok']):>14} {commas(r['res_tok']):>14} {commas(total):>14}" + ) + + if args.by: + bucket = parse_bucket(args.by) + _print_buckets(conn, bucket, args.top, args.tool) + + return 0 + + +def _session_filter_clause(conn, args) -> tuple[str, tuple]: + """Builds an optional WHERE clause for session_file filtering by --limit / --folder.""" + clauses, params = [], [] + if args.folder: + clauses.append("folder LIKE ?") + params.append(f"%{args.folder}%") + if args.limit > 0: + # Resolve to a concrete session_file IN (...) so other tables can reuse it. + rows = conn.execute( + f""" + SELECT session_file FROM ss_sessions + {('WHERE ' + ' AND '.join(clauses)) if clauses else ''} + ORDER BY mtime DESC LIMIT ? + """, + (*params, args.limit), + ).fetchall() + files = [r[0] for r in rows] + if not files: + return ("WHERE 0", ()) + placeholders = ",".join("?" * len(files)) + return (f"WHERE session_file IN ({placeholders})", tuple(files)) + if clauses: + return ("WHERE " + " AND ".join(clauses), tuple(params)) + return ("", ()) + + +def _print_buckets(conn, bucket_secs: int, top: int, tool_filter: str | None) -> None: + where = "WHERE c.tool_name = ?" if tool_filter else "" + params = (tool_filter,) if tool_filter else () + rows = conn.execute( + f""" + SELECT + (c.timestamp / 1000 / ?) * ? AS bucket, + c.tool_name, + COUNT(*) AS calls, + IFNULL(SUM(c.arg_tokens), 0) AS arg_tok, + IFNULL(SUM(r.result_tokens), 0) AS res_tok + FROM ss_tool_calls c + LEFT JOIN ss_tool_results r + ON r.session_file = c.session_file AND r.call_id = c.call_id + {where} + GROUP BY bucket, c.tool_name + ORDER BY bucket DESC + """, + (bucket_secs, bucket_secs) + params, + ).fetchall() + + by_bucket: dict[int, list[sqlite3.Row]] = defaultdict(list) + for r in rows: + by_bucket[r["bucket"]].append(r) + + print(f"\n=== per-tool tokens, bucketed by {bucket_secs}s " + f"({'all tools' if not tool_filter else tool_filter}) ===") + for bucket in sorted(by_bucket.keys(), reverse=True)[:20]: + from datetime import datetime, timezone + label = datetime.fromtimestamp(bucket, tz=timezone.utc).strftime("%Y-%m-%d %H:%MZ") + print(f"\n[{label}]") + ranked = sorted(by_bucket[bucket], key=lambda r: -(r["arg_tok"] + r["res_tok"])) + for r in ranked[:top]: + tot = r["arg_tok"] + r["res_tok"] + print(f" {r['tool_name']:<22} {r['calls']:>5}c " + f"args={commas(r['arg_tok']):>12} res={commas(r['res_tok']):>12} " + f"tot={commas(tot):>12}") + + +# --------------------------------------------------------------------------- # +# `edits` — edit-tool reliability audit (cmd_edits.rs port) + +_RE_TRUNCATED = re.compile(r"\[Output truncated", re.I) +_RE_ABORTED = re.compile( + r"Tool execution was aborted|Request was aborted|cancelled|canceled by user", re.I +) +_RE_SUCCESS = re.compile( + r"^(Updated|Successfully (wrote|replaced|edited|deleted|inserted)|Replaced|" + r"Applied|Deleted|Created|Wrote|edit applied|Edited|Inserted|OK\b)", + re.I, +) +_RE_ANCHOR_STALE = re.compile( + r"(Edit rejected:.*line[s]? .* changed since the last read|" + r"line[s]? ha(s|ve) changed since last read)", + re.I, +) +_RE_ANCHOR_MISSING = re.compile( + r"anchor .* (not found|unknown|missing)|loc requires the full anchor", re.I +) +_RE_NO_ENCLOSING = re.compile(r"No enclosing .* block", re.I) +_RE_PARSE_ERROR = re.compile(r"parse|syntax error|unbalanced|unexpected token", re.I) +_RE_SSR_NO_MATCH = re.compile( + r"0 matches|no replacements|no match found|No replacements made|Failed to find expected lines", + re.I, +) +_RE_FILE_NOT_READ = re.compile(r"must be read first|has not been read|not yet read", re.I) +_RE_FILE_CHANGED = re.compile(r"file has been (modified|changed) externally", re.I) +_RE_PERM_DENIED = re.compile(r"permission denied|not allowed", re.I) +_RE_GENERIC_REJECTED = re.compile(r"\b(rejected|failed|error|invalid)\b", re.I) + + +def classify_edit_result(text: str) -> str: + t = (text or "").strip() + if not t: + return "empty" + first = t.split("\n", 1)[0] + if _RE_TRUNCATED.search(first): + return "truncated" + if _RE_ABORTED.search(t): + return "aborted" + if _RE_SUCCESS.match(first): + return "success" + if _RE_ANCHOR_STALE.search(t): + return "fail:anchor-stale" + if _RE_NO_ENCLOSING.search(t): + return "fail:no-enclosing-block" + if _RE_ANCHOR_MISSING.search(t): + return "fail:anchor-missing" + if _RE_PARSE_ERROR.search(t): + return "fail:parse" + if _RE_SSR_NO_MATCH.search(t): + return "fail:no-match" + if _RE_FILE_NOT_READ.search(t): + return "fail:file-not-read" + if _RE_FILE_CHANGED.search(t): + return "fail:file-changed" + if _RE_PERM_DENIED.search(t): + return "fail:perm" + if _RE_GENERIC_REJECTED.search(first): + return "fail:other" + return "unknown" + + +_ANCHOR_BARE = re.compile(r"^[a-zA-Z]?[0-9]+[a-z]{2}$") + + +def _detect_edit_format(tool_name: str, args_obj: dict | None) -> str: + if tool_name == "write": + return "write" + if tool_name == "ast_edit": + return "ast_edit" + if not isinstance(args_obj, dict): + return "unknown" + has = lambda k: k in args_obj # noqa: E731 + if has("oldText") and has("newText"): + return "oldText/newText" + if has("old_text") and has("new_text"): + return "old_text/new_text" + if has("diff") and has("op"): + return "diff+op" + if has("diff") and has("operation"): + return "diff+operation" + if has("diff"): + return "diff" + if has("replace") or has("insert"): + return "replace/insert" + if has("input") and isinstance(args_obj.get("input"), str): + return "hashline" + edits = args_obj.get("edits") + if isinstance(edits, list) and edits and isinstance(edits[0], dict): + first = edits[0] + fh = lambda k: k in first # noqa: E731 + if fh("loc") and (fh("splice") or fh("pre") or fh("post") or fh("sed")): + return "loc+splice/pre/post/sed" + if fh("loc") and fh("content"): + return "loc+content" + if fh("set_line"): + return "set_line" + if fh("insert_after"): + return "insert_after" + if fh("op") and fh("pos") and fh("end") and fh("lines"): + return "op+pos+end+lines" + if fh("op") and fh("pos") and fh("lines"): + return "op+pos+lines" + if fh("op") and fh("sel") and fh("content"): + return "op+sel+content" + if fh("all") and (fh("new_text") or fh("old_text")): + return "per-edit:old_text/new_text" + return "edits[" + ",".join(sorted(first.keys())) + "]" + return ",".join(sorted(args_obj.keys())) + + +def _loc_shape(loc: str) -> str: + if not loc: + return "empty" + if loc == "$": + return "$file" + if ":" in loc and not loc.startswith("$"): + rest = loc.rsplit(":", 1)[1] + else: + rest = loc + if rest.startswith("(") and rest.endswith(")"): + return "bracket-(body)" + if rest.startswith("[") and rest.endswith("]"): + return "bracket-[block]" + if rest.startswith("(") or rest.startswith("["): + return "bracket-tail" + if rest.endswith(")") or rest.endswith("]"): + return "bracket-head" + if _ANCHOR_BARE.match(rest): + return "bare-anchor" + return "other" + + +def _classify_edit_args(tool_name: str, args_obj: dict | None) -> tuple[str, list[str], list[str]]: + """Returns (format, verbs, loc_shapes).""" + fmt = _detect_edit_format(tool_name, args_obj) + verbs: list[str] = [] + loc_shapes: list[str] = [] + if tool_name == "write": + verbs.append("write") + elif tool_name == "edit" and isinstance(args_obj, dict): + edits = args_obj.get("edits") + if isinstance(edits, list): + for op in edits: + if not isinstance(op, dict): + continue + loc_val = op.get("loc") + loc_shapes.append(_loc_shape(loc_val if isinstance(loc_val, str) else "")) + v: list[str] = [] + if op.get("splice"): + v.append("splice") + if op.get("pre"): + v.append("pre") + if op.get("post"): + v.append("post") + if op.get("sed"): + v.append("sed") + if not v: + v.append("none") + verbs.append("+".join(v)) + return fmt, verbs, loc_shapes + + +def cmd_edits(args: argparse.Namespace) -> int: + conn = open_ro() + where_session, where_args = _session_filter_clause(conn, args) + sf_clause = "AND c.session_file IN (" + ",".join("?" * len(where_args)) + ")" if where_args else "" + + rows = conn.execute( + f""" + SELECT + c.session_file, c.call_id, c.tool_name, c.arg_json, c.timestamp, + r.result_text, r.is_error + FROM ss_tool_calls c + LEFT JOIN ss_tool_results r + ON r.session_file = c.session_file AND r.call_id = c.call_id + WHERE c.tool_name IN ('edit','ast_edit','write') {sf_clause} + ORDER BY c.timestamp + """, + where_args, + ).fetchall() + + if not rows: + print("no edit-family tool calls found") + return 0 + + by_tool: Counter = Counter() + by_format: Counter = Counter() + status_by_tool: dict[str, Counter] = defaultdict(Counter) + status_by_format: dict[str, Counter] = defaultdict(Counter) + verb_count: Counter = Counter() + loc_count: Counter = Counter() + fails_by_verb: dict[str, Counter] = defaultdict(Counter) + fails_by_loc: dict[str, Counter] = defaultdict(Counter) + sessions = set() + failed_samples: list[tuple[str, str, list[str], list[str], str]] = [] + + for r in rows: + sessions.add(r["session_file"]) + tool = r["tool_name"] + try: + args_obj = json.loads(r["arg_json"]) if r["arg_json"] else None + except Exception: + args_obj = None + fmt, verbs, locs = _classify_edit_args(tool, args_obj) + status = classify_edit_result(r["result_text"] or "") + + by_tool[tool] += 1 + by_format[fmt] += 1 + status_by_tool[tool][status] += 1 + status_by_format[fmt][status] += 1 + for v in verbs: + verb_count[v] += 1 + fails_by_verb[v][status] += 1 + for l in locs: + loc_count[l] += 1 + fails_by_loc[l][status] += 1 + + if status.startswith("fail") and len(failed_samples) < 8: + text = r["result_text"] or "" + first = text.split("\n\n", 1)[0] + failed_samples.append((tool, status, verbs, locs, truncate_line(first, 220))) + + print("# Edit-tool usage") + print(f"\nTotal tool calls: {len(rows)} (across {len(sessions)} sessions)") + + _print_counter("\n## By tool", by_tool) + print("\n## Outcome by tool") + for tool in sorted(by_tool): + print(f"\n {tool} ({by_tool[tool]} calls):") + for st, n in sorted(status_by_tool[tool].items(), key=lambda kv: -kv[1]): + print(f" {st:<28} {n}") + + _print_counter("\n## edit verb distribution (per sub-edit)", verb_count) + _print_counter("\n## edit locator shape distribution", loc_count) + + print("\n## Failure rate per verb shape") + for v, _ in verb_count.most_common(): + total, failed = _fail_totals(fails_by_verb[v]) + print(f" {v:<20} {failed}/{total} failed ({pct(failed, total):.0f}%)") + + print("\n## Failure rate per locator shape") + for l, _ in loc_count.most_common(): + total, failed = _fail_totals(fails_by_loc[l]) + print(f" {l:<20} {failed}/{total} failed ({pct(failed, total):.0f}%)") + + _print_counter("\n## edit-tool argument-format usage", by_format) + + print("\n## Failure rate per argument format") + for f, _ in by_format.most_common(): + total, failed = _fail_totals(status_by_format[f]) + print(f" {f:<32} {failed:>6}/{total:<6} failed ({pct(failed, total):.0f}%)") + + print("\n## Failure breakdown per top format") + for f, _ in by_format.most_common(8): + print(f"\n {f} ({by_format[f]} total)") + for st, n in sorted(status_by_format[f].items(), key=lambda kv: -kv[1]): + print(f" {st:<28} {n}") + + print("\n## Sample failed edits") + for tool, status, verbs, locs, snippet in failed_samples: + print(f"\n— {tool} [{status}] verbs={verbs} loc={locs}\n result: {snippet}") + + return 0 + + +def _print_counter(header: str, c: Counter) -> None: + print(header) + for k, v in c.most_common(): + print(f" {k:<32} {v}") + + +def _fail_totals(c: Counter) -> tuple[int, int]: + total = sum(c.values()) + failed = sum(v for k, v in c.items() if k.startswith("fail")) + return total, failed + + +# --------------------------------------------------------------------------- # +# `followups` — five hashline-edit detectors (cmd_followups.rs port) + +_CLOSER_LINE_RE = re.compile(r"^\s*[\])}]+[;,]?\s*$") + +_FIX_PATTERNS = [ + "remove-single-closer", + "add-single-closer", + "one-line-modify", + "pure-delete-1", + "pure-insert-1", + "small-other", +] +_FIX_PRIORITY = {p: i for i, p in enumerate(_FIX_PATTERNS)} + + +def _classify_fix(deleted: int, payload_lines: list[str]) -> str: + closer = len(payload_lines) == 1 and bool(_CLOSER_LINE_RE.match(payload_lines[0])) + if deleted == 0 and closer: + return "add-single-closer" + if deleted == 1 and not payload_lines: + return "remove-single-closer" + if deleted == 0 and len(payload_lines) == 1: + return "pure-insert-1" + if deleted == 1 and len(payload_lines) == 1: + return "one-line-modify" + if deleted >= 1 and not payload_lines: + return "pure-delete-1" + return "small-other" + + +def cmd_followups(args: argparse.Namespace) -> int: + conn = open_ro() + where_session, where_args = _session_filter_clause(conn, args) + sf_clause = "AND c.session_file IN (" + ",".join("?" * len(where_args)) + ")" if where_args else "" + + # All edit calls + their sections, ordered per session. + call_rows = conn.execute( + f""" + SELECT c.session_file, c.call_id, c.seq, c.timestamp, c.raw_input_len, + c.success, c.warnings + FROM ss_edit_calls c + WHERE 1=1 {sf_clause} + ORDER BY c.session_file, c.seq + """, + where_args, + ).fetchall() + + section_rows = conn.execute( + f""" + SELECT s.session_file, s.call_id, s.seq, s.section_idx, s.target_file, + s.op_count, s.deleted_lines, s.payload_count, s.change_size, + s.min_line, s.max_line, s.payload_blocks, + s.longest_repeat_len, s.longest_repeat_block_idx, + s.longest_repeat_sample, s.dup_anchors + FROM ss_edit_sections s + JOIN ss_edit_calls c USING (session_file, call_id) + WHERE 1=1 {sf_clause.replace('c.session_file', 's.session_file')} + ORDER BY s.session_file, s.seq, s.section_idx + """, + where_args, + ).fetchall() + + # Index sections by (session_file, call_id). + sec_by_call: dict[tuple[str, str], list[sqlite3.Row]] = defaultdict(list) + for s in section_rows: + sec_by_call[(s["session_file"], s["call_id"])].append(s) + + # Build per-(session, target_file) ordered list of (call_meta, section). + by_session_file: dict[tuple[str, str], list[tuple[sqlite3.Row, sqlite3.Row]]] = defaultdict(list) + total_successful_edits = 0 + warning_hits: list[dict] = [] + payload_dups: list[dict] = [] + anchor_dups: list[dict] = [] + + for c in call_rows: + if c["success"] == 1: + total_successful_edits += 1 + warns = json.loads(c["warnings"] or "[]") + seen: list[str] = [] + for w in warns: + if w not in seen: + seen.append(w) + if seen: + files_csv = ",".join( + s["target_file"] for s in sec_by_call.get((c["session_file"], c["call_id"]), []) + ) + for kind in seen: + warning_hits.append({ + "session": c["session_file"], + "call_id": c["call_id"], + "kind": kind, + "files": files_csv, + "input_len": c["raw_input_len"], + }) + if c["success"] != 1: + continue + for s in sec_by_call.get((c["session_file"], c["call_id"]), []): + by_session_file[(c["session_file"], s["target_file"])].append((c, s)) + # Payload self-dup + if s["longest_repeat_len"] >= 4: + payload_dups.append({ + "session": c["session_file"], + "call_id": c["call_id"], + "file": s["target_file"], + "block_len": _block_len(s, s["longest_repeat_block_idx"]), + "repeat_len": s["longest_repeat_len"], + "sample": s["longest_repeat_sample"] or "", + }) + # Anchor reuse + try: + dups = json.loads(s["dup_anchors"] or "[]") + except Exception: + dups = [] + for d in dups: + anchor_dups.append({ + "session": c["session_file"], + "call_id": c["call_id"], + "files": d[2] if len(d) > 2 else s["target_file"], + "anchor": d[0], + "count": d[1], + }) + + # (1) small-fix follow-ups + (3) same-locus re-edits. + fix_hits: list[dict] = [] + locus_hits: list[dict] = [] + for (session, target), entries in by_session_file.items(): + for i in range(len(entries) - 1): + ac, asec = entries[i] + bc, bsec = entries[i + 1] + if ac["call_id"] == bc["call_id"]: + continue + first_size = asec["change_size"] + second_size = bsec["change_size"] + gap = max(0, (bc["timestamp"] - ac["timestamp"]) // 1000) + + # (1) small fix on big edit + if 0 < second_size <= args.max_fix and first_size > 2: + pl = _flatten_payload(bsec) + pattern = _classify_fix(bsec["deleted_lines"], pl) + summary = _render_section_summary(bsec, pl) + fix_hits.append({ + "session": session, "file": target, + "first_call_id": ac["call_id"], "second_call_id": bc["call_id"], + "first_size": first_size, "first_input_len": ac["raw_input_len"], + "second_size": second_size, + "pattern": pattern, "second_summary": summary, "gap_secs": gap, + }) + + # (3) same-locus re-edit (both > max-fix) + if ( + first_size > 2 and second_size > args.max_fix + and asec["min_line"] is not None and asec["max_line"] is not None + and bsec["min_line"] is not None and bsec["max_line"] is not None + ): + a_lo, a_hi = asec["min_line"], asec["max_line"] + b_lo, b_hi = bsec["min_line"], bsec["max_line"] + if max(a_lo, b_lo) <= min(a_hi, b_hi): + locus_hits.append({ + "session": session, "file": target, + "first_call_id": ac["call_id"], "second_call_id": bc["call_id"], + "first_range": (a_lo, a_hi), "second_range": (b_lo, b_hi), + "first_size": first_size, "second_size": second_size, "gap_secs": gap, + }) + + if args.max_gap > 0: + fix_hits = [h for h in fix_hits if h["gap_secs"] <= args.max_gap] + locus_hits = [h for h in locus_hits if h["gap_secs"] <= args.max_gap] + if args.pattern: + fix_hits = [h for h in fix_hits if h["pattern"] == args.pattern] + + fix_hits.sort(key=lambda h: (_FIX_PRIORITY.get(h["pattern"], 99), -h["first_input_len"])) + locus_hits.sort(key=lambda h: -h["first_size"]) + payload_dups = [p for p in payload_dups if p["repeat_len"] >= args.min_dup] + payload_dups.sort(key=lambda p: -p["repeat_len"]) + anchor_dups.sort(key=lambda a: -a["count"]) + + # ---- print ---- + by_pattern = Counter(h["pattern"] for h in fix_hits) + + print("=== heuristic followup hits ===") + print(f"total hits: {commas(len(fix_hits))}") + print("\nby pattern:") + for label, n in by_pattern.most_common(): + print(f" {label:<22} {n:>6}") + + shown = min(args.show, len(fix_hits)) + print(f"\n=== top {shown} hits (by first-edit input size) ===") + for h in fix_hits[:shown]: + print( + f"[{h['pattern']}] {h['file']} first={h['first_size']}L " + f"({h['first_input_len']}B) → second={h['second_size']}L gap={h['gap_secs']}s" + ) + print(f" session={h['session']}") + print(f" first_call={h['first_call_id']} second_call={h['second_call_id']}") + print(f" fix: {h['second_summary']}") + + print("\n=== tool self-corrections ===") + print("(emitted as warnings on otherwise-successful edits — the tool caught what the model wrote)") + by_kind = Counter(w["kind"] for w in warning_hits) + for kind, n in by_kind.most_common(): + suffix = (f" ({pct(n, total_successful_edits):.2f}% of " + f"{commas(total_successful_edits)} successful edits)") if total_successful_edits else "" + print(f" {kind:<16} {n:>6}{suffix}") + + warn_show = min(args.show, len(warning_hits)) + if warn_show: + print(f"\n--- top {warn_show} self-correction examples (by input size) ---") + sorted_warns = sorted(warning_hits, key=lambda w: -w["input_len"]) + for w in sorted_warns[:warn_show]: + print(f"[{w['kind']}] {truncate_line(w['files'], 80)} ({w['input_len']}B)") + print(f" session={w['session']} call={w['call_id']}") + + print("\n=== same-locus re-edits (overlapping anchor ranges, both > max-fix) ===") + print(f"hits: {commas(len(locus_hits))}") + locus_show = min(args.show, len(locus_hits)) + for h in locus_hits[:locus_show]: + print( + f"{h['file']} first={h['first_range'][0]}..{h['first_range'][1]} " + f"({h['first_size']}L) → second={h['second_range'][0]}..{h['second_range'][1]} " + f"({h['second_size']}L) gap={h['gap_secs']}s" + ) + print(f" session={h['session']}") + print(f" first_call={h['first_call_id']} second_call={h['second_call_id']}") + + print("\n=== payload self-duplication (model pasted same N-line chunk twice in one payload) ===") + print( + f"hits with repeat_len >= {args.min_dup}: {commas(len(payload_dups))} " + f"({pct(len(payload_dups), total_successful_edits):.2f}% of " + f"{commas(total_successful_edits)} successful edits)" + ) + for p in payload_dups[: args.show]: + print(f"k={p['repeat_len']} block={p['block_len']}L {p['file']}") + print(f" session={p['session']} call={p['call_id']}") + print(f" sample: {p['sample']}") + + print("\n=== same-anchor reused by multiple ops in one input ===") + print(f"hits: {commas(len(anchor_dups))}") + for a in anchor_dups[: args.show]: + print( + f"anchor {a['anchor']} referenced {a['count']}x " + f"files={truncate_line(a['files'], 80)}" + ) + print(f" session={a['session']} call={a['call_id']}") + + return 0 + + +def _flatten_payload(section_row: sqlite3.Row) -> list[str]: + try: + blocks = json.loads(section_row["payload_blocks"] or "[]") + except Exception: + return [] + out: list[str] = [] + for b in blocks: + out.extend(b) + return out + + +def _block_len(section_row: sqlite3.Row, idx: int | None) -> int: + if idx is None: + return 0 + try: + blocks = json.loads(section_row["payload_blocks"] or "[]") + except Exception: + return 0 + if 0 <= idx < len(blocks): + return len(blocks[idx]) + return 0 + + +def _render_section_summary(section_row: sqlite3.Row, payload_lines: list[str]) -> str: + bits: list[str] = [] + deleted = section_row["deleted_lines"] + if deleted > 0: + bits.append(f"-{deleted}") + if payload_lines: + bits.append(f"+{len(payload_lines)}") + out = " / ".join(bits) + if payload_lines: + out += f" | {truncate_line(payload_lines[0], 80)}" + return out + + +# --------------------------------------------------------------------------- # +# Entry point + +def main() -> int: + ap = argparse.ArgumentParser(description="session-stats analyses (sqlite-backed)") + sub = ap.add_subparsers(dest="cmd", required=True) + + common = argparse.ArgumentParser(add_help=False) + common.add_argument("-n", "--limit", type=int, default=0, + help="restrict to N most-recent sessions (0 = all)") + common.add_argument("--folder", default=None, + help="filter sessions whose folder contains this substring") + + ap_tools = sub.add_parser("tools", parents=[common], help="per-tool token totals") + ap_tools.add_argument("--by", default=None, + help="bucket per-call data: h, d, w, m, or {h,d,w}") + ap_tools.add_argument("--top", type=int, default=10, help="top tools per bucket") + ap_tools.add_argument("--tool", default=None, help="restrict bucket view to one tool") + ap_tools.set_defaults(func=cmd_tools) + + ap_edits = sub.add_parser("edits", parents=[common], help="edit reliability audit") + ap_edits.set_defaults(func=cmd_edits) + + ap_fu = sub.add_parser("followups", parents=[common], help="hashline edit followup detectors") + ap_fu.add_argument("--max-fix", type=int, default=2) + ap_fu.add_argument("--max-gap", type=int, default=0, help="cap seconds between paired edits") + ap_fu.add_argument("--min-dup", type=int, default=8, help="min payload-dup repeat length") + ap_fu.add_argument("--pattern", default=None, help="filter (1) to a single FixPattern") + ap_fu.add_argument("--show", type=int, default=60) + ap_fu.set_defaults(func=cmd_followups) + + args = ap.parse_args() + return args.func(args) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/session-stats/src/cmd_edits.rs b/scripts/session-stats/src/cmd_edits.rs deleted file mode 100644 index 3f7aa51c2..000000000 --- a/scripts/session-stats/src/cmd_edits.rs +++ /dev/null @@ -1,639 +0,0 @@ -//! `edits` subcommand — audits how agents have used the edit / ast_edit / -//! write tools across session jsonl files. -//! -//! For every edit-family toolCall we record: -//! - which argument-schema family is in use (the edit tool has shipped many -//! shapes over time: oldText/newText, op+pos+end+lines, loc+content, -//! loc+splice/pre/post/sed, etc.); -//! - the locator shape and verb combination (for the current -//! loc+splice/pre/post/sed schema); -//! then pair the call with its toolResult and classify success / failure -//! category (anchor-stale, no-match, parse, etc.). -//! -//! Output: markdown-ish report on stdout plus a per-call CSV at -//! `$EDIT_ANALYSIS_CSV` (default `./edit-analysis.csv`). - -use crate::common::*; -use anyhow::{Context, Result, bail}; -use regex::Regex; -use serde::Deserialize; -use serde_json::Value; -use serde_json::value::RawValue; -use std::collections::HashMap; -use std::fs::File; -use std::io::{BufRead, BufReader}; -use std::path::Path; -use std::sync::LazyLock; - -#[derive(Default, Clone)] -struct EditEntry { - file: String, - call_id: String, - tool_name: String, - num_edits: i64, - /// splice/pre/post/sed per sub-edit - verbs: Vec, - /// bare-anchor / bracket-(body) / ... - loc_shapes: Vec, - /// edit-tool argument schema family - format: String, - result_raw: String, - /// "success" / "fail:..." / etc. - status: String, -} - -#[derive(Deserialize, Default)] -struct EditOp { - #[serde(default)] - loc: String, - #[serde(default)] - splice: Option>, - #[serde(default)] - pre: Option>, - #[serde(default)] - post: Option>, - #[serde(default)] - sed: Option>, -} - -#[derive(Deserialize, Default)] -struct EditArgs { - #[serde(default)] - edits: Vec, - #[serde(default)] - ops: Vec>, -} - -pub fn run(args: Vec) -> Result<()> { - let mut limit: usize = 1_000; - let mut workers: usize = 0; - let mut date_filters: Vec = Vec::new(); - - let mut iter = args.into_iter(); - while let Some(a) = iter.next() { - match a.as_str() { - "-n" => { - limit = iter - .next() - .context("-n requires a value")? - .parse() - .context("-n value")?; - } - "-j" => { - workers = iter - .next() - .context("-j requires a value")? - .parse() - .context("-j value")?; - } - "-h" | "--help" => { - eprintln!( - "usage: session-stats edits [-n N] [-j workers] [date prefix ...]" - ); - return Ok(()); - } - other if other.starts_with('-') => bail!("unknown flag: {other}"), - other => date_filters.push(other.to_string()), - } - } - - let files = collect_sessions(&WalkOpts { - date_filters, - limit_most_recent: limit, - })?; - eprintln!("scanning {} session files", files.len()); - - let mut entries: Vec = parallel_collect(&files, workers, 5_000, |p| { - Some(process_file(p)) - }) - .into_iter() - .flatten() - .collect(); - - // Stable ordering for sample output. - entries.sort_by(|a, b| a.file.cmp(&b.file)); - - report_edits(&entries); - write_csv(&entries)?; - Ok(()) -} - -fn process_file(path: &Path) -> Vec { - let f = match File::open(path) { - Ok(f) => f, - Err(e) => { - eprintln!("open {}: {e}", path.display()); - return Vec::new(); - } - }; - let reader = BufReader::with_capacity(64 * 1024, f); - let path_str = path.to_string_lossy().into_owned(); - - let mut calls: HashMap = HashMap::new(); - let mut order: Vec = Vec::new(); - - for line in reader.lines() { - let Ok(line) = line else { continue }; - if line.is_empty() { - continue; - } - let Ok(ev) = serde_json::from_str::(&line) else { - continue; - }; - if ev.kind != "message" { - continue; - } - let Some(msg_raw) = ev.message else { continue }; - let Ok(m) = serde_json::from_str::(msg_raw.get()) else { - continue; - }; - let Some(content_raw) = m.content else { continue }; - let items = parse_content(&content_raw); - - match m.role.as_str() { - "assistant" => { - for it in items { - if it.kind != "toolCall" || !is_edit_tool(&it.name) { - continue; - } - let raw = it.arguments.as_deref(); - let mut e = classify_edit_args(&it.name, raw); - e.file.clone_from(&path_str); - e.call_id.clone_from(&it.id); - e.tool_name.clone_from(&it.name); - let id = it.id.clone(); - calls.insert(id.clone(), e); - order.push(id); - } - } - "toolResult" => { - if !is_edit_tool(&m.tool_name) { - continue; - } - let Some(e) = calls.get_mut(&m.tool_call_id) else { - continue; - }; - let text = join_text(&items); - e.status = classify_edit_result(&text); - e.result_raw = text; - } - _ => {} - } - } - - let mut out: Vec = Vec::with_capacity(order.len()); - for id in order { - if let Some(e) = calls.remove(&id) { - out.push(e); - } - } - out -} - -fn is_edit_tool(name: &str) -> bool { - matches!( - name.to_ascii_lowercase().as_str(), - "edit" | "ast_edit" | "write" - ) -} - -// ---- argument classification ---- - -static ANCHOR_BARE: LazyLock = - LazyLock::new(|| Regex::new(r"^[a-zA-Z]?[0-9]+[a-z]{2}$").expect("anchor_bare")); - -fn classify_edit_args(name: &str, raw: Option<&RawValue>) -> EditEntry { - let mut e = EditEntry { - format: detect_edit_format(name, raw), - ..EditEntry::default() - }; - let lname = name.to_ascii_lowercase(); - match lname.as_str() { - "edit" => { - let a: EditArgs = raw - .and_then(|r| serde_json::from_str(r.get()).ok()) - .unwrap_or_default(); - e.num_edits = a.edits.len() as i64; - for op in &a.edits { - e.loc_shapes.push(loc_shape(&op.loc)); - let mut verbs: Vec<&str> = Vec::new(); - if !is_null_or_empty(op.splice.as_deref()) { - verbs.push("splice"); - } - if !is_null_or_empty(op.pre.as_deref()) { - verbs.push("pre"); - } - if !is_null_or_empty(op.post.as_deref()) { - verbs.push("post"); - } - if !is_null_or_empty(op.sed.as_deref()) { - verbs.push("sed"); - } - if verbs.is_empty() { - verbs.push("none"); - } - e.verbs.push(verbs.join("+")); - } - } - "ast_edit" => { - let a: EditArgs = raw - .and_then(|r| serde_json::from_str(r.get()).ok()) - .unwrap_or_default(); - e.num_edits = a.ops.len() as i64; - } - "write" => { - e.num_edits = 1; - e.verbs.push("write".to_string()); - } - _ => {} - } - e -} - -/// Looks at top-level argument keys (and the first sub-edit for the `edit` -/// tool) to identify which schema is in use. Older sessions used many -/// incompatible schemas. -fn detect_edit_format(name: &str, raw: Option<&RawValue>) -> String { - match name.to_ascii_lowercase().as_str() { - "write" => return "write".to_string(), - "ast_edit" => return "ast_edit".to_string(), - _ => {} - } - let Some(raw) = raw else { - return "unknown".to_string(); - }; - let top: HashMap = match serde_json::from_str(raw.get()) { - Ok(v) => v, - Err(_) => return "unknown".to_string(), - }; - let has = |k: &str| top.contains_key(k); - - if has("oldText") && has("newText") { - return "oldText/newText".to_string(); - } - if has("old_text") && has("new_text") { - return "old_text/new_text".to_string(); - } - if has("diff") && has("op") { - return "diff+op".to_string(); - } - if has("diff") && has("operation") { - return "diff+operation".to_string(); - } - if has("diff") { - return "diff".to_string(); - } - if has("replace") || has("insert") { - return "replace/insert".to_string(); - } - - if let Some(edits_val) = top.get("edits") - && let Some(arr) = edits_val.as_array() - && let Some(first) = arr.first().and_then(Value::as_object) - { - let fh = |k: &str| first.contains_key(k); - if fh("loc") && (fh("splice") || fh("pre") || fh("post") || fh("sed")) { - return "loc+splice/pre/post/sed".to_string(); - } - if fh("loc") && fh("content") { - return "loc+content".to_string(); - } - if fh("set_line") { - return "set_line".to_string(); - } - if fh("insert_after") { - return "insert_after".to_string(); - } - if fh("op") && fh("pos") && fh("end") && fh("lines") { - return "op+pos+end+lines".to_string(); - } - if fh("op") && fh("pos") && fh("lines") { - return "op+pos+lines".to_string(); - } - if fh("op") && fh("sel") && fh("content") { - return "op+sel+content".to_string(); - } - if fh("all") && (fh("new_text") || fh("old_text")) { - return "per-edit:old_text/new_text".to_string(); - } - let mut keys: Vec<&str> = first.keys().map(String::as_str).collect(); - keys.sort_unstable(); - return format!("edits[{}]", keys.join(",")); - } - - let mut keys: Vec<&str> = top.keys().map(String::as_str).collect(); - keys.sort_unstable(); - keys.join(",") -} - -fn is_null_or_empty(b: Option<&RawValue>) -> bool { - let Some(b) = b else { return true }; - let s = b.get().trim(); - s.is_empty() || s == "null" -} - -fn loc_shape(loc: &str) -> String { - if loc.is_empty() { - return "empty".to_string(); - } - if loc == "$" { - return "$file".to_string(); - } - let rest = if let Some(i) = loc.rfind(':') - && !loc.starts_with('$') - { - &loc[i + 1..] - } else { - loc - }; - if rest.starts_with('(') && rest.ends_with(')') { - return "bracket-(body)".to_string(); - } - if rest.starts_with('[') && rest.ends_with(']') { - return "bracket-[block]".to_string(); - } - if rest.starts_with('(') || rest.starts_with('[') { - return "bracket-tail".to_string(); - } - if rest.ends_with(')') || rest.ends_with(']') { - return "bracket-head".to_string(); - } - if ANCHOR_BARE.is_match(rest) { - return "bare-anchor".to_string(); - } - "other".to_string() -} - -// ---- result classification ---- - -macro_rules! re { - ($pat:expr) => { - LazyLock::new(|| Regex::new($pat).expect("compile result regex")) - }; -} - -static RE_ANCHOR_STALE: LazyLock = re!( - r"(?i)(Edit rejected:.*line[s]? .* changed since the last read|line[s]? ha(s|ve) changed since last read)" -); -static RE_ANCHOR_MISSING: LazyLock = - re!(r"(?i)anchor .* (not found|unknown|missing)|loc requires the full anchor"); -static RE_NO_ENCLOSING: LazyLock = re!(r"(?i)No enclosing .* block"); -static RE_PARSE_ERROR: LazyLock = - re!(r"(?i)parse|syntax error|unbalanced|unexpected token"); -static RE_SSR_NO_MATCH: LazyLock = re!( - r"(?i)0 matches|no replacements|no match found|No replacements made|Failed to find expected lines" -); -static RE_FILE_NOT_READ: LazyLock = - re!(r"(?i)must be read first|has not been read|not yet read"); -static RE_FILE_CHANGED: LazyLock = - re!(r"(?i)file has been (modified|changed) externally"); -static RE_PERM_DENIED: LazyLock = re!(r"(?i)permission denied|not allowed"); -static RE_GENERIC_REJECTED: LazyLock = - re!(r"(?i)\b(rejected|failed|error|invalid)\b"); -static RE_TRUNCATED: LazyLock = re!(r"(?i)\[Output truncated"); -static RE_ABORTED: LazyLock = re!( - r"(?i)Tool execution was aborted|Request was aborted|cancelled|canceled by user" -); -static RE_SUCCESS: LazyLock = re!( - r"(?i)^(Updated|Successfully (wrote|replaced|edited|deleted|inserted)|Replaced|Applied|Deleted|Created|Wrote|edit applied|Edited|Inserted|OK\b)" -); - -fn classify_edit_result(text: &str) -> String { - let t = text.trim(); - if t.is_empty() { - return "empty".to_string(); - } - let first = t.split_once('\n').map_or(t, |(a, _)| a); - - if RE_TRUNCATED.is_match(first) { - return "truncated".to_string(); - } - if RE_ABORTED.is_match(t) { - return "aborted".to_string(); - } - if RE_SUCCESS.is_match(first) { - return "success".to_string(); - } - if RE_ANCHOR_STALE.is_match(t) { - return "fail:anchor-stale".to_string(); - } - if RE_NO_ENCLOSING.is_match(t) { - return "fail:no-enclosing-block".to_string(); - } - if RE_ANCHOR_MISSING.is_match(t) { - return "fail:anchor-missing".to_string(); - } - if RE_PARSE_ERROR.is_match(t) { - return "fail:parse".to_string(); - } - if RE_SSR_NO_MATCH.is_match(t) { - return "fail:no-match".to_string(); - } - if RE_FILE_NOT_READ.is_match(t) { - return "fail:file-not-read".to_string(); - } - if RE_FILE_CHANGED.is_match(t) { - return "fail:file-changed".to_string(); - } - if RE_PERM_DENIED.is_match(t) { - return "fail:perm".to_string(); - } - if RE_GENERIC_REJECTED.is_match(first) { - return "fail:other".to_string(); - } - "unknown".to_string() -} - -// ---- reporting ---- - -fn report_edits(entries: &[EditEntry]) { - if entries.is_empty() { - println!("no edit-family tool calls found in matched sessions"); - return; - } - - let mut by_tool: HashMap = HashMap::new(); - let mut by_format: HashMap = HashMap::new(); - let mut status_by_format: HashMap> = HashMap::new(); - let mut status_by_tool: HashMap> = HashMap::new(); - let mut verb_count: HashMap = HashMap::new(); - let mut loc_count: HashMap = HashMap::new(); - let mut fails_by_verb: HashMap> = HashMap::new(); - let mut fails_by_loc: HashMap> = HashMap::new(); - - for e in entries { - *by_tool.entry(e.tool_name.clone()).or_insert(0) += 1; - *status_by_tool - .entry(e.tool_name.clone()) - .or_default() - .entry(e.status.clone()) - .or_insert(0) += 1; - *by_format.entry(e.format.clone()).or_insert(0) += 1; - *status_by_format - .entry(e.format.clone()) - .or_default() - .entry(e.status.clone()) - .or_insert(0) += 1; - for v in &e.verbs { - *verb_count.entry(v.clone()).or_insert(0) += 1; - *fails_by_verb - .entry(v.clone()) - .or_default() - .entry(e.status.clone()) - .or_insert(0) += 1; - } - for l in &e.loc_shapes { - *loc_count.entry(l.clone()).or_insert(0) += 1; - *fails_by_loc - .entry(l.clone()) - .or_default() - .entry(e.status.clone()) - .or_insert(0) += 1; - } - } - - println!("# Edit-tool usage"); - println!( - "\nTotal tool calls: {} (across {} sessions)", - entries.len(), - count_edit_sessions(entries) - ); - - println!("\n## By tool"); - print_sorted(&by_tool); - - println!("\n## Outcome by tool"); - let mut tools: Vec<&String> = by_tool.keys().collect(); - tools.sort(); - for t in tools { - println!("\n {t} ({} calls):", by_tool[t.as_str()]); - if let Some(m) = status_by_tool.get(t.as_str()) { - print_sorted_indent(m, " "); - } - } - - println!("\n## edit verb distribution (per sub-edit)"); - print_sorted(&verb_count); - - println!("\n## edit locator shape distribution"); - print_sorted(&loc_count); - - println!("\n## Failure rate per verb shape"); - for v in sorted_by_count(&verb_count) { - let (total, failed) = fail_totals(fails_by_verb.get(v.as_str())); - println!( - " {v:<20} {failed}/{total} failed ({:.0}%)", - pct(failed, total) - ); - } - - println!("\n## Failure rate per locator shape"); - for l in sorted_by_count(&loc_count) { - let (total, failed) = fail_totals(fails_by_loc.get(l.as_str())); - println!( - " {l:<20} {failed}/{total} failed ({:.0}%)", - pct(failed, total) - ); - } - - println!("\n## edit-tool argument-format usage"); - print_sorted(&by_format); - - println!("\n## Failure rate per argument format"); - for fname in sorted_by_count(&by_format) { - let (total, failed) = fail_totals(status_by_format.get(fname.as_str())); - println!( - " {fname:<32} {failed:>6}/{total:<6} failed ({:.0}%)", - pct(failed, total) - ); - } - - println!("\n## Failure breakdown per top format"); - for fname in sorted_by_count(&by_format).into_iter().take(8) { - println!("\n {fname} ({} total)", by_format[fname.as_str()]); - if let Some(m) = status_by_format.get(fname.as_str()) { - print_sorted_indent(m, " "); - } - } - - println!("\n## Sample failed edits"); - let mut shown = 0; - for e in entries { - if !e.status.starts_with("fail") { - continue; - } - let first = e - .result_raw - .split_once("\n\n") - .map_or(e.result_raw.as_str(), |(a, _)| a); - println!( - "\n— {} [{}] verbs={:?} loc={:?}\n result: {}", - e.tool_name, - e.status, - e.verbs, - e.loc_shapes, - truncate_line(first, 220) - ); - shown += 1; - if shown >= 8 { - break; - } - } -} - -fn fail_totals(m: Option<&HashMap>) -> (i64, i64) { - let Some(m) = m else { return (0, 0) }; - let mut total = 0i64; - let mut failed = 0i64; - for (status, n) in m { - total += n; - if status.starts_with("fail") { - failed += n; - } - } - (total, failed) -} - -fn count_edit_sessions(entries: &[EditEntry]) -> usize { - let mut s: std::collections::HashSet<&str> = std::collections::HashSet::new(); - for e in entries { - s.insert(&e.file); - } - s.len() -} - -fn write_csv(entries: &[EditEntry]) -> Result<()> { - let path = std::env::var("EDIT_ANALYSIS_CSV").unwrap_or_else(|_| "edit-analysis.csv".to_string()); - let f = File::create(&path).with_context(|| format!("create {path}"))?; - let mut w = csv::Writer::from_writer(f); - w.write_record([ - "session", - "tool", - "status", - "num_edits", - "verbs", - "loc_shapes", - "result_first_line", - ])?; - for e in entries { - let first = e - .result_raw - .split_once('\n') - .map_or(e.result_raw.as_str(), |(a, _)| a); - let session = Path::new(&e.file) - .file_name() - .and_then(|s| s.to_str()) - .unwrap_or(&e.file); - w.write_record([ - session, - &e.tool_name, - &e.status, - &e.num_edits.to_string(), - &e.verbs.join(","), - &e.loc_shapes.join(","), - &truncate_line(first, 200), - ])?; - } - w.flush()?; - Ok(()) -} diff --git a/scripts/session-stats/src/cmd_followups.rs b/scripts/session-stats/src/cmd_followups.rs deleted file mode 100644 index 64086c852..000000000 --- a/scripts/session-stats/src/cmd_followups.rs +++ /dev/null @@ -1,972 +0,0 @@ -//! `followups` subcommand — heuristic for buggy hashline edits. -//! -//! Premise: when a hashline `edit` introduces a duplicate `}`, drops a line, -//! or otherwise breaks adjacent context, the next thing the agent does is -//! usually a tiny corrective edit on the same file. So we flag, per session, -//! pairs of edits where: -//! -//! - both are hashline `edit` toolCalls (input begins with a `@PATH`), -//! - both target the same file, -//! - the FIRST edit succeeded, -//! - the SECOND edit is small (<= --max-fix lines changed, default 2), -//! - the second edit is the next edit (in the same session) on that file, -//! and there is no other edit in between (so retries-after-failure are -//! not counted). -//! -//! For each hit we classify the small follow-up's pattern: adds a single -//! closing brace/paren/bracket, removes one (likely duplicate), pure single -//! insert, pure single delete, or one-line tweak. - -use crate::common::*; -use anyhow::{Context, Result, bail}; -use regex::Regex; -use serde::Deserialize; -use std::collections::HashMap; -use std::fs::File; -use std::io::{BufRead, BufReader}; -use std::path::Path; -use std::sync::LazyLock; - -// ---- parsed shapes -------------------------------------------------------- - -#[derive(Clone, Default)] -struct EditSection { - file: String, - payload_lines: Vec, - /// Payload lines grouped by op. Each inner vec is one op's payload (`+`, - /// `<`, or `=` followed by `~TEXT` lines), in source order. Used to detect - /// intra-block self-similarity (model duplicating an N-line chunk). - payload_blocks: Vec>, - /// Raw anchor refs (`123ab`) from ops in this section, in source order. - /// Used to detect the same anchor being referenced by multiple ops in the - /// same call. - op_anchors: Vec, - /// Total line count covered by `- A..B` and `= A..B` ranges (range size). - deleted_lines: i64, - op_count: i64, - /// Min/max anchor line touched by ops in this section. `None` for a - /// section whose only ops target BOF/EOF (no concrete line). - min_line: Option, - max_line: Option, -} - -impl EditSection { - fn change_size(&self) -> i64 { - self.payload_lines.len() as i64 + self.deleted_lines - } - fn touch(&mut self, line: i64) { - self.min_line = Some(self.min_line.map_or(line, |m| m.min(line))); - self.max_line = Some(self.max_line.map_or(line, |m| m.max(line))); - } -} - -#[derive(Clone)] -struct EditCall { - ts: i64, - call_id: String, - sections: Vec, - success: bool, - raw_input_len: usize, - /// Tool-emitted warnings extracted from a successful result text. Each - /// entry is one of `auto-rebased` / `auto-absorbed` / `auto-dropped`. - warnings: Vec<&'static str>, -} - -#[derive(Deserialize, Default)] -struct EditArgs { - #[serde(default)] - input: Option, -} - -// ---- per-section parsing -------------------------------------------------- - -// Op headers. We parse loosely: anything that doesn't match a known op or a -// `~` payload line is ignored (blank line, comment, etc.). -// -// Range sizes: `LINEhash..LINEhash` -> end_line - start_line + 1 (clamped to -// >= 1). For single-anchor `- A` we count 1. -static RANGE_RE: LazyLock = - LazyLock::new(|| Regex::new(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$").expect("RANGE_RE")); -static SINGLE_ANCHOR_RE: LazyLock = - LazyLock::new(|| Regex::new(r"^\s*(\d+)[a-z*]+\s*$").expect("SINGLE_ANCHOR_RE")); - -/// Returns (range_size, optional (start_line, end_line)). Size is at least 1. -fn parse_range(raw: &str) -> (i64, Option<(i64, i64)>) { - let Some(caps) = RANGE_RE.captures(raw.trim()) else { - return (1, None); - }; - let start: i64 = caps.get(1).and_then(|m| m.as_str().parse().ok()).unwrap_or(0); - let end: i64 = caps - .get(2) - .and_then(|m| m.as_str().parse().ok()) - .unwrap_or(start); - let size = (end - start + 1).max(1); - let lines = if start > 0 { Some((start, end.max(start))) } else { None }; - (size, lines) -} - -fn parse_anchor_line(raw: &str) -> Option { - let caps = SINGLE_ANCHOR_RE.captures(raw.trim())?; - caps.get(1)?.as_str().parse().ok() -} - -/// Parses a single hashline `input` arg into per-file sections. Returns an -/// empty vec if the input doesn't begin with `@PATH` (vim-mode edits etc.). -fn parse_hashline_input(input: &str) -> Vec { - let mut sections: Vec = Vec::new(); - let mut cur: Option = None; - // Index of the currently-open payload block in cur.payload_blocks, or - // `None` if no op is awaiting payload. - let mut open_block: Option = None; - - fn open_new_block(s: &mut EditSection, open_block: &mut Option) { - s.payload_blocks.push(Vec::new()); - *open_block = Some(s.payload_blocks.len() - 1); - } - fn close_block(open_block: &mut Option) { - *open_block = None; - } - - for raw_line in input.split('\n') { - let line = raw_line.strip_suffix('\r').unwrap_or(raw_line); - - // File header: `@`. Always starts a new section. - if let Some(rest) = line.strip_prefix('@') { - if let Some(s) = cur.take() { - sections.push(s); - } - cur = Some(EditSection { file: rest.trim().to_string(), ..Default::default() }); - open_block = None; - continue; - } - - let Some(s) = cur.as_mut() else { - // Stray content before any `@PATH` — not a hashline input. - continue; - }; - - // Payload lines (`~...`) belong to whatever op was last opened. - if let Some(payload) = line.strip_prefix('~') { - s.payload_lines.push(payload.to_string()); - if open_block.is_none() { - open_new_block(s, &mut open_block); - } - if let Some(idx) = open_block { - s.payload_blocks[idx].push(payload.to_string()); - } - continue; - } - - let trimmed = line.trim_start(); - let op_byte = trimmed.as_bytes().first().copied(); - match op_byte { - Some(b'+') | Some(b'<') => { - let body = trimmed[1..].trim_start(); - let (anchor_part, tail) = match body.split_once('~') { - Some((a, t)) => (a, Some(t)), - None => (body, None), - }; - let anchor_trimmed = anchor_part.trim(); - if !anchor_trimmed.is_empty() && anchor_trimmed != "BOF" && anchor_trimmed != "EOF" { - s.op_anchors.push(anchor_trimmed.to_string()); - } - if let Some(line) = parse_anchor_line(anchor_part) { - s.touch(line); - } - if let Some(tail) = tail { - // Inline `+ ANCHOR~text`: replaces a single line; not a - // payload-collecting op. - s.payload_lines.push(tail.to_string()); - s.deleted_lines += 1; - close_block(&mut open_block); - } else { - open_new_block(s, &mut open_block); - } - s.op_count += 1; - } - Some(b'-') | Some(b'=') => { - let body = trimmed[1..].trim_start(); - let (size, lines) = parse_range(body); - s.deleted_lines += size; - if let Some((lo, hi)) = lines { - s.touch(lo); - s.touch(hi); - } - // Collect raw anchor refs (`A` or `A..B`) for dup detection. - for part in body.trim().split("..") { - let t = part.trim(); - if !t.is_empty() { - s.op_anchors.push(t.to_string()); - } - } - s.op_count += 1; - if op_byte == Some(b'=') { - open_new_block(s, &mut open_block); - } else { - close_block(&mut open_block); - } - } - _ => { - // blank / unrecognized — leave payload state alone. - } - } - } - - if let Some(s) = cur.take() { - sections.push(s); - } - sections -} - -// ---- success classification ---------------------------------------------- - -static RE_FAILURE_HEAD: LazyLock = LazyLock::new(|| { - Regex::new( - r"(?i)^(edit rejected|error\b|failed\b|invalid\b|unrecognized\b|cannot\b|no enclosing|file has been (modified|changed)|file has not been read|permission denied|tool execution was aborted|request was aborted|cancelled|canceled|line \d+:|expected|unexpected|patch failed|no replacements|0 matches)", - ) - .expect("RE_FAILURE_HEAD") -}); - -fn looks_successful(text: &str) -> bool { - let head = text.lines().find(|l| !l.trim().is_empty()).unwrap_or(""); - if head.is_empty() { - return false; - } - !RE_FAILURE_HEAD.is_match(head.trim_start()) -} - -fn extract_warnings(text: &str) -> Vec<&'static str> { - let mut out: Vec<&'static str> = Vec::new(); - for line in text.lines() { - let t = line.trim_start(); - if t.starts_with("Auto-rebased anchor") { - out.push("auto-rebased"); - } else if t.starts_with("Auto-absorbed") { - out.push("auto-absorbed"); - } else if t.starts_with("Auto-dropped") { - out.push("auto-dropped"); - } - } - out -} - -// ---- followup pattern classification ------------------------------------- - -#[derive(Clone, Copy, PartialEq, Eq, Hash)] -enum FixPattern { - AddSingleCloser, // payload contains exactly one line that's pure }/]/) - RemoveSingleCloser, // single delete of a pure }/]/) line - PureInsertOneLine, - PureDeleteOneLine, - OneLineModify, - SmallOther, -} - -impl FixPattern { - fn label(&self) -> &'static str { - match self { - FixPattern::AddSingleCloser => "add-single-closer", - FixPattern::RemoveSingleCloser => "remove-single-closer", - FixPattern::PureInsertOneLine => "pure-insert-1", - FixPattern::PureDeleteOneLine => "pure-delete-1", - FixPattern::OneLineModify => "one-line-modify", - FixPattern::SmallOther => "small-other", - } - } -} - -fn pattern_priority(p: FixPattern) -> u8 { - match p { - FixPattern::RemoveSingleCloser => 0, - FixPattern::AddSingleCloser => 1, - FixPattern::OneLineModify => 2, - FixPattern::PureDeleteOneLine => 3, - FixPattern::PureInsertOneLine => 4, - FixPattern::SmallOther => 5, - } -} - -static CLOSER_LINE_RE: LazyLock = - LazyLock::new(|| Regex::new(r"^\s*[\])}]+[;,]?\s*$").expect("CLOSER_LINE_RE")); - -fn classify_fix(section: &EditSection) -> FixPattern { - let payload = §ion.payload_lines; - let deleted = section.deleted_lines; - let payload_is_one_closer = payload.len() == 1 && CLOSER_LINE_RE.is_match(&payload[0]); - - if deleted == 0 && payload_is_one_closer { - return FixPattern::AddSingleCloser; - } - if deleted == 1 && payload.is_empty() { - // We don't have the deleted line text, but a single-line delete on a - // sub-1KB follow-up is overwhelmingly the "drop duplicate brace" case - // when paired with the immediately-prior big edit. Mark it so; - // false-positives are easy to triage by hand. - return FixPattern::RemoveSingleCloser; - } - if deleted == 0 && payload.len() == 1 { - return FixPattern::PureInsertOneLine; - } - if deleted == 1 && payload.len() == 1 { - return FixPattern::OneLineModify; - } - if deleted >= 1 && payload.is_empty() { - return FixPattern::PureDeleteOneLine; - } - FixPattern::SmallOther -} - -// ---- per-file scanning --------------------------------------------------- - -#[derive(Clone)] -struct Hit { - session: String, - file: String, - first_call_id: String, - second_call_id: String, - first_size: i64, - first_input_len: usize, - second_size: i64, - pattern: FixPattern, - second_summary: String, - gap_secs: i64, -} - -/// A successful edit whose result text contained a tool self-correction. -#[derive(Clone)] -struct WarningHit { - session: String, - call_id: String, - kind: &'static str, // auto-rebased / auto-absorbed / auto-dropped - files: String, // comma-joined section files - input_len: usize, -} - -/// Two consecutive successful edits on the same file whose anchor line -/// ranges overlap (or one is contained in the other). Catches "agent -/// re-edited the same locus" cases that the small-fix detector misses. -#[derive(Clone)] -struct LocusHit { - session: String, - file: String, - first_call_id: String, - second_call_id: String, - first_range: (i64, i64), - second_range: (i64, i64), - first_size: i64, - second_size: i64, - gap_secs: i64, -} - -/// A single hashline edit whose payload contains a contiguous N-line sequence -/// that repeats inside the same payload block. Strong signal of "model pasted -/// the same chunk twice" when N is large. -#[derive(Clone)] -struct PayloadDupHit { - session: String, - call_id: String, - file: String, - block_len: usize, - repeat_len: usize, - sample: String, -} - -/// A single hashline edit that referenced the same anchor (e.g. `123ab`) from -/// two or more distinct ops. Often benign (insert before + insert after at -/// the same line), occasionally indicates a duplicated op in the input. -#[derive(Clone)] -struct AnchorDupHit { - session: String, - call_id: String, - files: String, - anchor: String, - count: usize, -} - -#[derive(Default)] -struct FileReport { - fix_hits: Vec, - warning_hits: Vec, - locus_hits: Vec, - payload_dups: Vec, - anchor_dups: Vec, - /// Total number of successful hashline edits scanned. - total_successful_edits: i64, -} - -/// Find the longest contiguous N-line sequence in `block` that occurs at two -/// distinct positions. Returns `(first_index, repeat_len)` if a match of at -/// least `min_len` lines exists with at least half its lines being -/// non-trivial (>= 4 chars after trim) — this filters out e.g. five repeated -/// `}` lines as boilerplate. -fn find_longest_repeat(block: &[String], min_len: usize) -> Option<(usize, usize)> { - let n = block.len(); - if n < 2 * min_len { - return None; - } - let mut best: Option<(usize, usize)> = None; - for i in 0..n.saturating_sub(min_len) { - for j in (i + min_len)..=n.saturating_sub(min_len) { - let mut k = 0; - while i + k < j && j + k < n && block[i + k] == block[j + k] { - k += 1; - } - if k < min_len { - continue; - } - let meaningful = block[i..i + k] - .iter() - .filter(|s| s.trim().len() >= 4) - .count(); - if meaningful < (k.div_ceil(2)).max(2) { - continue; - } - if best.is_none_or(|(_, bk)| k > bk) { - best = Some((i, k)); - } - } - } - best -} - -/// Returns the set of `(anchor, count, file)` pairs where the same raw -/// anchor was referenced by `count >= 2` distinct ops within ONE section -/// (i.e. on one file). Two files happening to have the same `LINE+hash` -/// anchor is a coincidence (2-char hashes collide), not a duplicate op. -/// Skips `BOF` / `EOF` and any anchor with a `*` interior-hash. -fn duplicated_anchors(sections: &[EditSection]) -> Vec<(String, usize, String)> { - let mut out: Vec<(String, usize, String)> = Vec::new(); - for sec in sections { - let mut counts: HashMap<&str, usize> = HashMap::new(); - for a in &sec.op_anchors { - if a == "BOF" || a == "EOF" || a.contains('*') { - continue; - } - *counts.entry(a.as_str()).or_default() += 1; - } - for (anchor, c) in counts { - if c >= 2 { - out.push((anchor.to_string(), c, sec.file.clone())); - } - } - } - out.sort_by(|a, b| b.1.cmp(&a.1)); - out -} - -fn process_file(path: &Path, max_fix: i64) -> FileReport { - let f = match File::open(path) { - Ok(f) => f, - Err(e) => { - eprintln!("open {}: {e}", path.display()); - return FileReport::default(); - } - }; - let reader = BufReader::with_capacity(64 * 1024, f); - let session = path - .file_stem() - .and_then(|s| s.to_str()) - .unwrap_or("") - .to_string(); - - // Walk message events in order. For each toolCall name=="edit", parse the - // input. Pair with its toolResult by callId. Keep the chronological list - // of EditCall records. - let mut pending: HashMap = HashMap::new(); - let mut order: Vec = Vec::new(); - let mut calls: HashMap = HashMap::new(); - - for line in reader.lines() { - let Ok(line) = line else { continue }; - if line.is_empty() { - continue; - } - let Ok(ev) = serde_json::from_str::(&line) else { - continue; - }; - if ev.kind != "message" { - continue; - } - let Some(msg_raw) = ev.message else { continue }; - let Ok(m) = serde_json::from_str::(msg_raw.get()) else { - continue; - }; - let Some(content_raw) = m.content else { continue }; - let items = parse_content(&content_raw); - - match m.role.as_str() { - "assistant" => { - for it in items { - if it.kind != "toolCall" || it.name != "edit" { - continue; - } - let raw = it.arguments.as_deref(); - let Some(raw) = raw else { continue }; - let Ok(args) = serde_json::from_str::(raw.get()) else { - continue; - }; - let Some(input) = args.input else { continue }; - if !input.trim_start().starts_with('@') { - // Vim-mode edit or other shape — skip. - continue; - } - let sections = parse_hashline_input(&input); - if sections.is_empty() { - continue; - } - let call = EditCall { - ts: parse_ts(&ev.timestamp), - call_id: it.id.clone(), - sections, - success: false, // filled in by toolResult - raw_input_len: input.len(), - warnings: Vec::new(), - }; - pending.insert(it.id.clone(), call); - } - } - "toolResult" => { - if m.tool_name != "edit" { - continue; - } - let Some(mut call) = pending.remove(&m.tool_call_id) else { - continue; - }; - let text = join_text(&items); - call.success = looks_successful(&text); - if call.success { - call.warnings = extract_warnings(&text); - } - let id = call.call_id.clone(); - calls.insert(id.clone(), call); - order.push(id); - } - _ => {} - } - } - - // Build per-file edit sequences. A single call can touch multiple files - // (multi-section input); we record an entry per (call, section). - struct Entry<'a> { - section: &'a EditSection, - call: &'a EditCall, - } - let mut by_file: HashMap<&str, Vec> = HashMap::new(); - for id in &order { - let Some(call) = calls.get(id) else { continue }; - for sec in &call.sections { - by_file - .entry(sec.file.as_str()) - .or_default() - .push(Entry { section: sec, call }); - } - } - - let mut report = FileReport::default(); - for call in calls.values() { - if call.success { - report.total_successful_edits += 1; - } - if !call.warnings.is_empty() { - // Dedup per kind so a call with two `Auto-rebased` lines counts once. - let mut seen: Vec<&'static str> = Vec::new(); - for w in &call.warnings { - if !seen.contains(w) { - seen.push(*w); - } - } - let files: Vec = call.sections.iter().map(|s| s.file.clone()).collect(); - for kind in seen { - report.warning_hits.push(WarningHit { - session: session.clone(), - call_id: call.call_id.clone(), - kind, - files: files.join(","), - input_len: call.raw_input_len, - }); - } - } - - // Only flag detectors on edits that actually applied. Failed edits - // are handled separately and their input is moot. - if call.success { - // Payload self-dup: per section, per block. Threshold is - // intentionally permissive (>=4) — caller filters by repeat_len. - for sec in &call.sections { - for block in &sec.payload_blocks { - if let Some((start, len)) = find_longest_repeat(block, 4) { - let sample = truncate_line(block.get(start).map(String::as_str).unwrap_or(""), 80); - report.payload_dups.push(PayloadDupHit { - session: session.clone(), - call_id: call.call_id.clone(), - file: sec.file.clone(), - block_len: block.len(), - repeat_len: len, - sample, - }); - } - } - } - - // Anchor dup within a single section's ops. - for (anchor, count, file) in duplicated_anchors(&call.sections) { - report.anchor_dups.push(AnchorDupHit { - session: session.clone(), - call_id: call.call_id.clone(), - files: file, - anchor, - count, - }); - } - } - } - - for (file, entries) in by_file { - for window in entries.windows(2) { - let a = &window[0]; - let b = &window[1]; - if !a.call.success || !b.call.success { - continue; - } - if a.call.call_id == b.call.call_id { - continue; - } - let first_size = a.section.change_size(); - let second_size = b.section.change_size(); - let gap = (b.call.ts - a.call.ts).max(0); - - // (1) Small-fix follow-up. - if second_size > 0 && second_size <= max_fix && first_size > 2 { - let pattern = classify_fix(b.section); - let summary = render_section_summary(b.section); - report.fix_hits.push(Hit { - session: session.clone(), - file: file.to_string(), - first_call_id: a.call.call_id.clone(), - second_call_id: b.call.call_id.clone(), - first_size, - first_input_len: a.call.raw_input_len, - second_size, - pattern, - second_summary: summary, - gap_secs: gap, - }); - } - - // (2) Same-locus re-edit. Skip when both are tiny (those are - // already noisy) and skip when the small-fix branch already - // fired (we don't want to double-count). - if first_size > 2 && second_size > max_fix { - if let (Some(a_lo), Some(a_hi), Some(b_lo), Some(b_hi)) = ( - a.section.min_line, - a.section.max_line, - b.section.min_line, - b.section.max_line, - ) { - let overlap_lo = a_lo.max(b_lo); - let overlap_hi = a_hi.min(b_hi); - if overlap_lo <= overlap_hi { - report.locus_hits.push(LocusHit { - session: session.clone(), - file: file.to_string(), - first_call_id: a.call.call_id.clone(), - second_call_id: b.call.call_id.clone(), - first_range: (a_lo, a_hi), - second_range: (b_lo, b_hi), - first_size, - second_size, - gap_secs: gap, - }); - } - } - } - } - } - report -} - -fn render_section_summary(section: &EditSection) -> String { - let mut bits: Vec = Vec::new(); - if section.deleted_lines > 0 { - bits.push(format!("-{}", section.deleted_lines)); - } - if !section.payload_lines.is_empty() { - bits.push(format!("+{}", section.payload_lines.len())); - } - let mut out = bits.join(" / "); - if !section.payload_lines.is_empty() { - let preview = truncate_line(§ion.payload_lines[0], 80); - out.push_str(&format!(" | {preview}")); - } - out -} - -// ---- entry point --------------------------------------------------------- - -pub fn run(args: Vec) -> Result<()> { - let mut limit: usize = 200; - let mut workers: usize = 0; - let mut max_fix: i64 = 2; - let mut show: usize = 60; - let mut max_gap: i64 = 0; - let mut pattern_filter: Option = None; - let mut min_dup: usize = 8; - - let mut iter = args.into_iter(); - while let Some(a) = iter.next() { - match a.as_str() { - "-n" => { - limit = iter - .next() - .context("-n requires a value")? - .parse() - .context("-n value")?; - } - "-j" => { - workers = iter - .next() - .context("-j requires a value")? - .parse() - .context("-j value")?; - } - "--max-fix" => { - max_fix = iter - .next() - .context("--max-fix requires a value")? - .parse() - .context("--max-fix value")?; - } - "--show" => { - show = iter - .next() - .context("--show requires a value")? - .parse() - .context("--show value")?; - } - "--max-gap" => { - max_gap = iter - .next() - .context("--max-gap requires seconds")? - .parse() - .context("--max-gap value")?; - } - "--pattern" => { - pattern_filter = Some(iter.next().context("--pattern requires a name")?); - } - "--min-dup" => { - min_dup = iter - .next() - .context("--min-dup requires a value")? - .parse() - .context("--min-dup value")?; - } - "-h" | "--help" => { - eprintln!( - "usage: session-stats followups [-n N] [-j workers] [--max-fix N] [--max-gap S] - [--min-dup K] [--pattern NAME] [--show N] - -Five detectors over hashline `edit` calls in the most-recent N sessions: - - 1. small-fix follow-ups - consecutive successful edits on the same file where the follow-up - changes <= --max-fix lines (default 2). The first edit must be > 2 - lines. Brace-related patterns surface first (remove-single-closer, - add-single-closer). --max-gap caps elapsed seconds between the pair. - - 2. tool self-corrections - warning lines emitted by the tool on otherwise-successful edits: - auto-rebased, auto-absorbed, auto-dropped. These are direct evidence - of the model writing stale anchors or duplicating adjacent context. - - 3. same-locus re-edits - two consecutive edits on the same file whose anchor line ranges - overlap, where both edits are > --max-fix lines (so they're not - already in (1)). - - 4. payload self-duplication - within one payload block, a contiguous K-line sequence appears at two - positions. K threshold via --min-dup (default 8). - - 5. same-anchor reuse - two or more ops in the same section reference the same `LINE+hash` - anchor. - ---pattern NAME filters (1) to a single FixPattern label." - ); - return Ok(()); - } - other => bail!("unknown flag: {other}"), - } - } - - let files = collect_sessions(&WalkOpts { - date_filters: Vec::new(), - limit_most_recent: limit, - })?; - eprintln!("scanning {} session files", files.len()); - - let max_fix_local = max_fix; - let reports: Vec = parallel_collect(&files, workers, 5_000, |p| { - let r = process_file(p, max_fix_local); - let empty = r.fix_hits.is_empty() && r.warning_hits.is_empty() && r.locus_hits.is_empty() - && r.total_successful_edits == 0; - if empty { None } else { Some(r) } - }); - - let mut hits: Vec = Vec::new(); - let mut warning_hits: Vec = Vec::new(); - let mut locus_hits: Vec = Vec::new(); - let mut payload_dups: Vec = Vec::new(); - let mut anchor_dups: Vec = Vec::new(); - let mut total_successful_edits: i64 = 0; - for r in reports { - hits.extend(r.fix_hits); - warning_hits.extend(r.warning_hits); - locus_hits.extend(r.locus_hits); - payload_dups.extend(r.payload_dups); - anchor_dups.extend(r.anchor_dups); - total_successful_edits += r.total_successful_edits; - } - - if max_gap > 0 { - hits.retain(|h| h.gap_secs <= max_gap); - locus_hits.retain(|h| h.gap_secs <= max_gap); - } - if let Some(ref name) = pattern_filter { - hits.retain(|h| h.pattern.label() == name); - } - hits.sort_by(|a, b| { - pattern_priority(a.pattern) - .cmp(&pattern_priority(b.pattern)) - .then_with(|| b.first_input_len.cmp(&a.first_input_len)) - }); - locus_hits.sort_by(|a, b| b.first_size.cmp(&a.first_size)); - - let mut by_pattern: HashMap<&'static str, i64> = HashMap::new(); - for h in &hits { - *by_pattern.entry(h.pattern.label()).or_default() += 1; - } - - println!("=== heuristic followup hits ==="); - println!("total hits: {}", commas(hits.len() as i64)); - println!(); - println!("by pattern:"); - let mut pats: Vec<(&&str, &i64)> = by_pattern.iter().collect(); - pats.sort_by(|a, b| b.1.cmp(a.1)); - for (label, n) in pats { - println!(" {:<22} {:>6}", label, commas(*n)); - } - println!(); - - let shown = show.min(hits.len()); - println!("=== top {shown} hits (by first-edit input size) ==="); - for h in hits.iter().take(shown) { - println!( - "[{pat}] {file} first={first_size}L ({first_len}B) → second={second_size}L gap={gap}s", - pat = h.pattern.label(), - file = h.file, - first_size = h.first_size, - first_len = h.first_input_len, - second_size = h.second_size, - gap = h.gap_secs, - ); - println!(" session={}", h.session); - println!(" first_call={} second_call={}", h.first_call_id, h.second_call_id); - println!(" fix: {}", h.second_summary); - } - println!(); - println!("=== tool self-corrections ==="); - println!("(emitted as warnings on otherwise-successful edits — the tool caught what the model wrote)"); - let mut warn_by_kind: HashMap<&'static str, i64> = HashMap::new(); - for w in &warning_hits { - *warn_by_kind.entry(w.kind).or_default() += 1; - } - let mut warn_kinds: Vec<(&&str, &i64)> = warn_by_kind.iter().collect(); - warn_kinds.sort_by(|a, b| b.1.cmp(a.1)); - let pct_of = |n: i64| -> String { - if total_successful_edits == 0 { - String::new() - } else { - format!(" ({:.2}% of {} successful edits)", pct(n, total_successful_edits), commas(total_successful_edits)) - } - }; - for (kind, n) in warn_kinds { - println!(" {:<16} {:>6}{}", kind, commas(*n), pct_of(*n)); - } - let warn_show = show.min(warning_hits.len()); - if warn_show > 0 { - println!(); - println!("--- top {warn_show} self-correction examples (by input size) ---"); - let mut sorted_warns = warning_hits.clone(); - sorted_warns.sort_by(|a, b| b.input_len.cmp(&a.input_len)); - for w in sorted_warns.iter().take(warn_show) { - println!( - "[{kind}] {files} ({len}B)", - kind = w.kind, - files = truncate_line(&w.files, 80), - len = w.input_len, - ); - println!(" session={} call={}", w.session, w.call_id); - } - } - - println!(); - println!("=== same-locus re-edits (overlapping anchor ranges, both > max-fix) ==="); - println!("hits: {}", commas(locus_hits.len() as i64)); - let locus_show = show.min(locus_hits.len()); - for h in locus_hits.iter().take(locus_show) { - println!( - "{file} first={a0}..{a1} ({fs}L) → second={b0}..{b1} ({ss}L) gap={gap}s", - file = h.file, - a0 = h.first_range.0, - a1 = h.first_range.1, - fs = h.first_size, - b0 = h.second_range.0, - b1 = h.second_range.1, - ss = h.second_size, - gap = h.gap_secs, - ); - println!(" session={}", h.session); - println!(" first_call={} second_call={}", h.first_call_id, h.second_call_id); - } - - println!(); - println!("=== payload self-duplication (model pasted same N-line chunk twice in one payload) ==="); - payload_dups.retain(|p| p.repeat_len >= min_dup); - payload_dups.sort_by(|a, b| b.repeat_len.cmp(&a.repeat_len)); - println!( - "hits with repeat_len >= {}: {} ({:.2}% of {} successful edits)", - min_dup, - commas(payload_dups.len() as i64), - pct(payload_dups.len() as i64, total_successful_edits), - commas(total_successful_edits) - ); - let dup_show = show.min(payload_dups.len()); - for p in payload_dups.iter().take(dup_show) { - println!( - "k={k} block={blk}L {file}", - k = p.repeat_len, - blk = p.block_len, - file = p.file, - ); - println!(" session={} call={}", p.session, p.call_id); - println!(" sample: {}", p.sample); - } - - println!(); - println!("=== same-anchor reused by multiple ops in one input ==="); - anchor_dups.sort_by(|a, b| b.count.cmp(&a.count)); - println!("hits: {}", commas(anchor_dups.len() as i64)); - let anchor_show = show.min(anchor_dups.len()); - for a in anchor_dups.iter().take(anchor_show) { - println!( - "anchor {anchor} referenced {n}x files={files}", - anchor = a.anchor, - n = a.count, - files = truncate_line(&a.files, 80), - ); - println!(" session={} call={}", a.session, a.call_id); - } - - Ok(()) -} diff --git a/scripts/session-stats/src/cmd_tools.rs b/scripts/session-stats/src/cmd_tools.rs deleted file mode 100644 index ebf70dac2..000000000 --- a/scripts/session-stats/src/cmd_tools.rs +++ /dev/null @@ -1,702 +0,0 @@ -//! `tools` subcommand — per-tool token totals across the most-recent N session -//! jsonl files. -//! -//! Token counting uses o200k_base via tiktoken-rs (the GPT-4o / GPT-5 family -//! tokenizer). It is not Claude's own BPE, but it is well-defined offline and -//! within ~5-10% across English/code in aggregate. -//! -//! Buckets: -//! tool ARGS — assistant tool-call argument JSON -//! tool RESULTS — tool result content text -//! assistant THINKING — assistant `thinking` blocks -//! assistant TEXT — assistant prose -//! user TEXT — user-authored text content -//! -//! Output: grand totals + per-tool breakdown sorted by total (arg+res) tokens. -//! Optional CSV at `$TOOL_USAGE_CSV` (per-tool totals) or -//! `--calls-csv PATH` / `$TOOL_CALLS_CSV` (one row per tool call). -//! -//! Pass `--by ` to bucket per-call data into rolling windows -//! and surface per-tool tokens/call (avg + p50 + p95) over time, so you can -//! spot regressions in tool efficiency. The buckets do not align to calendar -//! boundaries; they are pure `floor(unix_secs / N)` slices. - -use crate::common::*; -use anyhow::{Context, Result, bail}; -use std::collections::HashMap; -use std::fs::File; -use std::io::{BufRead, BufReader}; -use std::path::Path; - -#[derive(Default, Clone)] -struct ToolAgg { - calls: i64, - results: i64, - arg_tok: i64, - res_tok: i64, -} - -#[derive(Default, Clone)] -struct SessionTotals { - arg_tok: i64, - res_tok: i64, - thinking_tok: i64, - text_tok: i64, - user_tok: i64, - n_calls: i64, - n_results: i64, -} - -struct FileResult { - totals: SessionTotals, - tools: HashMap, - calls: Vec, -} - -#[derive(Clone)] -struct CallRecord { - ts: i64, - tool: String, - session: String, - model: String, - arg_tok: i32, - res_tok: i32, -} - -struct PendingCall { - tool: String, - ts: i64, - arg_tok: i32, - model: String, -} - -pub fn run(args: Vec) -> Result<()> { - let mut limit: usize = 1_000; - let mut workers: usize = 0; - let mut by: Option = None; - let mut top: usize = 12; - let mut tool_filter: Option = None; - let mut calls_csv: Option = std::env::var("TOOL_CALLS_CSV") - .ok() - .filter(|s| !s.is_empty()); - - let mut iter = args.into_iter(); - while let Some(a) = iter.next() { - match a.as_str() { - "-n" => { - limit = iter - .next() - .context("-n requires a value")? - .parse() - .context("-n value")?; - } - "-j" => { - workers = iter - .next() - .context("-j requires a value")? - .parse() - .context("-j value")?; - } - "--by" => { - let spec = iter.next().context("--by requires a bucket spec")?; - by = Some(parse_bucket(&spec)?); - } - "--top" => { - top = iter - .next() - .context("--top requires a value")? - .parse() - .context("--top value")?; - } - "--tool" => { - tool_filter = Some(iter.next().context("--tool requires a name")?); - } - "--calls-csv" => { - calls_csv = Some(iter.next().context("--calls-csv requires a path")?); - } - "-h" | "--help" => { - eprintln!( -"usage: session-stats tools [-n N] [-j workers] [--by SPEC] [--top N] - [--tool NAME] [--calls-csv PATH] - -Aggregates per-tool token usage across the most-recent N session -jsonl files (default 1000). Tokenizer: o200k_base. - - --by SPEC bucket per-call data into rolling windows. SPEC is one of: - hour, day, week, month, or {{h,d,w}} (e.g. 7d, 12h, 2w). - Buckets are pure floor(unix_secs / N); they do not align to - calendar boundaries. - --top N limit per-tool series to the N most-called tools (default 12). - --tool NAME show only this tool in the bucketed series. - --calls-csv PATH - emit one CSV row per tool call (ts, session, tool, model, - arg_tok, res_tok). Env fallback: TOOL_CALLS_CSV. - -TOOL_USAGE_CSV env still emits the per-tool grand-totals CSV." - ); - return Ok(()); - } - other => bail!("unknown flag: {other}"), - } - } - - let files = collect_sessions(&WalkOpts { - date_filters: Vec::new(), - limit_most_recent: limit, - })?; - eprintln!( - "scanning {} session files (tokenizer: o200k_base)", - files.len() - ); - - let results = parallel_collect(&files, workers, 5_000, process_file); - - let sessions = results.len(); - let mut grand = SessionTotals::default(); - let mut tools: HashMap = HashMap::new(); - let mut all_calls: Vec = Vec::new(); - for r in results { - grand.arg_tok += r.totals.arg_tok; - grand.res_tok += r.totals.res_tok; - grand.thinking_tok += r.totals.thinking_tok; - grand.text_tok += r.totals.text_tok; - grand.user_tok += r.totals.user_tok; - grand.n_calls += r.totals.n_calls; - grand.n_results += r.totals.n_results; - for (name, t) in r.tools { - let dst = tools.entry(name).or_default(); - dst.calls += t.calls; - dst.results += t.results; - dst.arg_tok += t.arg_tok; - dst.res_tok += t.res_tok; - } - all_calls.extend(r.calls); - } - - if let Some(ref name) = tool_filter { - all_calls.retain(|c| &c.tool == name); - } - - print_grand(&grand, sessions); - println!(); - print_table(&tools); - write_csv(&tools)?; - - if let Some(bucket_secs) = by { - println!(); - print_buckets(&all_calls, bucket_secs, top, tool_filter.as_deref()); - } - if let Some(path) = calls_csv { - write_calls_csv(&all_calls, &path)?; - eprintln!("wrote {} calls to {path}", commas(all_calls.len() as i64)); - } - Ok(()) -} - -fn process_file(path: &Path) -> Option { - let f = match File::open(path) { - Ok(f) => f, - Err(e) => { - eprintln!("open {}: {e}", path.display()); - return None; - } - }; - let reader = BufReader::with_capacity(64 * 1024, f); - - let mut totals = SessionTotals::default(); - let mut tools: HashMap = HashMap::new(); - let mut calls: Vec = Vec::new(); - let session = session_id_from_path(path); - // Pending call records: keyed by toolCallId so the matching toolResult - // can finalize a CallRecord with both arg+res tokens. Falls back to - // message.toolName when the id is absent (legacy sessions). - let mut pending: HashMap = HashMap::new(); - - for line in reader.lines() { - let Ok(line) = line else { continue }; - if line.is_empty() { - continue; - } - let Ok(ev) = serde_json::from_str::(&line) else { - continue; - }; - if ev.kind != "message" { - continue; - } - let Some(msg_raw) = ev.message else { continue }; - let Ok(m) = serde_json::from_str::(msg_raw.get()) else { - continue; - }; - let Some(content_raw) = m.content else { continue }; - let items = parse_content(&content_raw); - - match m.role.as_str() { - "assistant" => { - for it in items { - match it.kind.as_str() { - "toolCall" => { - let name = normalize_tool(&it.name); - let args_str = it.arguments.as_deref().map(RawValue::get).unwrap_or(""); - let tok = count_tokens(args_str) as i64; - totals.arg_tok += tok; - totals.n_calls += 1; - let t = tools.entry(name.clone()).or_default(); - t.calls += 1; - t.arg_tok += tok; - pending.insert( - it.id, - PendingCall { - tool: name, - ts: parse_ts(&ev.timestamp), - arg_tok: clamp_i32(tok), - model: m.model.clone(), - }, - ); - } - "thinking" => { - totals.thinking_tok += count_tokens(&it.thinking) as i64; - } - "text" => { - totals.text_tok += count_tokens(&it.text) as i64; - } - _ => {} - } - } - } - "toolResult" => { - let text = join_text(&items); - let tok = count_tokens(&text) as i64; - totals.res_tok += tok; - totals.n_results += 1; - let pc = pending.remove(&m.tool_call_id); - let name = pc - .as_ref() - .map(|p| p.tool.clone()) - .unwrap_or_else(|| normalize_tool(&m.tool_name)); - let t = tools.entry(name.clone()).or_default(); - t.results += 1; - t.res_tok += tok; - if let Some(p) = pc { - calls.push(CallRecord { - ts: p.ts, - tool: name, - session: session.clone(), - model: p.model, - arg_tok: p.arg_tok, - res_tok: clamp_i32(tok), - }); - } - } - "user" => { - for it in items { - if it.kind == "text" { - totals.user_tok += count_tokens(&it.text) as i64; - } - } - } - _ => {} - } - } - - Some(FileResult { totals, tools, calls }) -} - -use serde_json::value::RawValue; - -fn normalize_tool(name: &str) -> String { - if name.is_empty() { - "".to_string() - } else { - name.to_string() - } -} - -// ---- reporting ---- - -fn print_grand(g: &SessionTotals, sessions: usize) { - let total = g.arg_tok + g.res_tok + g.thinking_tok + g.text_tok + g.user_tok; - let rows: [(&str, i64); 5] = [ - ("tool call ARGS", g.arg_tok), - ("tool RESULTS", g.res_tok), - ("assistant THINKING", g.thinking_tok), - ("assistant TEXT", g.text_tok), - ("user TEXT", g.user_tok), - ]; - let label_w = rows.iter().map(|(l, _)| l.len()).max().unwrap_or(0); - let val_w = rows - .iter() - .map(|(_, n)| commas(*n).len()) - .chain(std::iter::once(commas(total).len())) - .max() - .unwrap_or(0); - - println!("=== Grand totals across {} sessions ===", commas(sessions as i64)); - for (label, n) in rows { - println!( - "{label:val_w$} tok ({:>5.1}%)", - commas(n), - pct(n, total), - ); - } - println!("{:val_w$} tok", "TOTAL", commas(total)); - println!(); - println!( - "tool calls: {}, tool results: {}", - commas(g.n_calls), - commas(g.n_results) - ); - if g.n_calls > 0 { - println!( - "avg arg tokens / call: {:.1}", - g.arg_tok as f64 / g.n_calls as f64 - ); - } - if g.n_results > 0 { - println!( - "avg result tokens / call: {:.1}", - g.res_tok as f64 / g.n_results as f64 - ); - } - if g.arg_tok > 0 { - println!( - "ratio result / arg: {:.2}x", - g.res_tok as f64 / g.arg_tok as f64 - ); - } -} - -struct ToolRow { - name: String, - calls: i64, - arg_tok: i64, - res_tok: i64, - total: i64, - avg_arg: f64, - avg_res: f64, - res_o_arg: f64, -} - -fn print_table(tools: &HashMap) { - let mut rows: Vec = tools - .iter() - .filter_map(|(name, t)| { - if t.calls == 0 && t.results == 0 { - return None; - } - let mut r = ToolRow { - name: name.clone(), - calls: t.calls, - arg_tok: t.arg_tok, - res_tok: t.res_tok, - total: t.arg_tok + t.res_tok, - avg_arg: 0.0, - avg_res: 0.0, - res_o_arg: 0.0, - }; - if t.calls > 0 { - r.avg_arg = t.arg_tok as f64 / t.calls as f64; - r.avg_res = t.res_tok as f64 / t.calls as f64; - } - if t.arg_tok > 0 { - r.res_o_arg = t.res_tok as f64 / t.arg_tok as f64; - } - Some(r) - }) - .collect(); - rows.sort_by(|a, b| b.total.cmp(&a.total)); - - const TOP: usize = 25; - let shown = TOP.min(rows.len()); - let head_rows = &rows[..shown]; - - // "(N others)" trailing summary, computed before width measurement so its - // string contents participate in column sizing. - let others = (rows.len() > TOP).then(|| { - let mut sc = 0i64; - let mut sa = 0i64; - let mut sr = 0i64; - for r in &rows[TOP..] { - sc += r.calls; - sa += r.arg_tok; - sr += r.res_tok; - } - OthersRow { - label: format!("({} others)", rows.len() - TOP), - calls: sc, - arg_tok: sa, - res_tok: sr, - total: sa + sr, - } - }); - - // Compute column widths from header label and every value that will - // appear under that header (including the "others" summary row, if any). - let max_str = |header: &str, vals: &[&str]| -> usize { - vals.iter().map(|s| s.len()).chain(std::iter::once(header.len())).max().unwrap_or(0) - }; - - let names: Vec<&str> = head_rows - .iter() - .map(|r| r.name.as_str()) - .chain(others.as_ref().map(|o| o.label.as_str())) - .collect(); - let calls: Vec = head_rows - .iter() - .map(|r| commas(r.calls)) - .chain(others.as_ref().map(|o| commas(o.calls))) - .collect(); - let arg_toks: Vec = head_rows - .iter() - .map(|r| commas(r.arg_tok)) - .chain(others.as_ref().map(|o| commas(o.arg_tok))) - .collect(); - let res_toks: Vec = head_rows - .iter() - .map(|r| commas(r.res_tok)) - .chain(others.as_ref().map(|o| commas(o.res_tok))) - .collect(); - let totals: Vec = head_rows - .iter() - .map(|r| commas(r.total)) - .chain(others.as_ref().map(|o| commas(o.total))) - .collect(); - let avg_args: Vec = head_rows.iter().map(|r| format!("{:.1}", r.avg_arg)).collect(); - let avg_ress: Vec = head_rows.iter().map(|r| format!("{:.1}", r.avg_res)).collect(); - let res_o_args: Vec = - head_rows.iter().map(|r| format!("{:.2}", r.res_o_arg)).collect(); - - let name_w = max_str("tool", &names); - let calls_w = max_str("calls", &calls.iter().map(String::as_str).collect::>()); - let arg_w = max_str("arg_tok", &arg_toks.iter().map(String::as_str).collect::>()); - let res_w = max_str("res_tok", &res_toks.iter().map(String::as_str).collect::>()); - let tot_w = max_str("total", &totals.iter().map(String::as_str).collect::>()); - let avga_w = max_str("avg_arg", &avg_args.iter().map(String::as_str).collect::>()); - let avgr_w = max_str("avg_res", &avg_ress.iter().map(String::as_str).collect::>()); - let ratio_w = max_str("res/arg", &res_o_args.iter().map(String::as_str).collect::>()); - - let total_width = name_w + 1 + calls_w + 1 + arg_w + 1 + res_w + 1 + tot_w + 1 + avga_w + 1 + avgr_w + 1 + ratio_w; - - println!( - "{:calls_w$} {:>arg_w$} {:>res_w$} {:>tot_w$} {:>avga_w$} {:>avgr_w$} {:>ratio_w$}", - "tool", "calls", "arg_tok", "res_tok", "total", "avg_arg", "avg_res", "res/arg" - ); - println!("{}", "-".repeat(total_width)); - - for (i, r) in head_rows.iter().enumerate() { - println!( - "{:calls_w$} {:>arg_w$} {:>res_w$} {:>tot_w$} {:>avga_w$} {:>avgr_w$} {:>ratio_w$}", - r.name, calls[i], arg_toks[i], res_toks[i], totals[i], avg_args[i], avg_ress[i], res_o_args[i], - ); - } - if let Some(o) = others { - let i = head_rows.len(); - println!( - "{:calls_w$} {:>arg_w$} {:>res_w$} {:>tot_w$}", - o.label, calls[i], arg_toks[i], res_toks[i], totals[i], - ); - } -} - -struct OthersRow { - label: String, - calls: i64, - arg_tok: i64, - res_tok: i64, - total: i64, -} - -fn write_csv(tools: &HashMap) -> Result<()> { - let path = std::env::var("TOOL_USAGE_CSV").unwrap_or_default(); - if path.is_empty() { - return Ok(()); - } - let f = File::create(&path).with_context(|| format!("create {path}"))?; - let mut w = csv::Writer::from_writer(f); - w.write_record(["tool", "calls", "results", "arg_tok", "res_tok", "total"])?; - let mut names: Vec<&String> = tools.keys().collect(); - names.sort_by(|a, b| { - let ai = { - let t = &tools[a.as_str()]; - t.arg_tok + t.res_tok - }; - let aj = { - let t = &tools[b.as_str()]; - t.arg_tok + t.res_tok - }; - aj.cmp(&ai) - }); - for n in names { - let t = &tools[n.as_str()]; - w.write_record([ - n.as_str(), - &t.calls.to_string(), - &t.results.to_string(), - &t.arg_tok.to_string(), - &t.res_tok.to_string(), - &(t.arg_tok + t.res_tok).to_string(), - ])?; - } - w.flush()?; - Ok(()) -} - -// ---- per-call helpers ---- - -fn session_id_from_path(path: &Path) -> String { - path.file_stem() - .and_then(|s| s.to_str()) - .unwrap_or("") - .to_string() -} - -fn clamp_i32(n: i64) -> i32 { - n.clamp(0, i32::MAX as i64) as i32 -} - -fn print_buckets( - calls: &[CallRecord], - bucket_secs: i64, - top: usize, - tool_filter: Option<&str>, -) { - if calls.is_empty() { - println!("(no per-call records — no toolCall/toolResult pairs found)"); - return; - } - - // Pick the tools to show: top-N by call count, ignoring records with ts==0 - // (events that lacked a parseable timestamp). - let mut per_tool: HashMap<&str, i64> = HashMap::new(); - for c in calls { - if c.ts == 0 { - continue; - } - *per_tool.entry(c.tool.as_str()).or_default() += 1; - } - let mut ranked: Vec<(&str, i64)> = per_tool.into_iter().collect(); - ranked.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(b.0))); - if tool_filter.is_none() { - ranked.truncate(top); - } - - let bucket_human = humanize_bucket(bucket_secs); - println!( - "=== per-call tokens by {} bucket (rolling, no calendar alignment) ===", - bucket_human - ); - println!( - "showing {} tool{} (sorted by call count). totals/calls match per-call records only;", - ranked.len(), - if ranked.len() == 1 { "" } else { "s" } - ); - println!( - "calls without a matching toolResult or without a parseable timestamp are skipped here." - ); - - // Group calls per (tool, bucket_id). - let mut by_bucket: HashMap<(&str, i64), Vec<&CallRecord>> = HashMap::new(); - for c in calls { - if c.ts == 0 { - continue; - } - if !ranked.iter().any(|(t, _)| *t == c.tool.as_str()) { - continue; - } - let bid = c.ts.div_euclid(bucket_secs); - by_bucket.entry((c.tool.as_str(), bid)).or_default().push(c); - } - - for (tool, total_calls) in ranked { - let mut bids: Vec = by_bucket - .keys() - .filter(|(t, _)| *t == tool) - .map(|(_, b)| *b) - .collect(); - if bids.is_empty() { - continue; - } - bids.sort(); - - println!(); - println!("=== {tool} ({} calls) ===", commas(total_calls)); - println!( - "{:<22} {:>7} {:>9} {:>9} {:>9} {:>9} {:>9}", - "window", "calls", "avg_arg", "avg_res", "avg_tot", "p50_tot", "p95_tot" - ); - let dashes = 22 + 1 + 7 + 5 * (1 + 9); - println!("{}", "-".repeat(dashes)); - - for bid in bids { - let records = by_bucket.get(&(tool, bid)).expect("present"); - let n = records.len() as i64; - let mut sum_arg = 0i64; - let mut sum_res = 0i64; - let mut totals: Vec = Vec::with_capacity(records.len()); - for r in records { - sum_arg += r.arg_tok as i64; - sum_res += r.res_tok as i64; - totals.push(r.arg_tok as i64 + r.res_tok as i64); - } - let avg_arg = sum_arg as f64 / n as f64; - let avg_res = sum_res as f64 / n as f64; - let avg_tot = avg_arg + avg_res; - let p50 = percentile(&mut totals.clone(), 50.0); - let p95 = percentile(&mut totals, 95.0); - let label = bucket_label(bid * bucket_secs, bucket_secs); - println!( - "{:<22} {:>7} {:>9} {:>9} {:>9} {:>9} {:>9}", - label, - commas(n), - commas(avg_arg.round() as i64), - commas(avg_res.round() as i64), - commas(avg_tot.round() as i64), - commas(p50.round() as i64), - commas(p95.round() as i64), - ); - } - } -} - -fn humanize_bucket(bucket_secs: i64) -> String { - if bucket_secs % (7 * 86_400) == 0 { - let n = bucket_secs / (7 * 86_400); - if n == 1 { "7d".into() } else { format!("{n}w") } - } else if bucket_secs % 86_400 == 0 { - format!("{}d", bucket_secs / 86_400) - } else if bucket_secs % 3600 == 0 { - format!("{}h", bucket_secs / 3600) - } else { - format!("{}s", bucket_secs) - } -} - -fn write_calls_csv(calls: &[CallRecord], path: &str) -> Result<()> { - let f = File::create(path).with_context(|| format!("create {path}"))?; - let mut w = csv::Writer::from_writer(f); - w.write_record([ - "ts_unix", - "ts_iso", - "session", - "tool", - "model", - "arg_tok", - "res_tok", - "total_tok", - ])?; - for c in calls { - let total = c.arg_tok as i64 + c.res_tok as i64; - w.write_record([ - &c.ts.to_string(), - &format_iso(c.ts), - &c.session, - &c.tool, - &c.model, - &c.arg_tok.to_string(), - &c.res_tok.to_string(), - &total.to_string(), - ])?; - } - w.flush()?; - Ok(()) -} diff --git a/scripts/session-stats/src/common.rs b/scripts/session-stats/src/common.rs deleted file mode 100644 index 2e363ef00..000000000 --- a/scripts/session-stats/src/common.rs +++ /dev/null @@ -1,344 +0,0 @@ -//! Shared JSONL shapes, walk helpers, tokenizer, and formatting helpers. - -use anyhow::{Context, Result}; -use rayon::prelude::*; -use serde::Deserialize; -use serde_json::value::RawValue; -use std::collections::HashMap; -use std::path::{Path, PathBuf}; -use std::sync::LazyLock; -use std::sync::atomic::{AtomicU64, Ordering}; -use std::time::SystemTime; -use tiktoken_rs::CoreBPE; -use walkdir::WalkDir; - -// ---- jsonl shapes ---- - -#[derive(Deserialize)] -pub struct RawEvent { - #[serde(rename = "type", default)] - pub kind: String, - #[serde(default)] - pub message: Option>, - #[serde(default)] - pub timestamp: String, -} - -#[derive(Deserialize)] -pub struct Message { - #[serde(default)] - pub role: String, - #[serde(default)] - pub content: Option>, - #[serde(default, rename = "toolName")] - pub tool_name: String, - #[serde(default, rename = "toolCallId")] - pub tool_call_id: String, - #[serde(default)] - pub model: String, -} - -#[derive(Deserialize)] -pub struct ContentItem { - #[serde(rename = "type", default)] - pub kind: String, - #[serde(default)] - pub text: String, - #[serde(default)] - pub thinking: String, - #[serde(default)] - pub name: String, - #[serde(default)] - pub id: String, - #[serde(default)] - pub arguments: Option>, -} - -// ---- session walking ---- - -pub fn sessions_root() -> Result { - let home = dirs::home_dir().context("could not resolve home directory")?; - Ok(home.join(".omp").join("agent").join("sessions")) -} - -pub struct WalkOpts { - /// Keeps only paths containing any of these substrings (e.g. "2026-04-28"). - /// Empty means accept all. - pub date_filters: Vec, - /// Keeps only the N most-recently-modified files (after the date filter). - /// 0 means no limit. - pub limit_most_recent: usize, -} - -/// Walks the sessions root and returns the matching `.jsonl` paths. -/// With `limit_most_recent > 0` the result is sorted by mtime descending and -/// truncated to N entries; otherwise it's lexically sorted. -pub fn collect_sessions(opts: &WalkOpts) -> Result> { - let base = sessions_root()?; - let need_mtime = opts.limit_most_recent > 0; - - let mut all: Vec<(PathBuf, SystemTime)> = Vec::new(); - for entry in WalkDir::new(&base).into_iter().filter_map(Result::ok) { - if !entry.file_type().is_file() { - continue; - } - let p = entry.path(); - if p.extension().and_then(|e| e.to_str()) != Some("jsonl") { - continue; - } - let path_str = p.to_string_lossy(); - if !match_date(&path_str, &opts.date_filters) { - continue; - } - let mt = if need_mtime { - entry - .metadata() - .ok() - .and_then(|m| m.modified().ok()) - .unwrap_or(SystemTime::UNIX_EPOCH) - } else { - SystemTime::UNIX_EPOCH - }; - all.push((p.to_path_buf(), mt)); - } - - if need_mtime { - all.sort_by(|a, b| b.1.cmp(&a.1)); - all.truncate(opts.limit_most_recent); - } else { - all.sort_by(|a, b| a.0.cmp(&b.0)); - } - Ok(all.into_iter().map(|(p, _)| p).collect()) -} - -fn match_date(p: &str, filters: &[String]) -> bool { - filters.is_empty() || filters.iter().any(|d| p.contains(d)) -} - -// ---- content helpers ---- - -pub fn parse_content(raw: &RawValue) -> Vec { - serde_json::from_str(raw.get()).unwrap_or_default() -} - -/// Concatenates all `text` items in a content array. -pub fn join_text(items: &[ContentItem]) -> String { - let mut out = String::new(); - for it in items { - if it.kind == "text" { - out.push_str(&it.text); - } - } - out -} - -// ---- tokenizer (o200k_base) ---- - -static BPE: LazyLock = - LazyLock::new(|| tiktoken_rs::o200k_base().expect("load o200k_base BPE")); - -/// Counts tokens for `s` using the o200k_base BPE (GPT-4o / GPT-5 family). -/// Uses the ordinary encoder so embedded `<|...|>` sequences in tool args do -/// not trigger special-token handling. -pub fn count_tokens(s: &str) -> usize { - if s.is_empty() { - return 0; - } - BPE.encode_ordinary(s).len() -} - -// ---- formatting helpers ---- - -/// Formats an integer with thousand separators. -pub fn commas(n: i64) -> String { - let neg = n < 0; - let mag = if neg { (n as i128).unsigned_abs() } else { n as u128 }; - let s = mag.to_string(); - let bytes = s.as_bytes(); - let mut out = String::with_capacity(bytes.len() + bytes.len() / 3 + 1); - if neg { - out.push('-'); - } - let pre = bytes.len() % 3; - if pre > 0 { - out.push_str(&s[..pre]); - if bytes.len() > pre { - out.push(','); - } - } - let mut i = pre; - while i + 3 <= bytes.len() { - out.push_str(&s[i..i + 3]); - if i + 3 < bytes.len() { - out.push(','); - } - i += 3; - } - out -} - -pub fn pct(a: i64, b: i64) -> f64 { - if b == 0 { - 0.0 - } else { - 100.0 * a as f64 / b as f64 - } -} - -/// Truncates a string to at most `n` chars, replacing newlines with " | ". -/// Adds an ellipsis when truncation occurs. -pub fn truncate_line(s: &str, n: usize) -> String { - let s = s.replace('\n', " | "); - if s.chars().count() <= n { - return s; - } - let mut out: String = s.chars().take(n).collect(); - out.push('…'); - out -} - -/// Returns map keys sorted by descending value, ties broken alphabetically. -pub fn sorted_by_count(m: &HashMap) -> Vec<&String> { - let mut keys: Vec<&String> = m.keys().collect(); - keys.sort_by(|a, b| { - let av = m.get(a.as_str()).copied().unwrap_or(0); - let bv = m.get(b.as_str()).copied().unwrap_or(0); - bv.cmp(&av).then_with(|| a.cmp(b)) - }); - keys -} - -pub fn print_sorted(m: &HashMap) { - print_sorted_indent(m, " "); -} - -pub fn print_sorted_indent(m: &HashMap, indent: &str) { - for k in sorted_by_count(m) { - println!("{indent}{k:<25} {}", m[k.as_str()]); - } -} - -// ---- parallel processing ---- - -/// Runs `handle(path)` in parallel across rayon workers and collects the -/// non-`None` results into a Vec. Logs progress every `progress_every` files -/// (set 0 to silence). -pub fn parallel_collect( - paths: &[PathBuf], - workers: usize, - progress_every: u64, - handle: H, -) -> Vec -where - R: Send, - H: Fn(&Path) -> Option + Sync, -{ - let total = paths.len(); - let done = AtomicU64::new(0); - - let pool = { - let mut b = rayon::ThreadPoolBuilder::new(); - if workers > 0 { - b = b.num_threads(workers); - } - b.build().expect("rayon thread pool") - }; - - pool.install(|| { - paths - .par_iter() - .filter_map(|p| { - let r = handle(p); - let n = done.fetch_add(1, Ordering::Relaxed) + 1; - if progress_every > 0 && n % progress_every == 0 { - eprintln!(" processed {n}/{total}"); - } - r - }) - .collect() - }) -} - -// ---- timestamp / bucket helpers ---- - -/// Parses an RFC3339 timestamp into unix seconds. Returns 0 on failure. -pub fn parse_ts(s: &str) -> i64 { - if s.is_empty() { - return 0; - } - chrono::DateTime::parse_from_rfc3339(s) - .map(|dt| dt.timestamp()) - .unwrap_or(0) -} - -/// Parses a bucket spec like "h", "day", "week", "month", "1h", "12h", "7d", "2w". -/// Returns the bucket size in seconds. -pub fn parse_bucket(spec: &str) -> anyhow::Result { - use anyhow::bail; - let s = spec.trim(); - let secs: i64 = match s { - "" => bail!("empty bucket spec"), - "hour" | "h" | "1h" => 3600, - "day" | "d" | "1d" => 86_400, - "week" | "w" | "1w" | "7d" => 7 * 86_400, - "month" | "mo" | "1mo" | "30d" => 30 * 86_400, - other => { - let bytes = other.as_bytes(); - let last = *bytes.last().unwrap(); - let unit_secs: i64 = match last { - b'h' => 3600, - b'd' => 86_400, - b'w' => 7 * 86_400, - _ => bail!("bad bucket spec {spec:?} (use h/d/w or e.g. 7d, 12h)"), - }; - let n: i64 = other[..other.len() - 1] - .parse() - .map_err(|_| anyhow::anyhow!("bad bucket count in {spec:?}"))?; - if n <= 0 { - bail!("bucket count must be > 0"); - } - n * unit_secs - } - }; - Ok(secs) -} - -/// Returns a label for the bucket starting at `start_secs`. -/// Buckets shorter than a day include the hour; multi-day buckets show start..end (exclusive). -pub fn bucket_label(start_secs: i64, bucket_secs: i64) -> String { - use chrono::TimeZone; - let start = chrono::Utc.timestamp_opt(start_secs, 0).single(); - let end = chrono::Utc.timestamp_opt(start_secs + bucket_secs - 1, 0).single(); - match (start, end) { - (Some(s), _) if bucket_secs < 86_400 => s.format("%Y-%m-%d %H:00").to_string(), - (Some(s), _) if bucket_secs == 86_400 => s.format("%Y-%m-%d").to_string(), - (Some(s), Some(e)) => format!( - "{}..{}", - s.format("%Y-%m-%d"), - e.format("%Y-%m-%d") - ), - _ => start_secs.to_string(), - } -} - -/// Formats a unix timestamp as an RFC3339 string (UTC). -pub fn format_iso(unix_secs: i64) -> String { - use chrono::TimeZone; - chrono::Utc - .timestamp_opt(unix_secs, 0) - .single() - .map(|dt| dt.format("%Y-%m-%dT%H:%M:%SZ").to_string()) - .unwrap_or_default() -} - -/// Computes a percentile over an integer slice. The slice is sorted in place. -/// `p` is 0..=100. -pub fn percentile(v: &mut [i64], p: f64) -> f64 { - if v.is_empty() { - return 0.0; - } - v.sort_unstable(); - let p = p.clamp(0.0, 100.0); - let idx = ((p / 100.0) * (v.len() as f64 - 1.0)).round() as usize; - v[idx.min(v.len() - 1)] as f64 -} diff --git a/scripts/session-stats/src/main.rs b/scripts/session-stats/src/main.rs deleted file mode 100644 index af07ea1f5..000000000 --- a/scripts/session-stats/src/main.rs +++ /dev/null @@ -1,63 +0,0 @@ -//! session-stats: ad-hoc analyses over the local agent session corpus -//! (`~/.omp/agent/sessions/`). -//! -//! Subcommands: -//! -//! edits [-n N] [date ...] audit edit/ast_edit/write tool usage by argument schema -//! tools [-n N] per-tool token totals across the most-recent N sessions -//! -//! Run with no subcommand for help. - -mod cmd_edits; -mod cmd_followups; -mod cmd_tools; -mod common; - -use std::process::ExitCode; - -fn usage() { - eprintln!( - "usage: session-stats [args...] - - edits [-n N] [date prefix ...] - audit edit-tool usage across N most-recent - sessions (default 1000). Optional date filters - (e.g. 2026-04-28) further narrow the set. - tools [-n N] [--by SPEC] [--top N] [--tool NAME] [--calls-csv PATH] - per-tool token totals across the N most-recent - session jsonl files (default 1000). With --by, - also bucket per-call tokens (avg + p50/p95) into - rolling windows so you can spot regressions. - -Token counting uses the o200k_base tokenizer (the GPT-4o / Claude-adjacent BPE). -Walk root: ~/.omp/agent/sessions/" - ); -} - -fn main() -> ExitCode { - let mut args = std::env::args().skip(1); - let Some(cmd) = args.next() else { - usage(); - return ExitCode::from(2); - }; - let rest: Vec = args.collect(); - let result = match cmd.as_str() { - "edits" => cmd_edits::run(rest), - "followups" => cmd_followups::run(rest), - "tools" => cmd_tools::run(rest), - "-h" | "--help" | "help" => { - usage(); - return ExitCode::SUCCESS; - } - other => { - eprintln!("unknown subcommand {other:?}\n"); - usage(); - return ExitCode::from(2); - } - }; - if let Err(err) = result { - eprintln!("fatal: {err:#}"); - return ExitCode::FAILURE; - } - ExitCode::SUCCESS -} diff --git a/scripts/session-stats/sync.py b/scripts/session-stats/sync.py new file mode 100644 index 000000000..d0ced3207 --- /dev/null +++ b/scripts/session-stats/sync.py @@ -0,0 +1,1019 @@ +#!/usr/bin/env python3 +""" +Sync ~/.omp/agent/sessions/**/*.jsonl into ~/.omp/stats.db (ss_* tables). + +Incremental: per-file byte offset + mtime is tracked in ss_sessions. Only new +bytes are parsed on re-runs. Tokenization (o200k_base) and the hashline edit +parser/detectors are computed once and persisted, so analyses become pure SQL +plus tiny Python loops. + +Schema (all tables prefixed `ss_` to avoid collision with packages/stats): + + ss_sessions one row per .jsonl, carries sync state + metadata + ss_tool_calls one row per toolCall content block + ss_tool_results one row per toolResult message + ss_assistant_msgs one row per assistant message (text + thinking blobs) + ss_user_msgs one row per user message (text blob) + ss_edit_calls one row per edit toolCall (success + warnings paired in) + ss_edit_sections one row per @PATH section inside an edit toolCall, with + precomputed detector outputs (longest_repeat_*, dup_anchors) + +Run: + python3 scripts/session-stats/sync.py + python3 scripts/session-stats/sync.py --workers 16 --full + python3 scripts/session-stats/sync.py --limit 200 # newest 200 files +""" + +from __future__ import annotations + +import argparse +import json +import os +import queue +import re +import sqlite3 +import sys +import threading +import time +from collections import defaultdict +from concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor, as_completed +from dataclasses import dataclass, field +from pathlib import Path + +try: + import tiktoken +except ImportError: + sys.exit("tiktoken not installed. Run: pip install tiktoken") + + +# --------------------------------------------------------------------------- # +# Config + +SESSIONS_ROOT = Path.home() / ".omp" / "agent" / "sessions" +DB_PATH = Path.home() / ".omp" / "stats.db" +TOKENIZER_NAME = "o200k_base" +SCHEMA_VERSION = 2 +# Bump whenever parse_hashline_input / find_longest_repeat / duplicated_anchors +# / looks_successful / extract_warnings semantics change. Bump invalidates +# previously-stored ss_edit_* rows on next sync. +EDIT_PARSER_VERSION = 1 + +SCHEMA_SQL = """ +CREATE TABLE IF NOT EXISTS ss_sessions ( + session_file TEXT PRIMARY KEY, + folder TEXT NOT NULL, + is_subagent INTEGER NOT NULL DEFAULT 0, + parent_session TEXT, + subagent_label TEXT, + started_at INTEGER, + title TEXT, + cwd TEXT, + session_uuid TEXT, + version INTEGER, + mtime INTEGER NOT NULL, + size INTEGER NOT NULL, + byte_offset INTEGER NOT NULL DEFAULT 0, + line_count INTEGER NOT NULL DEFAULT 0, + last_synced INTEGER NOT NULL, + tokenizer TEXT NOT NULL, + schema_version INTEGER NOT NULL DEFAULT 1, + parser_version INTEGER NOT NULL DEFAULT 0 +); + +CREATE TABLE IF NOT EXISTS ss_tool_calls ( + id INTEGER PRIMARY KEY, + session_file TEXT NOT NULL, + seq INTEGER NOT NULL, + entry_id TEXT, + call_id TEXT NOT NULL, + tool_name TEXT NOT NULL, + raw_tool_name TEXT NOT NULL, + timestamp INTEGER NOT NULL, + model TEXT, + provider TEXT, + arg_json TEXT, + arg_tokens INTEGER, + UNIQUE(session_file, call_id, seq) +); +CREATE INDEX IF NOT EXISTS ss_tc_tool_ts ON ss_tool_calls(tool_name, timestamp); +CREATE INDEX IF NOT EXISTS ss_tc_sess_seq ON ss_tool_calls(session_file, seq); + +CREATE TABLE IF NOT EXISTS ss_tool_results ( + id INTEGER PRIMARY KEY, + session_file TEXT NOT NULL, + seq INTEGER NOT NULL, + entry_id TEXT, + call_id TEXT NOT NULL, + tool_name TEXT NOT NULL, + raw_tool_name TEXT NOT NULL, + timestamp INTEGER NOT NULL, + result_text TEXT, + result_tokens INTEGER, + is_error INTEGER NOT NULL DEFAULT 0, + UNIQUE(session_file, call_id, seq) +); +CREATE INDEX IF NOT EXISTS ss_tr_tool_ts ON ss_tool_results(tool_name, timestamp); +CREATE INDEX IF NOT EXISTS ss_tr_sess_seq ON ss_tool_results(session_file, seq); +CREATE INDEX IF NOT EXISTS ss_tr_call_id ON ss_tool_results(session_file, call_id); + +CREATE TABLE IF NOT EXISTS ss_assistant_msgs ( + session_file TEXT NOT NULL, + seq INTEGER NOT NULL, + entry_id TEXT, + timestamp INTEGER NOT NULL, + model TEXT, + provider TEXT, + text_blob TEXT, + thinking_blob TEXT, + text_tokens INTEGER NOT NULL DEFAULT 0, + thinking_tokens INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY (session_file, seq) +); + +CREATE TABLE IF NOT EXISTS ss_user_msgs ( + session_file TEXT NOT NULL, + seq INTEGER NOT NULL, + entry_id TEXT, + timestamp INTEGER NOT NULL, + text_blob TEXT, + text_tokens INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY (session_file, seq) +); + +CREATE TABLE IF NOT EXISTS ss_edit_calls ( + session_file TEXT NOT NULL, + call_id TEXT NOT NULL, + seq INTEGER NOT NULL, + timestamp INTEGER NOT NULL, + raw_input_len INTEGER NOT NULL DEFAULT 0, + success INTEGER, -- nullable until result paired + warnings TEXT NOT NULL DEFAULT '[]', -- JSON list[str] + parser_version INTEGER NOT NULL DEFAULT 1, + PRIMARY KEY (session_file, call_id) +); +CREATE INDEX IF NOT EXISTS ss_ec_sess_seq ON ss_edit_calls(session_file, seq); +CREATE INDEX IF NOT EXISTS ss_ec_success ON ss_edit_calls(success); + +CREATE TABLE IF NOT EXISTS ss_edit_sections ( + id INTEGER PRIMARY KEY, + session_file TEXT NOT NULL, + call_id TEXT NOT NULL, + seq INTEGER NOT NULL, + section_idx INTEGER NOT NULL, + target_file TEXT NOT NULL, + op_count INTEGER NOT NULL, + deleted_lines INTEGER NOT NULL, + payload_count INTEGER NOT NULL, + change_size INTEGER NOT NULL, + min_line INTEGER, + max_line INTEGER, + payload_blocks TEXT NOT NULL DEFAULT '[]', -- JSON list[list[str]] + op_anchors TEXT NOT NULL DEFAULT '[]', -- JSON list[str] + longest_repeat_len INTEGER NOT NULL DEFAULT 0, + longest_repeat_block_idx INTEGER, + longest_repeat_sample TEXT, + dup_anchors TEXT NOT NULL DEFAULT '[]', -- JSON list[[anchor,count]] + parser_version INTEGER NOT NULL DEFAULT 1, + UNIQUE(session_file, call_id, section_idx) +); +CREATE INDEX IF NOT EXISTS ss_es_target ON ss_edit_sections(session_file, target_file, seq); +CREATE INDEX IF NOT EXISTS ss_es_repeat ON ss_edit_sections(longest_repeat_len); +""" + + +def _migrate(conn: sqlite3.Connection) -> None: + """Best-effort additive migrations for older databases.""" + cols = {r[1] for r in conn.execute("PRAGMA table_info(ss_sessions)").fetchall()} + if "parser_version" not in cols: + conn.execute( + "ALTER TABLE ss_sessions ADD COLUMN parser_version INTEGER NOT NULL DEFAULT 0" + ) + + +# --------------------------------------------------------------------------- # +# Tokenizer (one per worker thread). + +_tls = threading.local() + + +def get_encoder() -> "tiktoken.Encoding": + enc = getattr(_tls, "enc", None) + if enc is None: + enc = tiktoken.get_encoding(TOKENIZER_NAME) + _tls.enc = enc + return enc + + +def count_tokens(s: str) -> int: + if not s: + return 0 + return len(get_encoder().encode_ordinary(s)) + + +def batch_count_tokens(strings: list[str]) -> list[int]: + """Tokenize many strings in one FFI call. Empty strings short-circuit.""" + if not strings: + return [] + enc = get_encoder() + nonempty_idx = [i for i, s in enumerate(strings) if s] + if not nonempty_idx: + return [0] * len(strings) + nonempty = [strings[i] for i in nonempty_idx] + encoded = enc.encode_ordinary_batch(nonempty, num_threads=4) + out = [0] * len(strings) + for i, e in zip(nonempty_idx, encoded): + out[i] = len(e) + return out + + +# --------------------------------------------------------------------------- # +# Hashline edit parser (port of cmd_followups.rs::parse_hashline_input). + +_RANGE_RE = re.compile(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$") +_SINGLE_ANCHOR_RE = re.compile(r"^\s*(\d+)[a-z*]+\s*$") + + +def _parse_range(raw: str) -> tuple[int, tuple[int, int] | None]: + """Returns (range_size, optional (start_line, end_line)). Size >= 1.""" + m = _RANGE_RE.match(raw.strip()) + if not m: + return (1, None) + start = int(m.group(1)) + end_raw = m.group(2) + end = int(end_raw) if end_raw else start + size = max(end - start + 1, 1) + lines = (start, max(end, start)) if start > 0 else None + return (size, lines) + + +def _parse_anchor_line(raw: str) -> int | None: + m = _SINGLE_ANCHOR_RE.match(raw.strip()) + if not m: + return None + try: + return int(m.group(1)) + except ValueError: + return None + + +@dataclass +class EditSection: + target_file: str = "" + payload_blocks: list[list[str]] = field(default_factory=list) + op_anchors: list[str] = field(default_factory=list) + deleted_lines: int = 0 + op_count: int = 0 + min_line: int | None = None + max_line: int | None = None + + def touch(self, line: int) -> None: + self.min_line = line if self.min_line is None else min(self.min_line, line) + self.max_line = line if self.max_line is None else max(self.max_line, line) + + @property + def payload_count(self) -> int: + return sum(len(b) for b in self.payload_blocks) + + @property + def change_size(self) -> int: + return self.payload_count + self.deleted_lines + + +def parse_hashline_input(input_str: str) -> list[EditSection]: + sections: list[EditSection] = [] + cur: EditSection | None = None + open_idx: int | None = None # current open payload block in cur + + def open_new(s: EditSection) -> int: + s.payload_blocks.append([]) + return len(s.payload_blocks) - 1 + + for raw_line in input_str.split("\n"): + line = raw_line[:-1] if raw_line.endswith("\r") else raw_line + + if line.startswith("@"): + if cur is not None: + sections.append(cur) + cur = EditSection(target_file=line[1:].strip()) + open_idx = None + continue + if cur is None: + continue + + if line.startswith("~"): + payload = line[1:] + if open_idx is None: + open_idx = open_new(cur) + cur.payload_blocks[open_idx].append(payload) + continue + + trimmed = line.lstrip() + if not trimmed: + continue + op = trimmed[0] + if op in ("+", "<"): + body = trimmed[1:].lstrip() + if "~" in body: + anchor_part, tail = body.split("~", 1) + else: + anchor_part, tail = body, None + anchor_trimmed = anchor_part.strip() + if anchor_trimmed and anchor_trimmed not in ("BOF", "EOF"): + cur.op_anchors.append(anchor_trimmed) + line_no = _parse_anchor_line(anchor_part) + if line_no is not None: + cur.touch(line_no) + if tail is not None: + # Inline `+ ANCHOR~text`: replaces a single line. + if open_idx is None: + open_idx = open_new(cur) + cur.payload_blocks[open_idx].append(tail) + cur.deleted_lines += 1 + open_idx = None + else: + open_idx = open_new(cur) + cur.op_count += 1 + elif op in ("-", "="): + body = trimmed[1:].lstrip() + size, lines = _parse_range(body) + cur.deleted_lines += size + if lines is not None: + cur.touch(lines[0]) + cur.touch(lines[1]) + for part in body.strip().split(".."): + t = part.strip() + if t: + cur.op_anchors.append(t) + cur.op_count += 1 + if op == "=": + open_idx = open_new(cur) + else: + open_idx = None + # else: blank / unrecognized — keep payload state. + + if cur is not None: + sections.append(cur) + return sections + + +def find_longest_repeat(block: list[str], min_len: int = 4) -> tuple[int, int] | None: + """Returns (start_index, repeat_len) if a repeat of >= min_len with at least + half meaningful lines exists. O(n^2) per block — fine for typical edits.""" + n = len(block) + if n < 2 * min_len: + return None + best: tuple[int, int] | None = None + for i in range(n - min_len + 1): + for j in range(i + min_len, n - min_len + 1): + k = 0 + while i + k < j and j + k < n and block[i + k] == block[j + k]: + k += 1 + if k < min_len: + continue + meaningful = sum(1 for s in block[i : i + k] if len(s.strip()) >= 4) + if meaningful < max((k + 1) // 2, 2): + continue + if best is None or k > best[1]: + best = (i, k) + return best + + +def duplicated_anchors(sections: list[EditSection]) -> list[list]: + """Returns [[anchor, count, target_file], ...] for anchors referenced + by >= 2 ops within one section. Skips BOF/EOF/* anchors.""" + out: list[list] = [] + for sec in sections: + counts: dict[str, int] = defaultdict(int) + for a in sec.op_anchors: + if a in ("BOF", "EOF") or "*" in a: + continue + counts[a] += 1 + for anchor, c in counts.items(): + if c >= 2: + out.append([anchor, c, sec.target_file]) + out.sort(key=lambda r: -r[1]) + return out + + +# --------------------------------------------------------------------------- # +# Edit result classification (port of cmd_followups.rs success/warnings). + +_RE_FAILURE_HEAD = re.compile( + r"^(edit rejected|error\b|failed\b|invalid\b|unrecognized\b|cannot\b|" + r"no enclosing|file has been (modified|changed)|file has not been read|" + r"permission denied|tool execution was aborted|request was aborted|" + r"cancelled|canceled|line \d+:|expected|unexpected|patch failed|" + r"no replacements|0 matches)", + re.IGNORECASE, +) + + +def looks_successful(text: str) -> bool: + if not text: + return False + head = "" + for ln in text.split("\n"): + if ln.strip(): + head = ln + break + if not head: + return False + return _RE_FAILURE_HEAD.match(head.lstrip()) is None + + +def extract_warnings(text: str) -> list[str]: + out: list[str] = [] + for ln in text.split("\n"): + t = ln.lstrip() + if t.startswith("Auto-rebased anchor"): + out.append("auto-rebased") + elif t.startswith("Auto-absorbed"): + out.append("auto-absorbed") + elif t.startswith("Auto-dropped"): + out.append("auto-dropped") + return out + + +# --------------------------------------------------------------------------- # +# JSONL parsing + +def parse_iso_ms(s: str | None) -> int: + if not s: + return 0 + try: + if s.endswith("Z"): + s = s[:-1] + "+00:00" + from datetime import datetime + return int(datetime.fromisoformat(s).timestamp() * 1000) + except Exception: + return 0 + + +def join_text(items) -> str: + parts: list[str] = [] + for it in items or []: + if not isinstance(it, dict): + continue + t = it.get("text") + if isinstance(t, str) and t: + parts.append(t) + return "\n".join(parts) + + +def session_meta_from_path(path: Path) -> tuple[str, bool, str | None, str | None]: + parts = path.parts + try: + idx = parts.index("sessions") + except ValueError: + return (path.parent.name, False, None, None) + rel = parts[idx + 1 :] + if len(rel) == 2: + return (rel[0], False, None, None) + if len(rel) == 3: + folder = rel[0] + parent = str(path.parent) + ".jsonl" + label = path.stem + return (folder, True, parent, label) + return (rel[0] if rel else path.parent.name, False, None, None) + + +def _is_edit_call(name: str) -> bool: + # cmd_followups.rs targets `edit` only (hashline). + return name == "edit" + + +@dataclass +class SessionRecords: + session_meta: dict + tool_calls: list[list] = field(default_factory=list) + tool_results: list[list] = field(default_factory=list) + assistant_msgs: list[list] = field(default_factory=list) + user_msgs: list[list] = field(default_factory=list) + edit_calls: list[tuple] = field(default_factory=list) # initial stub on toolCall + edit_call_results: list[tuple] = field(default_factory=list) # success+warnings on toolResult + edit_sections: list[tuple] = field(default_factory=list) # one row per section + pending_tokens: list[tuple] = field(default_factory=list) # (row, field_idx, text) + starting_seq: int = 0 + full_rebuild: bool = False + starting_offset: int = 0 + final_offset: int = 0 + final_line_count: int = 0 + file_size: int = 0 + file_mtime: int = 0 + + +def parse_file( + path: Path, + starting_offset: int, + starting_seq: int, + full_rebuild: bool, +) -> SessionRecords | None: + try: + st = path.stat() + except FileNotFoundError: + return None + + folder, is_subagent, parent_session, subagent_label = session_meta_from_path(path) + rec = SessionRecords( + session_meta={ + "session_file": str(path), + "folder": folder, + "is_subagent": int(is_subagent), + "parent_session": parent_session, + "subagent_label": subagent_label, + "cwd": None, + "session_uuid": None, + "version": None, + "title": None, + "started_at": None, + }, + starting_seq=starting_seq, + starting_offset=starting_offset, + full_rebuild=full_rebuild, + file_size=st.st_size, + file_mtime=int(st.st_mtime * 1000), + ) + + seq = starting_seq + offset = starting_offset + try: + with path.open("rb") as f: + if starting_offset: + f.seek(starting_offset) + for raw in f: + offset += len(raw) + if not raw.strip(): + seq += 1 + continue + try: + ev = json.loads(raw) + except json.JSONDecodeError: + seq += 1 + continue + + kind = ev.get("type") + ts = parse_iso_ms(ev.get("timestamp")) + entry_id = ev.get("id") + + if kind == "session" and seq == 0: + rec.session_meta["session_uuid"] = ev.get("id") + rec.session_meta["version"] = ev.get("version") + rec.session_meta["title"] = ev.get("title") + rec.session_meta["cwd"] = ev.get("cwd") + rec.session_meta["started_at"] = ts or None + elif kind == "message": + msg = ev.get("message") or {} + role = msg.get("role") + content = msg.get("content") + if role == "assistant" and isinstance(content, list): + _ingest_assistant(rec, path, seq, entry_id, ts, msg, content) + elif role == "toolResult": + _ingest_tool_result(rec, path, seq, entry_id, ts, msg) + elif role == "user" and isinstance(content, list): + _ingest_user(rec, path, seq, entry_id, ts, content) + seq += 1 + except OSError as e: + print(f"!! {path}: {e}", file=sys.stderr) + return None + + rec.final_offset = offset + rec.final_line_count = seq + + # Single batched tokenization pass for the whole file. + if rec.pending_tokens: + texts = [p[2] for p in rec.pending_tokens] + tokens = batch_count_tokens(texts) + for (row, field_idx, _), n in zip(rec.pending_tokens, tokens): + row[field_idx] = n + rec.pending_tokens.clear() + return rec + + +def _parse_worker(item: tuple[Path, bool, int, int]) -> SessionRecords | None: + path, full_rebuild, start_off, start_seq = item + return parse_file(path, start_off, start_seq, full_rebuild) + + +def _ingest_assistant(rec, path, seq, entry_id, ts, msg, content) -> None: + sf = str(path) + model = msg.get("model") + provider = msg.get("provider") + + text_parts: list[str] = [] + thinking_parts: list[str] = [] + for it in content: + if not isinstance(it, dict): + continue + t = it.get("type") + if t == "toolCall": + call_id = it.get("id") or "" + raw_name = it.get("name") or "" + tool_name = raw_name or "" + arg_obj = it.get("arguments") + if arg_obj is None: + arg_json = "" + elif isinstance(arg_obj, str): + arg_json = arg_obj + else: + arg_json = json.dumps(arg_obj, separators=(",", ":"), ensure_ascii=False) + row = [ + sf, seq, entry_id, call_id, + tool_name, raw_name, ts, model, provider, + arg_json, 0, + ] + rec.tool_calls.append(row) + if arg_json: + rec.pending_tokens.append((row, 10, arg_json)) + if _is_edit_call(tool_name): + _ingest_edit_call(rec, sf, seq, ts, call_id, arg_obj, arg_json) + elif t == "thinking": + v = it.get("thinking") + if isinstance(v, str) and v: + thinking_parts.append(v) + elif t == "text": + v = it.get("text") + if isinstance(v, str) and v: + text_parts.append(v) + + text_blob = "\n".join(text_parts) if text_parts else None + thinking_blob = "\n".join(thinking_parts) if thinking_parts else None + text_tokens_slot = 0 + thinking_tokens_slot = 0 + if text_blob or thinking_blob: + row = [ + sf, seq, entry_id, ts, model, provider, + text_blob, thinking_blob, text_tokens_slot, thinking_tokens_slot, + ] + rec.assistant_msgs.append(row) + if text_blob: + rec.pending_tokens.append((row, 8, text_blob)) + if thinking_blob: + rec.pending_tokens.append((row, 9, thinking_blob)) + + +def _ingest_edit_call(rec, sf, seq, ts, call_id, arg_obj, arg_json) -> None: + """Parse the hashline `input` and emit ss_edit_calls + ss_edit_sections rows.""" + # Recover `input` from arg_obj (preferred) or arg_json (legacy). + input_str: str | None = None + if isinstance(arg_obj, dict): + v = arg_obj.get("input") + if isinstance(v, str): + input_str = v + if input_str is None and arg_json: + try: + parsed = json.loads(arg_json) + v = parsed.get("input") if isinstance(parsed, dict) else None + if isinstance(v, str): + input_str = v + except Exception: + pass + if input_str is None: + input_str = "" + raw_input_len = len(input_str.encode("utf-8")) + + # Stub call row (success + warnings come from toolResult later). + rec.edit_calls.append( + (sf, call_id, seq, ts, raw_input_len, EDIT_PARSER_VERSION) + ) + + if not input_str.lstrip().startswith("@"): + # Vim-mode or other shape — no sections to record. + return + + sections = parse_hashline_input(input_str) + for idx, sec in enumerate(sections): + repeat = None + repeat_block_idx: int | None = None + for bi, block in enumerate(sec.payload_blocks): + r = find_longest_repeat(block, 4) + if r is None: + continue + start_i, k = r + if repeat is None or k > repeat[1]: + repeat = (start_i, k) + repeat_block_idx = bi + if repeat is not None and repeat_block_idx is not None: + blk = sec.payload_blocks[repeat_block_idx] + sample_line = blk[repeat[0]] if repeat[0] < len(blk) else "" + sample = (sample_line[:80] + "…") if len(sample_line) > 80 else sample_line + longest_repeat_len = repeat[1] + else: + sample = None + longest_repeat_len = 0 + + dups = duplicated_anchors([sec]) + + rec.edit_sections.append( + ( + sf, call_id, seq, idx, sec.target_file, + sec.op_count, sec.deleted_lines, sec.payload_count, sec.change_size, + sec.min_line, sec.max_line, + json.dumps(sec.payload_blocks, ensure_ascii=False), + json.dumps(sec.op_anchors, ensure_ascii=False), + longest_repeat_len, repeat_block_idx, sample, + json.dumps(dups, ensure_ascii=False), + EDIT_PARSER_VERSION, + ) + ) + + +def _ingest_tool_result(rec, path, seq, entry_id, ts, msg) -> None: + sf = str(path) + call_id = msg.get("toolCallId") or "" + raw_name = msg.get("toolName") or "" + tool_name = raw_name or "" + content = msg.get("content") + text = join_text(content) if isinstance(content, list) else "" + is_error = 1 if msg.get("isError") else 0 + row = [sf, seq, entry_id, call_id, tool_name, raw_name, ts, text, 0, is_error] + rec.tool_results.append(row) + if text: + rec.pending_tokens.append((row, 8, text)) + if _is_edit_call(tool_name): + success = 1 if looks_successful(text) else 0 + warnings = extract_warnings(text) if success else [] + rec.edit_call_results.append( + (sf, call_id, success, json.dumps(warnings, ensure_ascii=False)) + ) + + +def _ingest_user(rec, path, seq, entry_id, ts, content) -> None: + text = join_text(content) + if not text: + return + row = [str(path), seq, entry_id, ts, text, 0] + rec.user_msgs.append(row) + rec.pending_tokens.append((row, 5, text)) + + +# --------------------------------------------------------------------------- # +# DB + +def open_db() -> sqlite3.Connection: + DB_PATH.parent.mkdir(parents=True, exist_ok=True) + conn = sqlite3.connect(DB_PATH, isolation_level=None, check_same_thread=False) + conn.execute("PRAGMA journal_mode = WAL") + conn.execute("PRAGMA synchronous = NORMAL") + conn.execute("PRAGMA temp_store = MEMORY") + conn.execute("PRAGMA mmap_size = 268435456") # 256 MiB + conn.executescript(SCHEMA_SQL) + _migrate(conn) + return conn + + +def existing_state(conn: sqlite3.Connection) -> dict[str, tuple[int, int, int, int, int]]: + """{session_file: (mtime, size, byte_offset, line_count, parser_version)}""" + rows = conn.execute( + "SELECT session_file, mtime, size, byte_offset, line_count, parser_version " + "FROM ss_sessions" + ).fetchall() + return {r[0]: (r[1], r[2], r[3], r[4], r[5]) for r in rows} + + +def write_records(conn: sqlite3.Connection, rec: SessionRecords, now_ms: int) -> None: + sf = rec.session_meta["session_file"] + cur = conn.cursor() + cur.execute("BEGIN IMMEDIATE") + try: + if rec.full_rebuild: + for tbl in ( + "ss_tool_calls", "ss_tool_results", + "ss_assistant_msgs", "ss_user_msgs", + "ss_edit_calls", "ss_edit_sections", + ): + cur.execute(f"DELETE FROM {tbl} WHERE session_file = ?", (sf,)) + + if rec.tool_calls: + cur.executemany( + "INSERT OR REPLACE INTO ss_tool_calls " + "(session_file, seq, entry_id, call_id, tool_name, raw_tool_name, " + " timestamp, model, provider, arg_json, arg_tokens) " + "VALUES (?,?,?,?,?,?,?,?,?,?,?)", + rec.tool_calls, + ) + if rec.tool_results: + cur.executemany( + "INSERT OR REPLACE INTO ss_tool_results " + "(session_file, seq, entry_id, call_id, tool_name, raw_tool_name, " + " timestamp, result_text, result_tokens, is_error) " + "VALUES (?,?,?,?,?,?,?,?,?,?)", + rec.tool_results, + ) + if rec.assistant_msgs: + cur.executemany( + "INSERT OR REPLACE INTO ss_assistant_msgs " + "(session_file, seq, entry_id, timestamp, model, provider, " + " text_blob, thinking_blob, text_tokens, thinking_tokens) " + "VALUES (?,?,?,?,?,?,?,?,?,?)", + rec.assistant_msgs, + ) + if rec.user_msgs: + cur.executemany( + "INSERT OR REPLACE INTO ss_user_msgs " + "(session_file, seq, entry_id, timestamp, text_blob, text_tokens) " + "VALUES (?,?,?,?,?,?)", + rec.user_msgs, + ) + if rec.edit_calls: + # Stub row when seeing toolCall; preserve any existing success/warnings + # if a prior sync already paired the result. + cur.executemany( + "INSERT INTO ss_edit_calls " + "(session_file, call_id, seq, timestamp, raw_input_len, parser_version) " + "VALUES (?,?,?,?,?,?) " + "ON CONFLICT(session_file, call_id) DO UPDATE SET " + " seq=excluded.seq, timestamp=excluded.timestamp, " + " raw_input_len=excluded.raw_input_len, " + " parser_version=excluded.parser_version", + rec.edit_calls, + ) + if rec.edit_call_results: + cur.executemany( + "INSERT INTO ss_edit_calls " + "(session_file, call_id, seq, timestamp, raw_input_len, success, warnings, parser_version) " + "VALUES (?,?,0,0,0,?,?,?) " + "ON CONFLICT(session_file, call_id) DO UPDATE SET " + " success=excluded.success, warnings=excluded.warnings, " + " parser_version=excluded.parser_version", + [(sf_, cid, succ, warn, EDIT_PARSER_VERSION) + for (sf_, cid, succ, warn) in rec.edit_call_results], + ) + if rec.edit_sections: + cur.executemany( + "INSERT OR REPLACE INTO ss_edit_sections " + "(session_file, call_id, seq, section_idx, target_file, " + " op_count, deleted_lines, payload_count, change_size, " + " min_line, max_line, payload_blocks, op_anchors, " + " longest_repeat_len, longest_repeat_block_idx, longest_repeat_sample, " + " dup_anchors, parser_version) " + "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)", + rec.edit_sections, + ) + + m = rec.session_meta + cur.execute( + "INSERT INTO ss_sessions " + "(session_file, folder, is_subagent, parent_session, subagent_label, " + " started_at, title, cwd, session_uuid, version, " + " mtime, size, byte_offset, line_count, last_synced, tokenizer, " + " schema_version, parser_version) " + "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?) " + "ON CONFLICT(session_file) DO UPDATE SET " + " folder=excluded.folder, " + " is_subagent=excluded.is_subagent, " + " parent_session=excluded.parent_session, " + " subagent_label=excluded.subagent_label, " + " started_at=COALESCE(excluded.started_at, ss_sessions.started_at), " + " title=COALESCE(excluded.title, ss_sessions.title), " + " cwd=COALESCE(excluded.cwd, ss_sessions.cwd), " + " session_uuid=COALESCE(excluded.session_uuid, ss_sessions.session_uuid), " + " version=COALESCE(excluded.version, ss_sessions.version), " + " mtime=excluded.mtime, " + " size=excluded.size, " + " byte_offset=excluded.byte_offset, " + " line_count=excluded.line_count, " + " last_synced=excluded.last_synced, " + " tokenizer=excluded.tokenizer, " + " schema_version=excluded.schema_version, " + " parser_version=excluded.parser_version", + ( + sf, m["folder"], m["is_subagent"], m["parent_session"], m["subagent_label"], + m["started_at"], m["title"], m["cwd"], m["session_uuid"], m["version"], + rec.file_mtime, rec.file_size, rec.final_offset, rec.final_line_count, + now_ms, TOKENIZER_NAME, SCHEMA_VERSION, EDIT_PARSER_VERSION, + ), + ) + cur.execute("COMMIT") + except Exception: + cur.execute("ROLLBACK") + raise + + +# --------------------------------------------------------------------------- # +# Driver + +def discover_sessions(root: Path, limit: int | None) -> list[Path]: + if not root.exists(): + return [] + files = [p for p in root.rglob("*.jsonl") if p.is_file()] + files.sort(key=lambda p: p.stat().st_mtime, reverse=True) + if limit and limit > 0: + files = files[:limit] + return files + + +def decide_action( + path: Path, + state: dict[str, tuple[int, int, int, int, int]], + full: bool, +) -> tuple[bool, int, int] | None: + """Returns (full_rebuild, starting_offset, starting_seq) or None to skip.""" + try: + st = path.stat() + except FileNotFoundError: + return None + mtime_ms = int(st.st_mtime * 1000) + size = st.st_size + prev = state.get(str(path)) + if full or prev is None: + return (True, 0, 0) + prev_mtime, prev_size, prev_offset, prev_lines, prev_parser = prev + if prev_parser < EDIT_PARSER_VERSION: + # Stale parser output → rebuild this file from scratch. + return (True, 0, 0) + if size == prev_size and mtime_ms <= prev_mtime: + return None + if size < prev_offset: + return (True, 0, 0) + if size == prev_offset and mtime_ms > prev_mtime: + return (True, 0, 0) + return (False, prev_offset, prev_lines) + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--workers", type=int, default=min(16, (os.cpu_count() or 4) * 2)) + ap.add_argument("--limit", type=int, default=0, + help="only sync the N most-recent files (0 = all)") + ap.add_argument("--full", action="store_true", + help="ignore stored state, re-ingest every file from scratch") + ap.add_argument("--root", default=str(SESSIONS_ROOT)) + args = ap.parse_args() + + root = Path(args.root).expanduser() + print(f"-> sessions root: {root}", file=sys.stderr) + print(f"-> db: {DB_PATH}", file=sys.stderr) + print(f"-> parser_version={EDIT_PARSER_VERSION} schema_version={SCHEMA_VERSION}", + file=sys.stderr) + + conn = open_db() + state = existing_state(conn) + print(f"-> known sessions in db: {len(state)}", file=sys.stderr) + + files = discover_sessions(root, args.limit or None) + print(f"-> on-disk sessions: {len(files)}", file=sys.stderr) + + work: list[tuple[Path, bool, int, int]] = [] + for p in files: + decision = decide_action(p, state, args.full) + if decision is None: + continue + full_rebuild, start_off, start_seq = decision + work.append((p, full_rebuild, start_off, start_seq)) + + print(f"-> dirty: {len(work)}", file=sys.stderr) + if not work: + return 0 + + out_q: queue.Queue[SessionRecords | None] = queue.Queue(maxsize=args.workers * 2) + + def writer_loop() -> None: + now_ms = int(time.time() * 1000) + n = 0 + t0 = time.monotonic() + last_log = t0 + while True: + rec = out_q.get() + if rec is None: + break + try: + write_records(conn, rec, now_ms) + except Exception as e: + print(f"!! write failed for {rec.session_meta['session_file']}: {e}", + file=sys.stderr) + n += 1 + now = time.monotonic() + if now - last_log >= 1.0: + rate = n / max(now - t0, 1e-6) + print(f" wrote {n}/{len(work)} files ({rate:.1f} files/s)", + file=sys.stderr) + last_log = now + rate = n / max(time.monotonic() - t0, 1e-6) + print(f"-> wrote {n} files total ({rate:.1f} files/s)", file=sys.stderr) + + writer_thread = threading.Thread(target=writer_loop, daemon=True) + writer_thread.start() + + t0 = time.monotonic() + with ProcessPoolExecutor(max_workers=args.workers) as ex: + futures = [ex.submit(_parse_worker, item) for item in work] + for fut in as_completed(futures): + try: + rec = fut.result() + except Exception as e: + print(f"!! parse failed: {e}", file=sys.stderr) + continue + if rec is not None: + out_q.put(rec) + + out_q.put(None) + writer_thread.join() + print(f"-> total wallclock: {time.monotonic() - t0:.1f}s", file=sys.stderr) + + conn.execute("PRAGMA optimize") + conn.close() + return 0 + + +if __name__ == "__main__": + sys.exit(main())