feat(scripts/session-stats): added session-stats edit/tools subcommands

This commit is contained in:
can1357
2026-04-30 01:54:05 +02:00
parent 863560ffb6
commit c39f7e0e20
12 changed files with 1458 additions and 1111 deletions
+4
View File
@@ -58,3 +58,7 @@ pi-*.html
packages/coding-agent/src/internal-urls/docs-index.generated.ts
/runs/
python/omp-rpc/src/omp_rpc.egg-info/
scripts/session-stats/Cargo.lock
scripts/session-stats/edit-analysis.csv
+3
View File
@@ -117,6 +117,9 @@
"ci:release:publish": "bun scripts/ci-release-publish.ts",
"bench:gen-fixtures": "bun --cwd=packages/typescript-edit-benchmark run src/generate.ts --typescript-dir /tmp/typescript-source --count-per-type 8",
"bench:edit": "bun --cwd=packages/typescript-edit-benchmark run start",
"stats:run": "cargo run --release --manifest-path scripts/session-stats/Cargo.toml --",
"stats:edits": "cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits",
"stats:tools": "cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools",
"prepublishOnly": "bun run check",
"prepare": "bun --cwd=packages/coding-agent run generate-docs-index",
"publish": "bun run prepublishOnly && npm publish -ws --access public",
-48
View File
@@ -1,48 +0,0 @@
# analyze-edit-formats
Audits how agents have used the `edit` / `ast_edit` / `write` tools across
historical session JSONLs in `~/.omp/agent/sessions/`.
For each tool call we:
- detect the **argument-schema family** in use (the edit tool has shipped many
shapes over time: `oldText/newText`, `op+pos+end+lines`, `loc+content`,
`loc+splice/pre/post/sed`, etc.);
- record the locator shape and verb combination (for the current
`loc+splice/pre/post/sed` schema);
- pair the call with its `toolResult` and classify the outcome
(`success` / `truncated` / `aborted` / `fail:anchor-stale` /
`fail:no-match` / `fail:parse` / `fail:no-enclosing-block` / …).
Output is a markdown-ish report on stdout plus per-call CSV at
`/tmp/edit-analysis/edits.csv` (or your CWD if you set that up differently).
## Usage
```sh
# Scan every session jsonl on disk (slow — ~25k files).
go run ./scripts/analyze-edit-formats
# Scan only files whose path contains the given date prefix(es).
go run ./scripts/analyze-edit-formats 2026-04-28
go run ./scripts/analyze-edit-formats 2026-04-27 2026-04-28
```
The walk root is `~/.omp/agent/sessions/`. Sub-session files (subagent
trajectories nested under `<session-id>/<n>-<name>.jsonl`) are picked up
automatically.
## Why Go
The session corpus is large (>25k files, >100k edit calls). Go iterates the
JSONL stream with negligible memory overhead and finishes in ~90s. The same
analysis in Bun/TS works but is noticeably slower for ad-hoc runs.
## What it's good for
- Comparing reliability across edit-tool argument schemas before changing the
current one.
- Spotting which verb / locator shapes have outsized failure rates so the
prompt can warn against them.
- Sanity-checking that a new edit-tool design isn't regressing the
failure-mode mix versus the previous design.
-3
View File
@@ -1,3 +0,0 @@
module github.com/oh-my-pi/scripts/analyze-edit-formats
go 1.26.2
-651
View File
@@ -1,651 +0,0 @@
// Analyzes how agents use the `edit` tool across today's session jsonl files
// in ~/.omp/agent/sessions/.
//
// For every edit-family toolCall (edit, ast_edit, write) we:
// - record what shape of arguments was supplied (loc, splice/pre/post/sed,
// bracket form of locator, line vs file targeted, etc.)
// - pair it with its toolResult and classify the result as success or as a
// specific failure category (anchor stale, anchor unknown, no enclosing
// block, parse error, ssr no match, etc.).
//
// Output is a markdown-ish report on stdout plus a CSV of every edit attempt
// to ./edit-analysis.csv (override with $EDIT_ANALYSIS_CSV).
package main
import (
"bufio"
"encoding/csv"
"encoding/json"
"fmt"
"os"
"path/filepath"
"regexp"
"sort"
"strings"
)
type rawMessage struct {
Type string `json:"type"`
Message json.RawMessage `json:"message"`
}
type message struct {
Role string `json:"role"`
Content json.RawMessage `json:"content"`
ToolName string `json:"toolName"`
ToolCallID string `json:"toolCallId"`
}
type contentItem struct {
Type string `json:"type"`
Text string `json:"text"`
Name string `json:"name"`
ID string `json:"id"`
Arguments json.RawMessage `json:"arguments"`
}
type editEntry struct {
File string
CallID string
ToolName string
NumEdits int
Verbs []string // splice/pre/post/sed per sub-edit
LocShapes []string // bare/bracket-(/bracket-[/bracket-tail/bracket-head/$pre/$post/$sed
HasNewFile bool
HasGlob bool
HasOps bool
Format string // edit-tool argument schema family
ResultRaw string
Status string // "success" or failure category
}
// matchDate accepts files whose path contains any of the supplied date
// prefixes (e.g. "2026-04-28"). With no flags it accepts every .jsonl.
var dateFilters []string
func matchDate(p string) bool {
if len(dateFilters) == 0 {
return true
}
for _, d := range dateFilters {
if strings.Contains(p, d) {
return true
}
}
return false
}
func main() {
dateFilters = os.Args[1:]
root, err := os.UserHomeDir()
must(err)
base := filepath.Join(root, ".omp", "agent", "sessions")
var files []string
must(filepath.Walk(base, func(p string, info os.FileInfo, err error) error {
if err != nil {
return nil
}
if info.IsDir() {
return nil
}
if !strings.HasSuffix(p, ".jsonl") {
return nil
}
if !matchDate(p) {
return nil
}
files = append(files, p)
return nil
}))
sort.Strings(files)
fmt.Fprintf(os.Stderr, "loaded %d session files for today\n", len(files))
var entries []editEntry
for _, f := range files {
entries = append(entries, processFile(f)...)
}
report(entries)
writeCSV(entries)
}
func processFile(path string) []editEntry {
fh, err := os.Open(path)
if err != nil {
fmt.Fprintln(os.Stderr, "open:", err)
return nil
}
defer fh.Close()
calls := map[string]*editEntry{}
var order []string
sc := bufio.NewScanner(fh)
sc.Buffer(make([]byte, 0, 64*1024), 64*1024*1024)
for sc.Scan() {
var rm rawMessage
if err := json.Unmarshal(sc.Bytes(), &rm); err != nil {
continue
}
if rm.Type != "message" {
continue
}
var m message
if err := json.Unmarshal(rm.Message, &m); err != nil {
continue
}
var items []contentItem
if err := json.Unmarshal(m.Content, &items); err != nil {
continue
}
switch m.Role {
case "assistant":
for _, it := range items {
if it.Type != "toolCall" {
continue
}
if !isEditTool(it.Name) {
continue
}
e := classifyArgs(it.Name, it.Arguments)
e.File = path
e.CallID = it.ID
e.ToolName = it.Name
calls[it.ID] = &e
order = append(order, it.ID)
}
case "toolResult":
if !isEditTool(m.ToolName) {
continue
}
e, ok := calls[m.ToolCallID]
if !ok {
// orphan result, skip
continue
}
text := joinText(items)
e.ResultRaw = text
e.Status = classifyResult(m.ToolName, text)
}
}
out := make([]editEntry, 0, len(order))
for _, id := range order {
if e, ok := calls[id]; ok {
out = append(out, *e)
}
}
return out
}
func isEditTool(name string) bool {
switch strings.ToLower(name) {
case "edit", "ast_edit", "write":
return true
}
return false
}
func joinText(items []contentItem) string {
var b strings.Builder
for _, it := range items {
if it.Type == "text" {
b.WriteString(it.Text)
}
}
return b.String()
}
// ---- argument classification ----
type editOp struct {
Loc string `json:"loc"`
Splice json.RawMessage `json:"splice"`
Pre json.RawMessage `json:"pre"`
Post json.RawMessage `json:"post"`
Sed json.RawMessage `json:"sed"`
}
type editArgs struct {
Path string `json:"path"`
Edits []editOp `json:"edits"`
// ast_edit
Ops []json.RawMessage `json:"ops"`
// Write
Content *string `json:"content,omitempty"`
}
var anchorBare = regexp.MustCompile(`^[a-zA-Z]?\d+[a-z]{2}$`)
var anchorWithFile = regexp.MustCompile(`^[^:]+:\d+[a-z]{2}$`)
func classifyArgs(name string, raw json.RawMessage) editEntry {
e := editEntry{}
e.Format = detectFormat(name, raw)
switch strings.ToLower(name) {
case "edit":
var a editArgs
_ = json.Unmarshal(raw, &a)
e.NumEdits = len(a.Edits)
for _, op := range a.Edits {
e.LocShapes = append(e.LocShapes, locShape(op.Loc))
verbs := []string{}
if !isNullOrEmpty(op.Splice) {
verbs = append(verbs, "splice")
}
if !isNullOrEmpty(op.Pre) {
verbs = append(verbs, "pre")
}
if !isNullOrEmpty(op.Post) {
verbs = append(verbs, "post")
}
if !isNullOrEmpty(op.Sed) {
verbs = append(verbs, "sed")
}
if len(verbs) == 0 {
verbs = append(verbs, "none")
}
e.Verbs = append(e.Verbs, strings.Join(verbs, "+"))
}
case "ast_edit":
var a editArgs
_ = json.Unmarshal(raw, &a)
e.HasOps = len(a.Ops) > 0
e.NumEdits = len(a.Ops)
if strings.ContainsAny(a.Path, "*?,") {
e.HasGlob = true
}
case "write":
e.HasNewFile = true
e.NumEdits = 1
e.Verbs = []string{"write"}
}
return e
}
// detectFormat figures out which edit-tool argument schema is in use by
// looking at the top-level argument keys and (for `edit`) the keys of the
// first sub-edit. Older sessions used many incompatible schemas.
func detectFormat(name string, raw json.RawMessage) string {
switch strings.ToLower(name) {
case "write":
return "write"
case "ast_edit":
return "ast_edit"
}
var top map[string]json.RawMessage
if err := json.Unmarshal(raw, &top); err != nil {
return "unknown"
}
has := func(k string) bool { _, ok := top[k]; return ok }
switch {
case has("oldText") && has("newText"):
return "oldText/newText"
case has("old_text") && has("new_text"):
return "old_text/new_text"
case has("diff") && has("op"):
return "diff+op"
case has("diff") && has("operation"):
return "diff+operation"
case has("diff"):
return "diff"
case has("replace") || has("insert"):
return "replace/insert"
}
if edits, ok := top["edits"]; ok {
var list []map[string]json.RawMessage
if err := json.Unmarshal(edits, &list); err == nil && len(list) > 0 {
first := list[0]
fh := func(k string) bool { _, ok := first[k]; return ok }
switch {
case fh("loc") && (fh("splice") || fh("pre") || fh("post") || fh("sed")):
return "loc+splice/pre/post/sed"
case fh("loc") && fh("content"):
return "loc+content"
case fh("set_line"):
return "set_line"
case fh("insert_after"):
return "insert_after"
case fh("op") && fh("pos") && fh("end") && fh("lines"):
return "op+pos+end+lines"
case fh("op") && fh("pos") && fh("lines"):
return "op+pos+lines"
case fh("op") && fh("sel") && fh("content"):
return "op+sel+content"
case fh("all") && (fh("new_text") || fh("old_text")):
return "per-edit:old_text/new_text"
}
keys := make([]string, 0, len(first))
for k := range first {
keys = append(keys, k)
}
sort.Strings(keys)
return "edits[" + strings.Join(keys, ",") + "]"
}
}
keys := make([]string, 0, len(top))
for k := range top {
keys = append(keys, k)
}
sort.Strings(keys)
return strings.Join(keys, ",")
}
func isNullOrEmpty(b json.RawMessage) bool {
s := strings.TrimSpace(string(b))
return s == "" || s == "null"
}
func locShape(loc string) string {
if loc == "" {
return "empty"
}
if loc == "$" {
return "$file"
}
// strip optional file: prefix
rest := loc
if i := strings.LastIndex(loc, ":"); i >= 0 && !strings.HasPrefix(loc, "$") {
rest = loc[i+1:]
}
switch {
case strings.HasPrefix(rest, "(") && strings.HasSuffix(rest, ")"):
return "bracket-(body)"
case strings.HasPrefix(rest, "[") && strings.HasSuffix(rest, "]"):
return "bracket-[block]"
case strings.HasPrefix(rest, "(") || strings.HasPrefix(rest, "["):
return "bracket-tail"
case strings.HasSuffix(rest, ")") || strings.HasSuffix(rest, "]"):
return "bracket-head"
case anchorBare.MatchString(rest):
return "bare-anchor"
}
return "other"
}
// ---- result classification ----
var (
reAnchorStale = regexp.MustCompile(`(?i)(Edit rejected:.*line[s]? .* changed since the last read|line[s]? ha(s|ve) changed since last read)`)
reAnchorMissing = regexp.MustCompile(`(?i)anchor .* (not found|unknown|missing)|loc requires the full anchor`)
reNoEnclosing = regexp.MustCompile(`(?i)No enclosing .* block`)
reParseError = regexp.MustCompile(`(?i)parse|syntax error|unbalanced|unexpected token`)
reSSRNoMatch = regexp.MustCompile(`(?i)0 matches|no replacements|no match found|No replacements made|Failed to find expected lines`)
reFileNotRead = regexp.MustCompile(`(?i)must be read first|has not been read|not yet read`)
reFileChanged = regexp.MustCompile(`(?i)file has been (modified|changed) externally`)
rePermDenied = regexp.MustCompile(`(?i)permission denied|not allowed`)
reGenericRejected = regexp.MustCompile(`(?i)\b(rejected|failed|error|invalid)\b`)
reTruncated = regexp.MustCompile(`(?i)\[Output truncated`)
reAborted = regexp.MustCompile(`(?i)Tool execution was aborted|Request was aborted|cancelled|canceled by user`)
reSuccess = regexp.MustCompile(`(?i)^(Updated|Successfully (wrote|replaced|edited|deleted|inserted)|Replaced|Applied|Deleted|Created|Wrote|edit applied|Edited|Inserted|OK\b)`)
)
func classifyResult(tool, text string) string {
t := strings.TrimSpace(text)
if t == "" {
return "empty"
}
first := strings.SplitN(t, "\n", 2)[0]
switch {
case reTruncated.MatchString(first):
return "truncated"
case reAborted.MatchString(t):
return "aborted"
case reSuccess.MatchString(first):
return "success"
case reAnchorStale.MatchString(t):
return "fail:anchor-stale"
case reNoEnclosing.MatchString(t):
return "fail:no-enclosing-block"
case reAnchorMissing.MatchString(t):
return "fail:anchor-missing"
case reParseError.MatchString(t):
return "fail:parse"
case reSSRNoMatch.MatchString(t):
return "fail:no-match"
case reFileNotRead.MatchString(t):
return "fail:file-not-read"
case reFileChanged.MatchString(t):
return "fail:file-changed"
case rePermDenied.MatchString(t):
return "fail:perm"
case reGenericRejected.MatchString(first):
return "fail:other"
}
return "unknown"
}
// ---- reporting ----
func report(entries []editEntry) {
if len(entries) == 0 {
fmt.Println("no edit-family tool calls found in today's sessions")
return
}
byTool := map[string]int{}
byFormat := map[string]int{}
statusByFormat := map[string]map[string]int{}
statusByTool := map[string]map[string]int{}
verbCount := map[string]int{}
locCount := map[string]int{}
failsByVerb := map[string]map[string]int{}
failsByLoc := map[string]map[string]int{}
for _, e := range entries {
byTool[e.ToolName]++
if statusByTool[e.ToolName] == nil {
statusByTool[e.ToolName] = map[string]int{}
}
statusByTool[e.ToolName][e.Status]++
byFormat[e.Format]++
if statusByFormat[e.Format] == nil {
statusByFormat[e.Format] = map[string]int{}
}
statusByFormat[e.Format][e.Status]++
for _, v := range e.Verbs {
verbCount[v]++
if failsByVerb[v] == nil {
failsByVerb[v] = map[string]int{}
}
failsByVerb[v][e.Status]++
}
for _, l := range e.LocShapes {
locCount[l]++
if failsByLoc[l] == nil {
failsByLoc[l] = map[string]int{}
}
failsByLoc[l][e.Status]++
}
}
fmt.Println("# Edit-tool usage in today's sessions")
fmt.Printf("\nTotal tool calls: %d (across %d sessions)\n",
len(entries), countSessions(entries))
fmt.Println("\n## By tool")
printSorted(byTool)
fmt.Println("\n## Outcome by tool")
tools := keys(byTool)
sort.Strings(tools)
for _, t := range tools {
fmt.Printf("\n %s (%d calls):\n", t, byTool[t])
printSortedIndent(statusByTool[t], " ")
}
fmt.Println("\n## edit verb distribution (per sub-edit)")
printSorted(verbCount)
fmt.Println("\n## edit locator shape distribution")
printSorted(locCount)
fmt.Println("\n## Failure rate per verb shape")
for _, v := range sortedKeys(verbCount) {
total, failed := 0, 0
for status, n := range failsByVerb[v] {
total += n
if strings.HasPrefix(status, "fail") {
failed += n
}
}
fmt.Printf(" %-20s %d/%d failed (%.0f%%)\n", v, failed, total, pct(failed, total))
}
fmt.Println("\n## Failure rate per locator shape")
for _, l := range sortedKeys(locCount) {
total, failed := 0, 0
for status, n := range failsByLoc[l] {
total += n
if strings.HasPrefix(status, "fail") {
failed += n
}
}
fmt.Printf(" %-20s %d/%d failed (%.0f%%)\n", l, failed, total, pct(failed, total))
}
fmt.Println("\n## edit-tool argument-format usage")
printSorted(byFormat)
fmt.Println("\n## Failure rate per argument format")
for _, fname := range sortedKeys(byFormat) {
total, failed := 0, 0
for status, n := range statusByFormat[fname] {
total += n
if strings.HasPrefix(status, "fail") {
failed += n
}
}
fmt.Printf(" %-32s %6d/%-6d failed (%.0f%%)\n", fname, failed, total, pct(failed, total))
}
fmt.Println("\n## Failure breakdown per top format")
cap := 0
for _, fname := range sortedKeys(byFormat) {
if cap >= 8 {
break
}
cap++
fmt.Printf("\n %s (%d total)\n", fname, byFormat[fname])
printSortedIndent(statusByFormat[fname], " ")
}
fmt.Println("\n## Sample failed edits")
shown := 0
for _, e := range entries {
if !strings.HasPrefix(e.Status, "fail") {
continue
}
fmt.Printf("\n— %s [%s] verbs=%v loc=%v\n result: %s\n",
e.ToolName, e.Status, e.Verbs, e.LocShapes,
truncate(strings.SplitN(e.ResultRaw, "\n\n", 2)[0], 220))
shown++
if shown >= 8 {
break
}
}
}
func countSessions(es []editEntry) int {
s := map[string]struct{}{}
for _, e := range es {
s[e.File] = struct{}{}
}
return len(s)
}
func printSorted(m map[string]int) {
for _, k := range sortedKeys(m) {
fmt.Printf(" %-25s %d\n", k, m[k])
}
}
func printSortedIndent(m map[string]int, indent string) {
for _, k := range sortedKeys(m) {
fmt.Printf("%s%-25s %d\n", indent, k, m[k])
}
}
func sortedKeys(m map[string]int) []string {
type kv struct {
k string
v int
}
pairs := make([]kv, 0, len(m))
for k, v := range m {
pairs = append(pairs, kv{k, v})
}
sort.Slice(pairs, func(i, j int) bool {
if pairs[i].v != pairs[j].v {
return pairs[i].v > pairs[j].v
}
return pairs[i].k < pairs[j].k
})
out := make([]string, len(pairs))
for i, p := range pairs {
out[i] = p.k
}
return out
}
func keys(m map[string]int) []string {
out := make([]string, 0, len(m))
for k := range m {
out = append(out, k)
}
return out
}
func pct(a, b int) float64 {
if b == 0 {
return 0
}
return 100 * float64(a) / float64(b)
}
func truncate(s string, n int) string {
s = strings.ReplaceAll(s, "\n", " | ")
if len(s) <= n {
return s
}
return s[:n] + "…"
}
func writeCSV(entries []editEntry) {
csvPath := os.Getenv("EDIT_ANALYSIS_CSV")
if csvPath == "" {
csvPath = "edit-analysis.csv"
}
f, err := os.Create(csvPath)
if err != nil {
fmt.Fprintln(os.Stderr, "csv:", err)
return
}
defer f.Close()
w := csv.NewWriter(f)
defer w.Flush()
_ = w.Write([]string{"session", "tool", "status", "num_edits", "verbs", "loc_shapes", "result_first_line"})
for _, e := range entries {
first := strings.SplitN(e.ResultRaw, "\n", 2)[0]
_ = w.Write([]string{
filepath.Base(e.File),
e.ToolName,
e.Status,
fmt.Sprintf("%d", e.NumEdits),
strings.Join(e.Verbs, ","),
strings.Join(e.LocShapes, ","),
truncate(first, 200),
})
}
}
func must(err error) {
if err != nil {
fmt.Fprintln(os.Stderr, "fatal:", err)
os.Exit(1)
}
}
-409
View File
@@ -1,409 +0,0 @@
#!/usr/bin/env bun
/**
* Dump and analyze edit tool attempts from session JSONL files.
*
* Usage:
* bun scripts/dump-edit-history.ts <session-file.jsonl> [options]
*
* Options:
* --failures Show only failed attempts
* --successes Show only successful attempts
* --json Output as JSON
* --stats Show statistics only
* --context Include thinking context before each attempt
* --compact Compact output (no diff content)
*/
import { Glob } from "bun";
import { basename } from "node:path";
// ═══════════════════════════════════════════════════════════════════════════
// Types
// ═══════════════════════════════════════════════════════════════════════════
interface Message {
type: string;
id?: string;
message?: {
role?: string;
content?: Array<{
type: string;
name?: string;
id?: string;
arguments?: Record<string, unknown>;
text?: string;
thinking?: string;
}>;
toolCallId?: string;
isError?: boolean;
};
}
interface EditAttempt {
id: string;
path: string;
op: string;
diff: string;
isError: boolean;
resultText: string;
errorType?: string;
thinkingContext?: string;
}
interface SessionResult {
file: string;
attempts: EditAttempt[];
}
// ═══════════════════════════════════════════════════════════════════════════
// Parsing
// ═══════════════════════════════════════════════════════════════════════════
function classifyError(resultText: string): string {
if (resultText.includes("Failed to find context")) return "context-not-found";
if (resultText.includes("matches for context")) return "ambiguous-context";
if (resultText.includes("Unexpected line in hunk")) return "parse-error";
if (resultText.includes("Failed to find expected lines")) return "lines-not-found";
if (resultText.includes("File not found")) return "file-not-found";
if (resultText.includes("occurrences")) return "ambiguous-match";
return "unknown";
}
async function extractEditAttempts(sessionPath: string): Promise<EditAttempt[]> {
const content = await Bun.file(sessionPath).bytes();
const messages = Bun.JSONL.parse(content) as Message[];
const editAttempts: EditAttempt[] = [];
for (let i = 0; i < messages.length; i++) {
const msg = messages[i];
if (msg.type !== "message") continue;
const msgContent = msg.message?.content;
if (!Array.isArray(msgContent)) continue;
// Extract thinking from this message
const thinking = msgContent.find((c) => c.type === "thinking")?.thinking;
for (const item of msgContent) {
if (item.type === "toolCall" && item.name === "edit") {
const toolId = item.id!;
const args = item.arguments as { path?: string; op?: string; diff?: string };
// Find result
let result: Message["message"] | null = null;
for (let j = i + 1; j < messages.length; j++) {
const resultMsg = messages[j];
if (resultMsg.type === "message" && resultMsg.message?.role === "toolResult") {
if (resultMsg.message.toolCallId === toolId) {
result = resultMsg.message;
break;
}
}
}
const resultContent = result?.content;
const resultText =
Array.isArray(resultContent) && resultContent[0]?.type === "text"
? (resultContent[0].text ?? "")
: "";
const isError = result?.isError ?? false;
editAttempts.push({
id: toolId,
path: args.path ?? "",
op: args.op ?? "update",
diff: args.diff ?? "",
isError,
resultText,
errorType: isError ? classifyError(resultText) : undefined,
thinkingContext: thinking,
});
}
}
}
return editAttempts;
}
// ═══════════════════════════════════════════════════════════════════════════
// Formatting
// ═══════════════════════════════════════════════════════════════════════════
const colors = {
reset: "\x1b[0m",
bold: "\x1b[1m",
dim: "\x1b[2m",
red: "\x1b[31m",
green: "\x1b[32m",
yellow: "\x1b[33m",
blue: "\x1b[34m",
magenta: "\x1b[35m",
cyan: "\x1b[36m",
};
function colorize(text: string, color: keyof typeof colors): string {
return `${colors[color]}${text}${colors.reset}`;
}
function formatDiff(diff: string): string {
return diff
.split("\n")
.map((line) => {
if (line.startsWith("+")) return colorize(line, "green");
if (line.startsWith("-")) return colorize(line, "red");
if (line.startsWith("@@")) return colorize(line, "cyan");
return colorize(line, "dim");
})
.join("\n");
}
function formatAttempt(attempt: EditAttempt, index: number, options: Options): string {
const status = attempt.isError
? `${colorize("✗ FAILED", "red")} ${colorize(`[${attempt.errorType}]`, "yellow")}`
: colorize("✓ SUCCESS", "green");
const lines: string[] = [
``,
`${colorize(`### Attempt ${index}`, "bold")}: ${status}`,
`${colorize("Path:", "dim")} ${attempt.path}`,
`${colorize("Operation:", "dim")} ${attempt.op}`,
];
if (options.context && attempt.thinkingContext) {
const truncated =
attempt.thinkingContext.length > 300
? `${attempt.thinkingContext.slice(0, 300)}…`
: attempt.thinkingContext;
lines.push(`${colorize("Thinking:", "dim")} ${truncated}`);
}
if (!options.compact) {
lines.push(`${colorize("Diff:", "dim")}`);
lines.push(formatDiff(attempt.diff));
}
lines.push(``);
const resultPreview = attempt.resultText.slice(0, 200);
const truncatedResult = attempt.resultText.length > 200 ? `${resultPreview}…` : resultPreview;
lines.push(`${colorize("Result:", "dim")} ${truncatedResult}`);
lines.push(colorize("-".repeat(80), "dim"));
return lines.join("\n");
}
function formatStats(results: SessionResult[]): string {
const allAttempts = results.flatMap((r) => r.attempts);
const failed = allAttempts.filter((a) => a.isError);
const succeeded = allAttempts.filter((a) => !a.isError);
// Group failures by error type
const errorGroups: Record<string, EditAttempt[]> = {};
for (const attempt of failed) {
const type = attempt.errorType ?? "unknown";
if (!errorGroups[type]) errorGroups[type] = [];
errorGroups[type].push(attempt);
}
const lines: string[] = [
``,
colorize("═".repeat(60), "dim"),
colorize(" Statistics", "bold"),
colorize("═".repeat(60), "dim"),
``,
`Total attempts: ${colorize(String(allAttempts.length), "bold")}`,
` ${colorize("✓", "green")} Succeeded: ${succeeded.length}`,
` ${colorize("✗", "red")} Failed: ${failed.length}`,
``,
];
if (Object.keys(errorGroups).length > 0) {
lines.push(colorize("Failures by type:", "bold"));
for (const [type, attempts] of Object.entries(errorGroups).sort((a, b) => b[1].length - a[1].length)) {
lines.push(` ${colorize(type, "yellow")}: ${attempts.length}`);
// Show example paths
const uniquePaths = [...new Set(attempts.map((a) => a.path))].slice(0, 3);
for (const p of uniquePaths) {
lines.push(` ${colorize("→", "dim")} ${p}`);
}
}
}
// Show unique @@ contexts that failed
const failedContexts = failed
.map((a) => {
const match = a.diff.match(/^@@\s*(.+)$/m);
return match?.[1]?.trim();
})
.filter(Boolean);
if (failedContexts.length > 0) {
lines.push(``);
lines.push(colorize("Failed @@ contexts:", "bold"));
const uniqueContexts = [...new Set(failedContexts)].slice(0, 10);
for (const ctx of uniqueContexts) {
lines.push(` ${colorize("@@", "cyan")} ${ctx}`);
}
}
return lines.join("\n");
}
function formatJson(results: SessionResult[]): string {
return JSON.stringify(
results.map((r) => ({
file: r.file,
attempts: r.attempts.map((a) => ({
path: a.path,
op: a.op,
diff: a.diff,
isError: a.isError,
errorType: a.errorType,
result: a.resultText,
})),
})),
null,
2,
);
}
// ═══════════════════════════════════════════════════════════════════════════
// Main
// ═══════════════════════════════════════════════════════════════════════════
interface Options {
failures: boolean;
successes: boolean;
json: boolean;
stats: boolean;
context: boolean;
compact: boolean;
}
function parseArgs(): { paths: string[]; options: Options } {
const args = process.argv.slice(2);
const options: Options = {
failures: false,
successes: false,
json: false,
stats: false,
context: false,
compact: false,
};
const paths: string[] = [];
for (const arg of args) {
if (arg === "--failures") options.failures = true;
else if (arg === "--successes") options.successes = true;
else if (arg === "--json") options.json = true;
else if (arg === "--stats") options.stats = true;
else if (arg === "--context") options.context = true;
else if (arg === "--compact") options.compact = true;
else if (!arg.startsWith("-")) paths.push(arg);
}
return { paths, options };
}
async function expandGlobs(patterns: string[]): Promise<string[]> {
const files: string[] = [];
for (const pattern of patterns) {
if (pattern.includes("*")) {
const glob = new Glob(pattern);
for await (const file of glob.scan({ absolute: true })) {
files.push(file);
}
} else {
try {
await Bun.file(pattern).text();
files.push(pattern);
} catch (err) {
const error = err as NodeJS.ErrnoException;
if (typeof err === "object" && err !== null && "code" in err ) continue;
if (error.code === "EISDIR" || error.code === "EACCES" || error.code === "EPERM" || error.code === "ENOENT") continue;
throw err;
}
}
}
return files;
}
async function main() {
const { paths, options } = parseArgs();
if (paths.length === 0) {
console.error(`Usage: bun scripts/dump-edit-history.ts <session-file.jsonl> [options]
Options:
--failures Show only failed attempts
--successes Show only successful attempts
--json Output as JSON
--stats Show statistics only
--context Include thinking context before each attempt
--compact Compact output (no diff content)
Examples:
bun scripts/dump-edit-history.ts session.jsonl
bun scripts/dump-edit-history.ts ~/.omp/agent/sessions/**/*.jsonl --stats
bun scripts/dump-edit-history.ts session.jsonl --failures --compact`);
process.exit(1);
}
const files = await expandGlobs(paths);
if (files.length === 0) {
console.error("No matching files found");
process.exit(1);
}
const results: SessionResult[] = [];
for (const file of files) {
try {
let attempts = await extractEditAttempts(file);
// Filter
if (options.failures) attempts = attempts.filter((a) => a.isError);
if (options.successes) attempts = attempts.filter((a) => !a.isError);
if (attempts.length > 0) {
results.push({ file, attempts });
}
} catch (e) {
console.error(`Error processing ${file}: ${e}`);
}
}
// Output
if (options.json) {
console.log(formatJson(results));
return;
}
if (options.stats) {
console.log(formatStats(results));
return;
}
for (const result of results) {
if (results.length > 1) {
console.log(`\n${colorize("═".repeat(80), "cyan")}`);
console.log(colorize(` ${basename(result.file)}`, "bold"));
console.log(colorize("═".repeat(80), "cyan"));
}
console.log(`Found ${result.attempts.length} edit attempt(s)`);
for (let i = 0; i < result.attempts.length; i++) {
console.log(formatAttempt(result.attempts[i], i + 1, options));
}
}
// Always show summary
const total = results.reduce((sum, r) => sum + r.attempts.length, 0);
const failed = results.reduce((sum, r) => sum + r.attempts.filter((a) => a.isError).length, 0);
console.log(`\n${colorize("Summary:", "bold")} ${total - failed} succeeded, ${failed} failed`);
}
main();
+29
View File
@@ -0,0 +1,29 @@
[package]
name = "session-stats"
version = "0.1.0"
edition = "2024"
publish = false
# Standalone crate: do not inherit from the parent workspace.
[workspace]
[[bin]]
name = "session-stats"
path = "src/main.rs"
[dependencies]
anyhow = "1"
csv = "1"
dirs = "5"
rayon = "1.10"
regex = "1"
serde = { version = "1", features = ["derive"] }
serde_json = { version = "1", features = ["raw_value"] }
tiktoken-rs = "0.7"
walkdir = "2"
[profile.release]
opt-level = 3
lto = "thin"
codegen-units = 1
strip = true
+82
View File
@@ -0,0 +1,82 @@
# session-stats
Ad-hoc analyses over the local agent session corpus
(`~/.omp/agent/sessions/`). Single Rust binary with subcommands.
## Subcommands
### `edits` — edit-tool reliability audit
Audits how agents have used the `edit` / `ast_edit` / `write` tools.
For each call we:
- detect the **argument-schema family** in use (the edit tool has shipped many
shapes over time: `oldText/newText`, `op+pos+end+lines`, `loc+content`,
`loc+splice/pre/post/sed`, etc.);
- record the locator shape and verb combination (for the current schema);
- pair the call with its `toolResult` and classify the outcome
(`success` / `truncated` / `aborted` / `fail:anchor-stale` /
`fail:no-match` / `fail:parse` / `fail:no-enclosing-block` / …).
Output: markdown-ish report on stdout plus per-call CSV at `$EDIT_ANALYSIS_CSV`
(default `./edit-analysis.csv`).
### `tools` — per-tool token budget
Aggregates token usage across the most-recent N sessions. Buckets:
- `tool ARGS` — assistant tool-call argument JSON
- `tool RESULTS` — tool result content text
- `assistant THINKING` — assistant `thinking` blocks
- `assistant TEXT` — assistant prose
- `user TEXT` — user-authored text content
Token counting uses **`o200k_base`** via `tiktoken-rs` (the GPT-4o / GPT-5
family BPE — well-defined offline and within ~5-10% of Claude's own counts in
aggregate across English/code).
Output: grand totals + per-tool breakdown sorted by total (arg+res) tokens.
Optional CSV at `$TOOL_USAGE_CSV`.
## Usage
```sh
# Edit audit on the most-recent sessions.
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits
# Edit audit on the 200 most-recent sessions.
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits -n 200
# Edit audit on a specific date.
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- edits 2026-04-28
# Tool token budget on the 1000 most-recent sessions.
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 1000
# Tool token budget on every jsonl on disk.
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 0
# Dump per-tool CSV alongside the report.
TOOL_USAGE_CSV=tools.csv \
cargo run --release --manifest-path scripts/session-stats/Cargo.toml -- tools -n 200
```
The walk root is `~/.omp/agent/sessions/`. Subagent jsonls
(`<session-id>/<n>-<name>.jsonl`) count as their own session and are included
in the recency window independently.
## Layout
```
scripts/session-stats/
Cargo.toml
src/
main.rs # subcommand dispatch
common.rs # shared JSONL shapes, walk, tokenizer, formatting helpers
cmd_edits.rs # edits subcommand
cmd_tools.rs # tools subcommand
```
The crate is a standalone Cargo project (it carries its own `[workspace]`
declaration) so it does not perturb the main workspace's lockfile.
+639
View File
@@ -0,0 +1,639 @@
//! `edits` subcommand — audits how agents have used the edit / ast_edit /
//! write tools across session jsonl files.
//!
//! For every edit-family toolCall we record:
//! - which argument-schema family is in use (the edit tool has shipped many
//! shapes over time: oldText/newText, op+pos+end+lines, loc+content,
//! loc+splice/pre/post/sed, etc.);
//! - the locator shape and verb combination (for the current
//! loc+splice/pre/post/sed schema);
//! then pair the call with its toolResult and classify success / failure
//! category (anchor-stale, no-match, parse, etc.).
//!
//! Output: markdown-ish report on stdout plus a per-call CSV at
//! `$EDIT_ANALYSIS_CSV` (default `./edit-analysis.csv`).
use crate::common::*;
use anyhow::{Context, Result, bail};
use regex::Regex;
use serde::Deserialize;
use serde_json::Value;
use serde_json::value::RawValue;
use std::collections::HashMap;
use std::fs::File;
use std::io::{BufRead, BufReader};
use std::path::Path;
use std::sync::LazyLock;
#[derive(Default, Clone)]
struct EditEntry {
file: String,
call_id: String,
tool_name: String,
num_edits: i64,
/// splice/pre/post/sed per sub-edit
verbs: Vec<String>,
/// bare-anchor / bracket-(body) / ...
loc_shapes: Vec<String>,
/// edit-tool argument schema family
format: String,
result_raw: String,
/// "success" / "fail:..." / etc.
status: String,
}
#[derive(Deserialize, Default)]
struct EditOp {
#[serde(default)]
loc: String,
#[serde(default)]
splice: Option<Box<RawValue>>,
#[serde(default)]
pre: Option<Box<RawValue>>,
#[serde(default)]
post: Option<Box<RawValue>>,
#[serde(default)]
sed: Option<Box<RawValue>>,
}
#[derive(Deserialize, Default)]
struct EditArgs {
#[serde(default)]
edits: Vec<EditOp>,
#[serde(default)]
ops: Vec<Box<RawValue>>,
}
pub fn run(args: Vec<String>) -> Result<()> {
let mut limit: usize = 100_000;
let mut workers: usize = 0;
let mut date_filters: Vec<String> = Vec::new();
let mut iter = args.into_iter();
while let Some(a) = iter.next() {
match a.as_str() {
"-n" => {
limit = iter
.next()
.context("-n requires a value")?
.parse()
.context("-n value")?;
}
"-j" => {
workers = iter
.next()
.context("-j requires a value")?
.parse()
.context("-j value")?;
}
"-h" | "--help" => {
eprintln!(
"usage: session-stats edits [-n N] [-j workers] [date prefix ...]"
);
return Ok(());
}
other if other.starts_with('-') => bail!("unknown flag: {other}"),
other => date_filters.push(other.to_string()),
}
}
let files = collect_sessions(&WalkOpts {
date_filters,
limit_most_recent: limit,
})?;
eprintln!("scanning {} session files", files.len());
let mut entries: Vec<EditEntry> = parallel_collect(&files, workers, 5_000, |p| {
Some(process_file(p))
})
.into_iter()
.flatten()
.collect();
// Stable ordering for sample output.
entries.sort_by(|a, b| a.file.cmp(&b.file));
report_edits(&entries);
write_csv(&entries)?;
Ok(())
}
fn process_file(path: &Path) -> Vec<EditEntry> {
let f = match File::open(path) {
Ok(f) => f,
Err(e) => {
eprintln!("open {}: {e}", path.display());
return Vec::new();
}
};
let reader = BufReader::with_capacity(64 * 1024, f);
let path_str = path.to_string_lossy().into_owned();
let mut calls: HashMap<String, EditEntry> = HashMap::new();
let mut order: Vec<String> = Vec::new();
for line in reader.lines() {
let Ok(line) = line else { continue };
if line.is_empty() {
continue;
}
let Ok(ev) = serde_json::from_str::<RawEvent>(&line) else {
continue;
};
if ev.kind != "message" {
continue;
}
let Some(msg_raw) = ev.message else { continue };
let Ok(m) = serde_json::from_str::<Message>(msg_raw.get()) else {
continue;
};
let Some(content_raw) = m.content else { continue };
let items = parse_content(&content_raw);
match m.role.as_str() {
"assistant" => {
for it in items {
if it.kind != "toolCall" || !is_edit_tool(&it.name) {
continue;
}
let raw = it.arguments.as_deref();
let mut e = classify_edit_args(&it.name, raw);
e.file.clone_from(&path_str);
e.call_id.clone_from(&it.id);
e.tool_name.clone_from(&it.name);
let id = it.id.clone();
calls.insert(id.clone(), e);
order.push(id);
}
}
"toolResult" => {
if !is_edit_tool(&m.tool_name) {
continue;
}
let Some(e) = calls.get_mut(&m.tool_call_id) else {
continue;
};
let text = join_text(&items);
e.status = classify_edit_result(&text);
e.result_raw = text;
}
_ => {}
}
}
let mut out: Vec<EditEntry> = Vec::with_capacity(order.len());
for id in order {
if let Some(e) = calls.remove(&id) {
out.push(e);
}
}
out
}
fn is_edit_tool(name: &str) -> bool {
matches!(
name.to_ascii_lowercase().as_str(),
"edit" | "ast_edit" | "write"
)
}
// ---- argument classification ----
static ANCHOR_BARE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"^[a-zA-Z]?[0-9]+[a-z]{2}$").expect("anchor_bare"));
fn classify_edit_args(name: &str, raw: Option<&RawValue>) -> EditEntry {
let mut e = EditEntry {
format: detect_edit_format(name, raw),
..EditEntry::default()
};
let lname = name.to_ascii_lowercase();
match lname.as_str() {
"edit" => {
let a: EditArgs = raw
.and_then(|r| serde_json::from_str(r.get()).ok())
.unwrap_or_default();
e.num_edits = a.edits.len() as i64;
for op in &a.edits {
e.loc_shapes.push(loc_shape(&op.loc));
let mut verbs: Vec<&str> = Vec::new();
if !is_null_or_empty(op.splice.as_deref()) {
verbs.push("splice");
}
if !is_null_or_empty(op.pre.as_deref()) {
verbs.push("pre");
}
if !is_null_or_empty(op.post.as_deref()) {
verbs.push("post");
}
if !is_null_or_empty(op.sed.as_deref()) {
verbs.push("sed");
}
if verbs.is_empty() {
verbs.push("none");
}
e.verbs.push(verbs.join("+"));
}
}
"ast_edit" => {
let a: EditArgs = raw
.and_then(|r| serde_json::from_str(r.get()).ok())
.unwrap_or_default();
e.num_edits = a.ops.len() as i64;
}
"write" => {
e.num_edits = 1;
e.verbs.push("write".to_string());
}
_ => {}
}
e
}
/// Looks at top-level argument keys (and the first sub-edit for the `edit`
/// tool) to identify which schema is in use. Older sessions used many
/// incompatible schemas.
fn detect_edit_format(name: &str, raw: Option<&RawValue>) -> String {
match name.to_ascii_lowercase().as_str() {
"write" => return "write".to_string(),
"ast_edit" => return "ast_edit".to_string(),
_ => {}
}
let Some(raw) = raw else {
return "unknown".to_string();
};
let top: HashMap<String, Value> = match serde_json::from_str(raw.get()) {
Ok(v) => v,
Err(_) => return "unknown".to_string(),
};
let has = |k: &str| top.contains_key(k);
if has("oldText") && has("newText") {
return "oldText/newText".to_string();
}
if has("old_text") && has("new_text") {
return "old_text/new_text".to_string();
}
if has("diff") && has("op") {
return "diff+op".to_string();
}
if has("diff") && has("operation") {
return "diff+operation".to_string();
}
if has("diff") {
return "diff".to_string();
}
if has("replace") || has("insert") {
return "replace/insert".to_string();
}
if let Some(edits_val) = top.get("edits")
&& let Some(arr) = edits_val.as_array()
&& let Some(first) = arr.first().and_then(Value::as_object)
{
let fh = |k: &str| first.contains_key(k);
if fh("loc") && (fh("splice") || fh("pre") || fh("post") || fh("sed")) {
return "loc+splice/pre/post/sed".to_string();
}
if fh("loc") && fh("content") {
return "loc+content".to_string();
}
if fh("set_line") {
return "set_line".to_string();
}
if fh("insert_after") {
return "insert_after".to_string();
}
if fh("op") && fh("pos") && fh("end") && fh("lines") {
return "op+pos+end+lines".to_string();
}
if fh("op") && fh("pos") && fh("lines") {
return "op+pos+lines".to_string();
}
if fh("op") && fh("sel") && fh("content") {
return "op+sel+content".to_string();
}
if fh("all") && (fh("new_text") || fh("old_text")) {
return "per-edit:old_text/new_text".to_string();
}
let mut keys: Vec<&str> = first.keys().map(String::as_str).collect();
keys.sort_unstable();
return format!("edits[{}]", keys.join(","));
}
let mut keys: Vec<&str> = top.keys().map(String::as_str).collect();
keys.sort_unstable();
keys.join(",")
}
fn is_null_or_empty(b: Option<&RawValue>) -> bool {
let Some(b) = b else { return true };
let s = b.get().trim();
s.is_empty() || s == "null"
}
fn loc_shape(loc: &str) -> String {
if loc.is_empty() {
return "empty".to_string();
}
if loc == "$" {
return "$file".to_string();
}
let rest = if let Some(i) = loc.rfind(':')
&& !loc.starts_with('$')
{
&loc[i + 1..]
} else {
loc
};
if rest.starts_with('(') && rest.ends_with(')') {
return "bracket-(body)".to_string();
}
if rest.starts_with('[') && rest.ends_with(']') {
return "bracket-[block]".to_string();
}
if rest.starts_with('(') || rest.starts_with('[') {
return "bracket-tail".to_string();
}
if rest.ends_with(')') || rest.ends_with(']') {
return "bracket-head".to_string();
}
if ANCHOR_BARE.is_match(rest) {
return "bare-anchor".to_string();
}
"other".to_string()
}
// ---- result classification ----
macro_rules! re {
($pat:expr) => {
LazyLock::new(|| Regex::new($pat).expect("compile result regex"))
};
}
static RE_ANCHOR_STALE: LazyLock<Regex> = re!(
r"(?i)(Edit rejected:.*line[s]? .* changed since the last read|line[s]? ha(s|ve) changed since last read)"
);
static RE_ANCHOR_MISSING: LazyLock<Regex> =
re!(r"(?i)anchor .* (not found|unknown|missing)|loc requires the full anchor");
static RE_NO_ENCLOSING: LazyLock<Regex> = re!(r"(?i)No enclosing .* block");
static RE_PARSE_ERROR: LazyLock<Regex> =
re!(r"(?i)parse|syntax error|unbalanced|unexpected token");
static RE_SSR_NO_MATCH: LazyLock<Regex> = re!(
r"(?i)0 matches|no replacements|no match found|No replacements made|Failed to find expected lines"
);
static RE_FILE_NOT_READ: LazyLock<Regex> =
re!(r"(?i)must be read first|has not been read|not yet read");
static RE_FILE_CHANGED: LazyLock<Regex> =
re!(r"(?i)file has been (modified|changed) externally");
static RE_PERM_DENIED: LazyLock<Regex> = re!(r"(?i)permission denied|not allowed");
static RE_GENERIC_REJECTED: LazyLock<Regex> =
re!(r"(?i)\b(rejected|failed|error|invalid)\b");
static RE_TRUNCATED: LazyLock<Regex> = re!(r"(?i)\[Output truncated");
static RE_ABORTED: LazyLock<Regex> = re!(
r"(?i)Tool execution was aborted|Request was aborted|cancelled|canceled by user"
);
static RE_SUCCESS: LazyLock<Regex> = re!(
r"(?i)^(Updated|Successfully (wrote|replaced|edited|deleted|inserted)|Replaced|Applied|Deleted|Created|Wrote|edit applied|Edited|Inserted|OK\b)"
);
fn classify_edit_result(text: &str) -> String {
let t = text.trim();
if t.is_empty() {
return "empty".to_string();
}
let first = t.split_once('\n').map_or(t, |(a, _)| a);
if RE_TRUNCATED.is_match(first) {
return "truncated".to_string();
}
if RE_ABORTED.is_match(t) {
return "aborted".to_string();
}
if RE_SUCCESS.is_match(first) {
return "success".to_string();
}
if RE_ANCHOR_STALE.is_match(t) {
return "fail:anchor-stale".to_string();
}
if RE_NO_ENCLOSING.is_match(t) {
return "fail:no-enclosing-block".to_string();
}
if RE_ANCHOR_MISSING.is_match(t) {
return "fail:anchor-missing".to_string();
}
if RE_PARSE_ERROR.is_match(t) {
return "fail:parse".to_string();
}
if RE_SSR_NO_MATCH.is_match(t) {
return "fail:no-match".to_string();
}
if RE_FILE_NOT_READ.is_match(t) {
return "fail:file-not-read".to_string();
}
if RE_FILE_CHANGED.is_match(t) {
return "fail:file-changed".to_string();
}
if RE_PERM_DENIED.is_match(t) {
return "fail:perm".to_string();
}
if RE_GENERIC_REJECTED.is_match(first) {
return "fail:other".to_string();
}
"unknown".to_string()
}
// ---- reporting ----
fn report_edits(entries: &[EditEntry]) {
if entries.is_empty() {
println!("no edit-family tool calls found in matched sessions");
return;
}
let mut by_tool: HashMap<String, i64> = HashMap::new();
let mut by_format: HashMap<String, i64> = HashMap::new();
let mut status_by_format: HashMap<String, HashMap<String, i64>> = HashMap::new();
let mut status_by_tool: HashMap<String, HashMap<String, i64>> = HashMap::new();
let mut verb_count: HashMap<String, i64> = HashMap::new();
let mut loc_count: HashMap<String, i64> = HashMap::new();
let mut fails_by_verb: HashMap<String, HashMap<String, i64>> = HashMap::new();
let mut fails_by_loc: HashMap<String, HashMap<String, i64>> = HashMap::new();
for e in entries {
*by_tool.entry(e.tool_name.clone()).or_insert(0) += 1;
*status_by_tool
.entry(e.tool_name.clone())
.or_default()
.entry(e.status.clone())
.or_insert(0) += 1;
*by_format.entry(e.format.clone()).or_insert(0) += 1;
*status_by_format
.entry(e.format.clone())
.or_default()
.entry(e.status.clone())
.or_insert(0) += 1;
for v in &e.verbs {
*verb_count.entry(v.clone()).or_insert(0) += 1;
*fails_by_verb
.entry(v.clone())
.or_default()
.entry(e.status.clone())
.or_insert(0) += 1;
}
for l in &e.loc_shapes {
*loc_count.entry(l.clone()).or_insert(0) += 1;
*fails_by_loc
.entry(l.clone())
.or_default()
.entry(e.status.clone())
.or_insert(0) += 1;
}
}
println!("# Edit-tool usage");
println!(
"\nTotal tool calls: {} (across {} sessions)",
entries.len(),
count_edit_sessions(entries)
);
println!("\n## By tool");
print_sorted(&by_tool);
println!("\n## Outcome by tool");
let mut tools: Vec<&String> = by_tool.keys().collect();
tools.sort();
for t in tools {
println!("\n {t} ({} calls):", by_tool[t.as_str()]);
if let Some(m) = status_by_tool.get(t.as_str()) {
print_sorted_indent(m, " ");
}
}
println!("\n## edit verb distribution (per sub-edit)");
print_sorted(&verb_count);
println!("\n## edit locator shape distribution");
print_sorted(&loc_count);
println!("\n## Failure rate per verb shape");
for v in sorted_by_count(&verb_count) {
let (total, failed) = fail_totals(fails_by_verb.get(v.as_str()));
println!(
" {v:<20} {failed}/{total} failed ({:.0}%)",
pct(failed, total)
);
}
println!("\n## Failure rate per locator shape");
for l in sorted_by_count(&loc_count) {
let (total, failed) = fail_totals(fails_by_loc.get(l.as_str()));
println!(
" {l:<20} {failed}/{total} failed ({:.0}%)",
pct(failed, total)
);
}
println!("\n## edit-tool argument-format usage");
print_sorted(&by_format);
println!("\n## Failure rate per argument format");
for fname in sorted_by_count(&by_format) {
let (total, failed) = fail_totals(status_by_format.get(fname.as_str()));
println!(
" {fname:<32} {failed:>6}/{total:<6} failed ({:.0}%)",
pct(failed, total)
);
}
println!("\n## Failure breakdown per top format");
for fname in sorted_by_count(&by_format).into_iter().take(8) {
println!("\n {fname} ({} total)", by_format[fname.as_str()]);
if let Some(m) = status_by_format.get(fname.as_str()) {
print_sorted_indent(m, " ");
}
}
println!("\n## Sample failed edits");
let mut shown = 0;
for e in entries {
if !e.status.starts_with("fail") {
continue;
}
let first = e
.result_raw
.split_once("\n\n")
.map_or(e.result_raw.as_str(), |(a, _)| a);
println!(
"\n— {} [{}] verbs={:?} loc={:?}\n result: {}",
e.tool_name,
e.status,
e.verbs,
e.loc_shapes,
truncate_line(first, 220)
);
shown += 1;
if shown >= 8 {
break;
}
}
}
fn fail_totals(m: Option<&HashMap<String, i64>>) -> (i64, i64) {
let Some(m) = m else { return (0, 0) };
let mut total = 0i64;
let mut failed = 0i64;
for (status, n) in m {
total += n;
if status.starts_with("fail") {
failed += n;
}
}
(total, failed)
}
fn count_edit_sessions(entries: &[EditEntry]) -> usize {
let mut s: std::collections::HashSet<&str> = std::collections::HashSet::new();
for e in entries {
s.insert(&e.file);
}
s.len()
}
fn write_csv(entries: &[EditEntry]) -> Result<()> {
let path = std::env::var("EDIT_ANALYSIS_CSV").unwrap_or_else(|_| "edit-analysis.csv".to_string());
let f = File::create(&path).with_context(|| format!("create {path}"))?;
let mut w = csv::Writer::from_writer(f);
w.write_record([
"session",
"tool",
"status",
"num_edits",
"verbs",
"loc_shapes",
"result_first_line",
])?;
for e in entries {
let first = e
.result_raw
.split_once('\n')
.map_or(e.result_raw.as_str(), |(a, _)| a);
let session = Path::new(&e.file)
.file_name()
.and_then(|s| s.to_str())
.unwrap_or(&e.file);
w.write_record([
session,
&e.tool_name,
&e.status,
&e.num_edits.to_string(),
&e.verbs.join(","),
&e.loc_shapes.join(","),
&truncate_line(first, 200),
])?;
}
w.flush()?;
Ok(())
}
+387
View File
@@ -0,0 +1,387 @@
//! `tools` subcommand — per-tool token totals across the most-recent N session
//! jsonl files.
//!
//! Token counting uses o200k_base via tiktoken-rs (the GPT-4o / GPT-5 family
//! tokenizer). It is not Claude's own BPE, but it is well-defined offline and
//! within ~5-10% across English/code in aggregate.
//!
//! Buckets:
//! tool ARGS — assistant tool-call argument JSON
//! tool RESULTS — tool result content text
//! assistant THINKING — assistant `thinking` blocks
//! assistant TEXT — assistant prose
//! user TEXT — user-authored text content
//!
//! Output: grand totals + per-tool breakdown sorted by total (arg+res) tokens.
//! Optional CSV at `$TOOL_USAGE_CSV`.
use crate::common::*;
use anyhow::{Context, Result, bail};
use std::collections::HashMap;
use std::fs::File;
use std::io::{BufRead, BufReader};
use std::path::Path;
#[derive(Default, Clone)]
struct ToolAgg {
calls: i64,
results: i64,
arg_tok: i64,
res_tok: i64,
}
#[derive(Default, Clone)]
struct SessionTotals {
arg_tok: i64,
res_tok: i64,
thinking_tok: i64,
text_tok: i64,
user_tok: i64,
n_calls: i64,
n_results: i64,
}
struct FileResult {
totals: SessionTotals,
tools: HashMap<String, ToolAgg>,
}
pub fn run(args: Vec<String>) -> Result<()> {
let mut limit: usize = 100_000;
let mut workers: usize = 0;
let mut iter = args.into_iter();
while let Some(a) = iter.next() {
match a.as_str() {
"-n" => {
limit = iter
.next()
.context("-n requires a value")?
.parse()
.context("-n value")?;
}
"-j" => {
workers = iter
.next()
.context("-j requires a value")?
.parse()
.context("-j value")?;
}
"-h" | "--help" => {
eprintln!(
"usage: session-stats tools [-n N] [-j workers]\n\
\n\
Aggregates per-tool token usage across the most-recent N session\n\
jsonl files (default 100000). Tokenizer: o200k_base."
);
return Ok(());
}
other => bail!("unknown flag: {other}"),
}
}
let files = collect_sessions(&WalkOpts {
date_filters: Vec::new(),
limit_most_recent: limit,
})?;
eprintln!(
"scanning {} session files (tokenizer: o200k_base)",
files.len()
);
let results = parallel_collect(&files, workers, 5_000, process_file);
let sessions = results.len();
let mut grand = SessionTotals::default();
let mut tools: HashMap<String, ToolAgg> = HashMap::new();
for r in results {
grand.arg_tok += r.totals.arg_tok;
grand.res_tok += r.totals.res_tok;
grand.thinking_tok += r.totals.thinking_tok;
grand.text_tok += r.totals.text_tok;
grand.user_tok += r.totals.user_tok;
grand.n_calls += r.totals.n_calls;
grand.n_results += r.totals.n_results;
for (name, t) in r.tools {
let dst = tools.entry(name).or_default();
dst.calls += t.calls;
dst.results += t.results;
dst.arg_tok += t.arg_tok;
dst.res_tok += t.res_tok;
}
}
print_grand(&grand, sessions);
println!();
print_table(&tools);
write_csv(&tools)?;
Ok(())
}
fn process_file(path: &Path) -> Option<FileResult> {
let f = match File::open(path) {
Ok(f) => f,
Err(e) => {
eprintln!("open {}: {e}", path.display());
return None;
}
};
let reader = BufReader::with_capacity(64 * 1024, f);
let mut totals = SessionTotals::default();
let mut tools: HashMap<String, ToolAgg> = HashMap::new();
// Pending arg attribution: when a result arrives we credit the tool listed
// here; otherwise we fall back to message.toolName on the result event.
let mut pending: HashMap<String, String> = HashMap::new();
for line in reader.lines() {
let Ok(line) = line else { continue };
if line.is_empty() {
continue;
}
let Ok(ev) = serde_json::from_str::<RawEvent>(&line) else {
continue;
};
if ev.kind != "message" {
continue;
}
let Some(msg_raw) = ev.message else { continue };
let Ok(m) = serde_json::from_str::<Message>(msg_raw.get()) else {
continue;
};
let Some(content_raw) = m.content else { continue };
let items = parse_content(&content_raw);
match m.role.as_str() {
"assistant" => {
for it in items {
match it.kind.as_str() {
"toolCall" => {
let name = normalize_tool(&it.name);
let args_str = it.arguments.as_deref().map(RawValue::get).unwrap_or("");
let tok = count_tokens(args_str) as i64;
totals.arg_tok += tok;
totals.n_calls += 1;
let t = tools.entry(name.clone()).or_default();
t.calls += 1;
t.arg_tok += tok;
pending.insert(it.id, name);
}
"thinking" => {
totals.thinking_tok += count_tokens(&it.thinking) as i64;
}
"text" => {
totals.text_tok += count_tokens(&it.text) as i64;
}
_ => {}
}
}
}
"toolResult" => {
let text = join_text(&items);
let tok = count_tokens(&text) as i64;
totals.res_tok += tok;
totals.n_results += 1;
let name = pending
.remove(&m.tool_call_id)
.unwrap_or_else(|| normalize_tool(&m.tool_name));
let t = tools.entry(name).or_default();
t.results += 1;
t.res_tok += tok;
}
"user" => {
for it in items {
if it.kind == "text" {
totals.user_tok += count_tokens(&it.text) as i64;
}
}
}
_ => {}
}
}
Some(FileResult { totals, tools })
}
use serde_json::value::RawValue;
fn normalize_tool(name: &str) -> String {
if name.is_empty() {
"<unknown>".to_string()
} else {
name.to_string()
}
}
// ---- reporting ----
fn print_grand(g: &SessionTotals, sessions: usize) {
let total = g.arg_tok + g.res_tok + g.thinking_tok + g.text_tok + g.user_tok;
let share = |n: i64| pct(n, total);
println!("=== Grand totals across {sessions} sessions ===");
println!(
"tool call ARGS: {:>10} tok ({:>5.1}%)",
commas(g.arg_tok),
share(g.arg_tok)
);
println!(
"tool RESULTS: {:>10} tok ({:>5.1}%)",
commas(g.res_tok),
share(g.res_tok)
);
println!(
"assistant THINKING: {:>10} tok ({:>5.1}%)",
commas(g.thinking_tok),
share(g.thinking_tok)
);
println!(
"assistant TEXT: {:>10} tok ({:>5.1}%)",
commas(g.text_tok),
share(g.text_tok)
);
println!(
"user TEXT: {:>10} tok ({:>5.1}%)",
commas(g.user_tok),
share(g.user_tok)
);
println!(" ---------------");
println!("TOTAL: {:>10} tok", commas(total));
println!();
println!(
"tool calls: {}, tool results: {}",
commas(g.n_calls),
commas(g.n_results)
);
if g.n_calls > 0 {
println!(
"avg arg tokens / call: {:.1}",
g.arg_tok as f64 / g.n_calls as f64
);
}
if g.n_results > 0 {
println!(
"avg result tokens / call: {:.1}",
g.res_tok as f64 / g.n_results as f64
);
}
if g.arg_tok > 0 {
println!(
"ratio result / arg: {:.2}x",
g.res_tok as f64 / g.arg_tok as f64
);
}
}
struct ToolRow {
name: String,
calls: i64,
arg_tok: i64,
res_tok: i64,
total: i64,
avg_arg: f64,
avg_res: f64,
res_o_arg: f64,
}
fn print_table(tools: &HashMap<String, ToolAgg>) {
let mut rows: Vec<ToolRow> = tools
.iter()
.filter_map(|(name, t)| {
if t.calls == 0 && t.results == 0 {
return None;
}
let mut r = ToolRow {
name: name.clone(),
calls: t.calls,
arg_tok: t.arg_tok,
res_tok: t.res_tok,
total: t.arg_tok + t.res_tok,
avg_arg: 0.0,
avg_res: 0.0,
res_o_arg: 0.0,
};
if t.calls > 0 {
r.avg_arg = t.arg_tok as f64 / t.calls as f64;
r.avg_res = t.res_tok as f64 / t.calls as f64;
}
if t.arg_tok > 0 {
r.res_o_arg = t.res_tok as f64 / t.arg_tok as f64;
}
Some(r)
})
.collect();
rows.sort_by(|a, b| b.total.cmp(&a.total));
println!(
"{:<22} {:>6} {:>10} {:>10} {:>10} {:>8} {:>8} {:>8}",
"tool", "calls", "arg_tok", "res_tok", "total", "avg_arg", "avg_res", "res/arg"
);
println!("{}", "-".repeat(100));
const TOP: usize = 25;
let shown = TOP.min(rows.len());
for r in &rows[..shown] {
println!(
"{:<22} {:>6} {:>10} {:>10} {:>10} {:>8.1} {:>8.1} {:>8.2}",
r.name,
commas(r.calls),
commas(r.arg_tok),
commas(r.res_tok),
commas(r.total),
r.avg_arg,
r.avg_res,
r.res_o_arg
);
}
if rows.len() > TOP {
let (mut sc, mut sa, mut sr) = (0i64, 0i64, 0i64);
for r in &rows[TOP..] {
sc += r.calls;
sa += r.arg_tok;
sr += r.res_tok;
}
println!(
"{:<22} {:>6} {:>10} {:>10} {:>10}",
format!("({} others)", rows.len() - TOP),
commas(sc),
commas(sa),
commas(sr),
commas(sa + sr),
);
}
}
fn write_csv(tools: &HashMap<String, ToolAgg>) -> Result<()> {
let path = std::env::var("TOOL_USAGE_CSV").unwrap_or_default();
if path.is_empty() {
return Ok(());
}
let f = File::create(&path).with_context(|| format!("create {path}"))?;
let mut w = csv::Writer::from_writer(f);
w.write_record(["tool", "calls", "results", "arg_tok", "res_tok", "total"])?;
let mut names: Vec<&String> = tools.keys().collect();
names.sort_by(|a, b| {
let ai = {
let t = &tools[a.as_str()];
t.arg_tok + t.res_tok
};
let aj = {
let t = &tools[b.as_str()];
t.arg_tok + t.res_tok
};
aj.cmp(&ai)
});
for n in names {
let t = &tools[n.as_str()];
w.write_record([
n.as_str(),
&t.calls.to_string(),
&t.results.to_string(),
&t.arg_tok.to_string(),
&t.res_tok.to_string(),
&(t.arg_tok + t.res_tok).to_string(),
])?;
}
w.flush()?;
Ok(())
}
+256
View File
@@ -0,0 +1,256 @@
//! Shared JSONL shapes, walk helpers, tokenizer, and formatting helpers.
use anyhow::{Context, Result};
use rayon::prelude::*;
use serde::Deserialize;
use serde_json::value::RawValue;
use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::sync::LazyLock;
use std::sync::atomic::{AtomicU64, Ordering};
use std::time::SystemTime;
use tiktoken_rs::CoreBPE;
use walkdir::WalkDir;
// ---- jsonl shapes ----
#[derive(Deserialize)]
pub struct RawEvent {
#[serde(rename = "type", default)]
pub kind: String,
#[serde(default)]
pub message: Option<Box<RawValue>>,
}
#[derive(Deserialize)]
pub struct Message {
#[serde(default)]
pub role: String,
#[serde(default)]
pub content: Option<Box<RawValue>>,
#[serde(default, rename = "toolName")]
pub tool_name: String,
#[serde(default, rename = "toolCallId")]
pub tool_call_id: String,
}
#[derive(Deserialize)]
pub struct ContentItem {
#[serde(rename = "type", default)]
pub kind: String,
#[serde(default)]
pub text: String,
#[serde(default)]
pub thinking: String,
#[serde(default)]
pub name: String,
#[serde(default)]
pub id: String,
#[serde(default)]
pub arguments: Option<Box<RawValue>>,
}
// ---- session walking ----
pub fn sessions_root() -> Result<PathBuf> {
let home = dirs::home_dir().context("could not resolve home directory")?;
Ok(home.join(".omp").join("agent").join("sessions"))
}
pub struct WalkOpts {
/// Keeps only paths containing any of these substrings (e.g. "2026-04-28").
/// Empty means accept all.
pub date_filters: Vec<String>,
/// Keeps only the N most-recently-modified files (after the date filter).
/// 0 means no limit.
pub limit_most_recent: usize,
}
/// Walks the sessions root and returns the matching `.jsonl` paths.
/// With `limit_most_recent > 0` the result is sorted by mtime descending and
/// truncated to N entries; otherwise it's lexically sorted.
pub fn collect_sessions(opts: &WalkOpts) -> Result<Vec<PathBuf>> {
let base = sessions_root()?;
let need_mtime = opts.limit_most_recent > 0;
let mut all: Vec<(PathBuf, SystemTime)> = Vec::new();
for entry in WalkDir::new(&base).into_iter().filter_map(Result::ok) {
if !entry.file_type().is_file() {
continue;
}
let p = entry.path();
if p.extension().and_then(|e| e.to_str()) != Some("jsonl") {
continue;
}
let path_str = p.to_string_lossy();
if !match_date(&path_str, &opts.date_filters) {
continue;
}
let mt = if need_mtime {
entry
.metadata()
.ok()
.and_then(|m| m.modified().ok())
.unwrap_or(SystemTime::UNIX_EPOCH)
} else {
SystemTime::UNIX_EPOCH
};
all.push((p.to_path_buf(), mt));
}
if need_mtime {
all.sort_by(|a, b| b.1.cmp(&a.1));
all.truncate(opts.limit_most_recent);
} else {
all.sort_by(|a, b| a.0.cmp(&b.0));
}
Ok(all.into_iter().map(|(p, _)| p).collect())
}
fn match_date(p: &str, filters: &[String]) -> bool {
filters.is_empty() || filters.iter().any(|d| p.contains(d))
}
// ---- content helpers ----
pub fn parse_content(raw: &RawValue) -> Vec<ContentItem> {
serde_json::from_str(raw.get()).unwrap_or_default()
}
/// Concatenates all `text` items in a content array.
pub fn join_text(items: &[ContentItem]) -> String {
let mut out = String::new();
for it in items {
if it.kind == "text" {
out.push_str(&it.text);
}
}
out
}
// ---- tokenizer (o200k_base) ----
static BPE: LazyLock<CoreBPE> =
LazyLock::new(|| tiktoken_rs::o200k_base().expect("load o200k_base BPE"));
/// Counts tokens for `s` using the o200k_base BPE (GPT-4o / GPT-5 family).
/// Uses the ordinary encoder so embedded `<|...|>` sequences in tool args do
/// not trigger special-token handling.
pub fn count_tokens(s: &str) -> usize {
if s.is_empty() {
return 0;
}
BPE.encode_ordinary(s).len()
}
// ---- formatting helpers ----
/// Formats an integer with thousand separators.
pub fn commas(n: i64) -> String {
let neg = n < 0;
let mag = if neg { (n as i128).unsigned_abs() } else { n as u128 };
let s = mag.to_string();
let bytes = s.as_bytes();
let mut out = String::with_capacity(bytes.len() + bytes.len() / 3 + 1);
if neg {
out.push('-');
}
let pre = bytes.len() % 3;
if pre > 0 {
out.push_str(&s[..pre]);
if bytes.len() > pre {
out.push(',');
}
}
let mut i = pre;
while i + 3 <= bytes.len() {
out.push_str(&s[i..i + 3]);
if i + 3 < bytes.len() {
out.push(',');
}
i += 3;
}
out
}
pub fn pct(a: i64, b: i64) -> f64 {
if b == 0 {
0.0
} else {
100.0 * a as f64 / b as f64
}
}
/// Truncates a string to at most `n` chars, replacing newlines with " | ".
/// Adds an ellipsis when truncation occurs.
pub fn truncate_line(s: &str, n: usize) -> String {
let s = s.replace('\n', " | ");
if s.chars().count() <= n {
return s;
}
let mut out: String = s.chars().take(n).collect();
out.push('…');
out
}
/// Returns map keys sorted by descending value, ties broken alphabetically.
pub fn sorted_by_count(m: &HashMap<String, i64>) -> Vec<&String> {
let mut keys: Vec<&String> = m.keys().collect();
keys.sort_by(|a, b| {
let av = m.get(a.as_str()).copied().unwrap_or(0);
let bv = m.get(b.as_str()).copied().unwrap_or(0);
bv.cmp(&av).then_with(|| a.cmp(b))
});
keys
}
pub fn print_sorted(m: &HashMap<String, i64>) {
print_sorted_indent(m, " ");
}
pub fn print_sorted_indent(m: &HashMap<String, i64>, indent: &str) {
for k in sorted_by_count(m) {
println!("{indent}{k:<25} {}", m[k.as_str()]);
}
}
// ---- parallel processing ----
/// Runs `handle(path)` in parallel across rayon workers and collects the
/// non-`None` results into a Vec. Logs progress every `progress_every` files
/// (set 0 to silence).
pub fn parallel_collect<R, H>(
paths: &[PathBuf],
workers: usize,
progress_every: u64,
handle: H,
) -> Vec<R>
where
R: Send,
H: Fn(&Path) -> Option<R> + Sync,
{
let total = paths.len();
let done = AtomicU64::new(0);
let pool = {
let mut b = rayon::ThreadPoolBuilder::new();
if workers > 0 {
b = b.num_threads(workers);
}
b.build().expect("rayon thread pool")
};
pool.install(|| {
paths
.par_iter()
.filter_map(|p| {
let r = handle(p);
let n = done.fetch_add(1, Ordering::Relaxed) + 1;
if progress_every > 0 && n % progress_every == 0 {
eprintln!(" processed {n}/{total}");
}
r
})
.collect()
})
}
+58
View File
@@ -0,0 +1,58 @@
//! session-stats: ad-hoc analyses over the local agent session corpus
//! (`~/.omp/agent/sessions/`).
//!
//! Subcommands:
//!
//! edits [-n N] [date ...] audit edit/ast_edit/write tool usage by argument schema
//! tools [-n N] per-tool token totals across the most-recent N sessions
//!
//! Run with no subcommand for help.
mod cmd_edits;
mod cmd_tools;
mod common;
use std::process::ExitCode;
fn usage() {
eprintln!(
"usage: session-stats <subcommand> [args...]
edits [-n N] [date prefix ...]
audit edit-tool usage across N most-recent
sessions (default 100000). Optional date filters
(e.g. 2026-04-28) further narrow the set.
tools [-n N] per-tool token totals across the N most-recent
session jsonl files (default 100000).
Token counting uses the o200k_base tokenizer (the GPT-4o / Claude-adjacent BPE).
Walk root: ~/.omp/agent/sessions/"
);
}
fn main() -> ExitCode {
let mut args = std::env::args().skip(1);
let Some(cmd) = args.next() else {
usage();
return ExitCode::from(2);
};
let rest: Vec<String> = args.collect();
let result = match cmd.as_str() {
"edits" => cmd_edits::run(rest),
"tools" => cmd_tools::run(rest),
"-h" | "--help" | "help" => {
usage();
return ExitCode::SUCCESS;
}
other => {
eprintln!("unknown subcommand {other:?}\n");
usage();
return ExitCode::from(2);
}
};
if let Err(err) = result {
eprintln!("fatal: {err:#}");
return ExitCode::FAILURE;
}
ExitCode::SUCCESS
}