diff --git a/scripts/session-stats/README.md b/scripts/session-stats/README.md index cabb086e9..74ac9151a 100644 --- a/scripts/session-stats/README.md +++ b/scripts/session-stats/README.md @@ -47,7 +47,7 @@ All tables are prefixed `ss_` to avoid collision with `packages/stats`. |`ss_assistant_msgs`|per assistant message text + thinking blobs and token counts| |`ss_user_msgs`|per user message text and token count| |`ss_edit_calls`|per `edit` call: `success`, `warnings`, `raw_input_len`| -|`ss_edit_sections`|per `§PATH` section in an edit; precomputed `longest_repeat_*`, `dup_anchors`| +|`ss_edit_sections`|per `¶PATH` section in an edit; precomputed `longest_repeat_*`, `dup_anchors`. Legacy `§PATH` sections from pre-2026-05 sessions still parse.| Indexes on `(tool_name, timestamp)` and `(session_file, seq)` make per-tool aggregations and ordered session walks cheap. diff --git a/scripts/session-stats/harmony_backtest.py b/scripts/session-stats/harmony_backtest.py index 21acf9d95..8b4a0c82e 100755 --- a/scripts/session-stats/harmony_backtest.py +++ b/scripts/session-stats/harmony_backtest.py @@ -53,12 +53,28 @@ SCRIPT_RUN_RE = re.compile( "]{2,}" ) -HEADER_RE = re.compile(r"^(?:§+(?P.*)|\*\*\* Update File:\s+(?P\S.*))\s*$") +# Header detector — matches any of: +# ¶PATH or ¶PATH#hash (current hashline format) +# §PATH (legacy hashline format, pre-2026-05) +# *** Update File: PATH (Codex apply_patch envelope) +HEADER_RE = re.compile( + r"^(?:§+(?P.*)|¶+\s*(?P[^\s#¶]+)(?:#[0-9a-f]{4})?|\*\*\* Update File:\s+(?P\S.*))\s*$" +) BEGIN_PATCH_RE = re.compile(r"^\*\*\* Begin Patch\s*$") END_PATCH_RE = re.compile(r"^\*\*\* End Patch\s*$") -INSERT_RE = re.compile(r"^(?P[«»])\s*(?PBOF|EOF|[1-9][0-9]*[A-Za-z]{2})\s*$") -RANGE_RE = re.compile(r"(?P[1-9][0-9]*[A-Za-z]{2})(?:\.\.(?P[1-9][0-9]*[A-Za-z]{2}))?") -REPLACE_RE = re.compile(r"^≔\s*(?P[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$") + +# Legacy hashline ops (kept for historical session corpus). +LEGACY_INSERT_RE = re.compile(r"^(?P[«»])\s*(?PBOF|EOF|[1-9][0-9]*[A-Za-z]{2})\s*$") +LEGACY_RANGE_RE = re.compile(r"(?P[1-9][0-9]*[A-Za-z]{2})(?:\.\.(?P[1-9][0-9]*[A-Za-z]{2}))?") +LEGACY_REPLACE_RE = re.compile(r"^≔\s*(?P[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$") + +# Current hashline ops. +NEW_INSERT_RE = re.compile( + r"^\s*(?:[>+\-*]+\s*)?(?P[1-9][0-9]*|BOF|EOF)(?P[↑↓])(?P.*)$" +) +NEW_RANGE_RE = re.compile( + r"^\s*(?:[>+\-*]+\s*)?(?P[1-9][0-9]*)(?:-(?P[1-9][0-9]*))?(?P[:!])(?P.*)$" +) JSON_DECODER = json.JSONDecoder() @@ -410,8 +426,8 @@ def anchor_line_no(anchor: str) -> int | None: return int(m.group(1)) if m else None -def range_deleted_lines(raw_range: str) -> int: - m = RANGE_RE.fullmatch(raw_range) +def legacy_range_deleted_lines(raw_range: str) -> int: + m = LEGACY_RANGE_RE.fullmatch(raw_range) if not m: return 1 a = anchor_line_no(m.group("a")) or 0 @@ -419,16 +435,41 @@ def range_deleted_lines(raw_range: str) -> int: return max(1, b - a + 1) +def new_range_deleted_lines(a_raw: str, b_raw: str | None) -> int: + try: + a = int(a_raw) + b = int(b_raw) if b_raw else a + except ValueError: + return 1 + return max(1, b - a + 1) + + def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBoundary: + """Find the longest tool-input prefix that parses as a complete sequence of + edit sections. + + Supports both the current hashline format (``¶PATH#hash`` / ``↑↓:!``) and + the legacy format (``§PATH`` / ``«»≔``) since the analyzed corpus spans the + format transition. Codex ``*** Update File:`` envelopes are also recognized. + """ sections: list[EditSection] = [] cur: EditSection | None = None + # Per-section format: "new", "legacy_hl", or "codex". Determines which op + # regexes apply and whether `needs_payload` semantics are in effect. + cur_format: str | None = None parsed_end = 0 line_no = 0 + # Legacy hashline (`«»`) inserts demand at least one payload line before the + # boundary advances. New-format inserts are self-completing on their own + # line, so these flags only matter when ``cur_format == "legacy_hl"``. needs_payload = False payload_allowed = False saw_required_payload = False seen_content = False + legacy_payload_blockers = {"«", "»", "≔", "§"} + new_payload_blockers = ("¶",) + for line, _start, end in line_spans(text): line_no += 1 stripped = line.strip() @@ -456,7 +497,15 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo if header: if needs_payload and not saw_required_payload: break - target = (header.group("hl") or header.group("upd") or "").strip() + if header.group("hl_new") is not None: + target = header.group("hl_new").strip() + cur_format = "new" + elif header.group("hl_legacy") is not None: + target = header.group("hl_legacy").strip() + cur_format = "legacy_hl" + else: + target = (header.group("upd") or "").strip() + cur_format = "codex" cur = EditSection(target_file=target) sections.append(cur) parsed_end = end @@ -468,7 +517,49 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo if cur is None: break - if payload_allowed and line[:1] not in {"«", "»", "≔", "§"}: + if cur_format == "new": + # New format: payload lines are anything that does not start a new + # op or header. `↑`/`↓`/`:` ops may be followed by additional + # payload lines; `!` ops are self-contained. + if payload_allowed and not line.startswith(new_payload_blockers): + # Re-check whether the line itself is an op — an op line ends + # the current payload run and starts a new op. + if not NEW_INSERT_RE.match(line) and not NEW_RANGE_RE.match(line): + cur.payload_lines += 1 + parsed_end = end + continue + + if not stripped: + parsed_end = end + payload_allowed = False + continue + + ins = NEW_INSERT_RE.match(line) + if ins: + cur.op_count += 1 + if ins.group("inline"): + cur.payload_lines += 1 + parsed_end = end + payload_allowed = True + continue + + rng = NEW_RANGE_RE.match(line) + if rng: + sigil = rng.group("sigil") + cur.op_count += 1 + cur.deleted_lines += new_range_deleted_lines(rng.group("a"), rng.group("b")) + if sigil == ":" and rng.group("inline"): + cur.payload_lines += 1 + parsed_end = end + # `!` deletes do not accept payload; `:` replaces may carry + # subsequent payload lines. + payload_allowed = sigil == ":" + continue + + break + + # Legacy hashline state machine (preserved verbatim for §«»≔ corpus). + if payload_allowed and line[:1] not in legacy_payload_blockers: cur.payload_lines += 1 parsed_end = end saw_required_payload = True @@ -487,7 +578,7 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo if needs_payload and not saw_required_payload: break - ins = INSERT_RE.match(line) + ins = LEGACY_INSERT_RE.match(line) if ins: cur.op_count += 1 needs_payload = True @@ -496,11 +587,10 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo # Not complete until at least one payload line appears. continue - - repl = REPLACE_RE.match(line) + repl = LEGACY_REPLACE_RE.match(line) if repl: cur.op_count += 1 - cur.deleted_lines += range_deleted_lines(repl.group("range")) + cur.deleted_lines += legacy_range_deleted_lines(repl.group("range")) parsed_end = end needs_payload = False payload_allowed = True diff --git a/scripts/session-stats/sync.py b/scripts/session-stats/sync.py index 385749f95..eb38531b8 100644 --- a/scripts/session-stats/sync.py +++ b/scripts/session-stats/sync.py @@ -15,7 +15,7 @@ Schema (all tables prefixed `ss_` to avoid collision with packages/stats): ss_assistant_msgs one row per assistant message (text + thinking blobs) ss_user_msgs one row per user message (text blob) ss_edit_calls one row per edit toolCall (success + warnings paired in) - ss_edit_sections one row per §PATH section inside an edit toolCall, with + ss_edit_sections one row per ¶PATH section inside an edit toolCall, with precomputed detector outputs (longest_repeat_*, dup_anchors) Run: @@ -56,7 +56,7 @@ SCHEMA_VERSION = 3 # Bump whenever parse_hashline_input / find_longest_repeat / duplicated_anchors # / looks_successful / extract_warnings semantics change. Bump invalidates # previously-stored ss_edit_* rows on next sync. -EDIT_PARSER_VERSION = 4 +EDIT_PARSER_VERSION = 5 SCHEMA_SQL = """ CREATE TABLE IF NOT EXISTS ss_sessions ( @@ -227,17 +227,39 @@ def batch_count_tokens(strings: list[str]) -> list[int]: # --------------------------------------------------------------------------- # -# Hashline edit parser (port of cmd_followups.rs::parse_hashline_input). +# Hashline edit parser. +# +# Supports two on-the-wire formats so the analytic tables stay coherent across +# the format transition: +# +# new (current): ¶PATH[#HASH], LINE↑[body], LINE↓[body], A[-B]:[body], A[-B]! +# legacy: §PATH, «ANCHOR, »ANCHOR, ≔ANCHOR[..ANCHOR] +# +# Legacy "anchor" tokens were `<2-letter-hash>` (e.g. `4fb`, `12*`); +# new ops use bare line numbers and hoist the file hash into the header. + +_LEGACY_RANGE_RE = re.compile(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$") +_LEGACY_SINGLE_ANCHOR_RE = re.compile(r"^\s*(\d+)[a-z*]+\s*$") +_LEGACY_OP_RE = re.compile(r"^([«»≔])\s*(\S+)\s*$") + +# Header: one or more `¶`, optional whitespace, path (no whitespace/#/¶), +# optional `#HASH` (4 lowercase hex). +_HEADER_NEW_RE = re.compile(r"^¶+\s*([^\s#¶]+)(?:#([0-9a-f]{4}))?\s*$") +# Insert op: LINE↑BODY / LINE↓BODY / BOF↑BODY / EOF↓BODY … +_OP_INSERT_NEW_RE = re.compile( + r"^\s*(?:[>+\-*]+\s*)?(?P[1-9][0-9]*|BOF|EOF)(?P[↑↓])(?P.*)$" +) +# Replace / delete op: A:BODY / A-B:BODY / A! / A-B! +_OP_RANGE_NEW_RE = re.compile( + r"^\s*(?:[>+\-*]+\s*)?(?P[1-9][0-9]*)(?:-(?P[1-9][0-9]*))?(?P[:!])(?P.*)$" +) -_RANGE_RE = re.compile(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$") -_SINGLE_ANCHOR_RE = re.compile(r"^\s*(\d+)[a-z*]+\s*$") -_HASHLINE_OP_RE = re.compile(r"^([«»≔])\s*(\S+)\s*$") _HASHLINE_ENVELOPE_MARKERS = {"*** Begin Patch", "*** End Patch", "*** Abort"} -def _parse_range(raw: str) -> tuple[int, tuple[int, int] | None]: +def _parse_legacy_range(raw: str) -> tuple[int, tuple[int, int] | None]: """Returns (range_size, optional (start_line, end_line)). Size >= 1.""" - m = _RANGE_RE.match(raw.strip()) + m = _LEGACY_RANGE_RE.match(raw.strip()) if not m: return (1, None) start = int(m.group(1)) @@ -248,8 +270,8 @@ def _parse_range(raw: str) -> tuple[int, tuple[int, int] | None]: return (size, lines) -def _parse_anchor_line(raw: str) -> int | None: - m = _SINGLE_ANCHOR_RE.match(raw.strip()) +def _parse_legacy_anchor_line(raw: str) -> int | None: + m = _LEGACY_SINGLE_ANCHOR_RE.match(raw.strip()) if not m: return None try: @@ -284,6 +306,7 @@ class EditSection: def parse_hashline_input(input_str: str) -> list[EditSection]: sections: list[EditSection] = [] cur: EditSection | None = None + cur_format: str | None = None # "new" | "legacy" open_idx: int | None = None # current open payload block in cur def open_new(s: EditSection) -> int: @@ -299,6 +322,15 @@ def parse_hashline_input(input_str: str) -> list[EditSection]: break continue + # Headers — new format first, then legacy. + new_header = _HEADER_NEW_RE.match(line) + if new_header: + if cur is not None: + sections.append(cur) + cur = EditSection(target_file=new_header.group(1)) + cur_format = "new" + open_idx = None + continue if line.startswith("§"): if cur is not None: sections.append(cur) @@ -306,38 +338,86 @@ def parse_hashline_input(input_str: str) -> list[EditSection]: while prefix_end < len(line) and line[prefix_end] == "§": prefix_end += 1 cur = EditSection(target_file=line[prefix_end:].strip()) + cur_format = "legacy" open_idx = None continue + if cur is None: continue - op_match = _HASHLINE_OP_RE.match(line) - if op_match: - op = op_match.group(1) - body = op_match.group(2) - if op in ("«", "»"): - anchor_trimmed = body.strip() - if anchor_trimmed and anchor_trimmed not in ("BOF", "EOF"): - cur.op_anchors.append(anchor_trimmed) - line_no = _parse_anchor_line(body) - if line_no is not None: - cur.touch(line_no) + if cur_format == "new": + ins = _OP_INSERT_NEW_RE.match(line) + if ins: + anchor = ins.group("anchor") + inline = ins.group("inline") + cur.op_anchors.append(anchor) + if anchor not in ("BOF", "EOF"): + try: + cur.touch(int(anchor)) + except ValueError: + pass open_idx = open_new(cur) cur.op_count += 1 + if inline: + cur.payload_blocks[open_idx].append(inline) continue - if op == "≔": - size, lines = _parse_range(body) + + rng = _OP_RANGE_NEW_RE.match(line) + if rng: + sigil = rng.group("sigil") + a_str = rng.group("a") + b_str = rng.group("b") or a_str + inline = rng.group("inline") + try: + a = int(a_str) + b = int(b_str) + size = max(b - a + 1, 1) + cur.touch(a) + cur.touch(b) + except ValueError: + size = 1 + cur.op_anchors.append(a_str) + if b_str != a_str: + cur.op_anchors.append(b_str) cur.deleted_lines += size - if lines is not None: - cur.touch(lines[0]) - cur.touch(lines[1]) - for part in body.strip().split(".."): - t = part.strip() - if t: - cur.op_anchors.append(t) cur.op_count += 1 + if sigil == "!": + # Delete op: payload forbidden by the production parser; close. + open_idx = None + continue + # sigil == ":" — replace. open_idx = open_new(cur) + if inline: + cur.payload_blocks[open_idx].append(inline) continue + else: + op_match = _LEGACY_OP_RE.match(line) + if op_match: + op = op_match.group(1) + body = op_match.group(2) + if op in ("«", "»"): + anchor_trimmed = body.strip() + if anchor_trimmed and anchor_trimmed not in ("BOF", "EOF"): + cur.op_anchors.append(anchor_trimmed) + line_no = _parse_legacy_anchor_line(body) + if line_no is not None: + cur.touch(line_no) + open_idx = open_new(cur) + cur.op_count += 1 + continue + if op == "≔": + size, lines = _parse_legacy_range(body) + cur.deleted_lines += size + if lines is not None: + cur.touch(lines[0]) + cur.touch(lines[1]) + for part in body.strip().split(".."): + t = part.strip() + if t: + cur.op_anchors.append(t) + cur.op_count += 1 + open_idx = open_new(cur) + continue if open_idx is not None: cur.payload_blocks[open_idx].append(line) @@ -669,7 +749,7 @@ def _ingest_edit_call(rec, sf, seq, ts, call_id, arg_obj, arg_json) -> None: (sf, call_id, seq, ts, raw_input_len, EDIT_PARSER_VERSION) ) - if not any(line.startswith("§") for line in input_str.lstrip("\ufeff").splitlines()): + if not any(line.startswith(("¶", "§")) for line in input_str.lstrip("\ufeff").splitlines()): # Vim-mode or other shape — no sections to record. return