feat(scripts/session-stats): added dual-format hashline parsing support
- Enhanced parser to accept both new ¶PATH and legacy §PATH section headers in edit input. - Added dedicated new-format regex handlers for insert, replace, and delete ops while retaining legacy anchor parsing semantics. - Bumped EDIT_PARSER_VERSION to 5 so downstream syncs refresh rows for the parser change.
This commit is contained in:
@@ -47,7 +47,7 @@ All tables are prefixed `ss_` to avoid collision with `packages/stats`.
|
||||
|`ss_assistant_msgs`|per assistant message text + thinking blobs and token counts|
|
||||
|`ss_user_msgs`|per user message text and token count|
|
||||
|`ss_edit_calls`|per `edit` call: `success`, `warnings`, `raw_input_len`|
|
||||
|`ss_edit_sections`|per `§PATH` section in an edit; precomputed `longest_repeat_*`, `dup_anchors`|
|
||||
|`ss_edit_sections`|per `¶PATH` section in an edit; precomputed `longest_repeat_*`, `dup_anchors`. Legacy `§PATH` sections from pre-2026-05 sessions still parse.|
|
||||
|
||||
Indexes on `(tool_name, timestamp)` and `(session_file, seq)` make per-tool
|
||||
aggregations and ordered session walks cheap.
|
||||
|
||||
@@ -53,12 +53,28 @@ SCRIPT_RUN_RE = re.compile(
|
||||
"]{2,}"
|
||||
)
|
||||
|
||||
HEADER_RE = re.compile(r"^(?:§+(?P<hl>.*)|\*\*\* Update File:\s+(?P<upd>\S.*))\s*$")
|
||||
# Header detector — matches any of:
|
||||
# ¶PATH or ¶PATH#hash (current hashline format)
|
||||
# §PATH (legacy hashline format, pre-2026-05)
|
||||
# *** Update File: PATH (Codex apply_patch envelope)
|
||||
HEADER_RE = re.compile(
|
||||
r"^(?:§+(?P<hl_legacy>.*)|¶+\s*(?P<hl_new>[^\s#¶]+)(?:#[0-9a-f]{4})?|\*\*\* Update File:\s+(?P<upd>\S.*))\s*$"
|
||||
)
|
||||
BEGIN_PATCH_RE = re.compile(r"^\*\*\* Begin Patch\s*$")
|
||||
END_PATCH_RE = re.compile(r"^\*\*\* End Patch\s*$")
|
||||
INSERT_RE = re.compile(r"^(?P<op>[«»])\s*(?P<anchor>BOF|EOF|[1-9][0-9]*[A-Za-z]{2})\s*$")
|
||||
RANGE_RE = re.compile(r"(?P<a>[1-9][0-9]*[A-Za-z]{2})(?:\.\.(?P<b>[1-9][0-9]*[A-Za-z]{2}))?")
|
||||
REPLACE_RE = re.compile(r"^≔\s*(?P<range>[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$")
|
||||
|
||||
# Legacy hashline ops (kept for historical session corpus).
|
||||
LEGACY_INSERT_RE = re.compile(r"^(?P<op>[«»])\s*(?P<anchor>BOF|EOF|[1-9][0-9]*[A-Za-z]{2})\s*$")
|
||||
LEGACY_RANGE_RE = re.compile(r"(?P<a>[1-9][0-9]*[A-Za-z]{2})(?:\.\.(?P<b>[1-9][0-9]*[A-Za-z]{2}))?")
|
||||
LEGACY_REPLACE_RE = re.compile(r"^≔\s*(?P<range>[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$")
|
||||
|
||||
# Current hashline ops.
|
||||
NEW_INSERT_RE = re.compile(
|
||||
r"^\s*(?:[>+\-*]+\s*)?(?P<anchor>[1-9][0-9]*|BOF|EOF)(?P<sigil>[↑↓])(?P<inline>.*)$"
|
||||
)
|
||||
NEW_RANGE_RE = re.compile(
|
||||
r"^\s*(?:[>+\-*]+\s*)?(?P<a>[1-9][0-9]*)(?:-(?P<b>[1-9][0-9]*))?(?P<sigil>[:!])(?P<inline>.*)$"
|
||||
)
|
||||
|
||||
JSON_DECODER = json.JSONDecoder()
|
||||
|
||||
@@ -410,8 +426,8 @@ def anchor_line_no(anchor: str) -> int | None:
|
||||
return int(m.group(1)) if m else None
|
||||
|
||||
|
||||
def range_deleted_lines(raw_range: str) -> int:
|
||||
m = RANGE_RE.fullmatch(raw_range)
|
||||
def legacy_range_deleted_lines(raw_range: str) -> int:
|
||||
m = LEGACY_RANGE_RE.fullmatch(raw_range)
|
||||
if not m:
|
||||
return 1
|
||||
a = anchor_line_no(m.group("a")) or 0
|
||||
@@ -419,16 +435,41 @@ def range_deleted_lines(raw_range: str) -> int:
|
||||
return max(1, b - a + 1)
|
||||
|
||||
|
||||
def new_range_deleted_lines(a_raw: str, b_raw: str | None) -> int:
|
||||
try:
|
||||
a = int(a_raw)
|
||||
b = int(b_raw) if b_raw else a
|
||||
except ValueError:
|
||||
return 1
|
||||
return max(1, b - a + 1)
|
||||
|
||||
|
||||
def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBoundary:
|
||||
"""Find the longest tool-input prefix that parses as a complete sequence of
|
||||
edit sections.
|
||||
|
||||
Supports both the current hashline format (``¶PATH#hash`` / ``↑↓:!``) and
|
||||
the legacy format (``§PATH`` / ``«»≔``) since the analyzed corpus spans the
|
||||
format transition. Codex ``*** Update File:`` envelopes are also recognized.
|
||||
"""
|
||||
sections: list[EditSection] = []
|
||||
cur: EditSection | None = None
|
||||
# Per-section format: "new", "legacy_hl", or "codex". Determines which op
|
||||
# regexes apply and whether `needs_payload` semantics are in effect.
|
||||
cur_format: str | None = None
|
||||
parsed_end = 0
|
||||
line_no = 0
|
||||
# Legacy hashline (`«»`) inserts demand at least one payload line before the
|
||||
# boundary advances. New-format inserts are self-completing on their own
|
||||
# line, so these flags only matter when ``cur_format == "legacy_hl"``.
|
||||
needs_payload = False
|
||||
payload_allowed = False
|
||||
saw_required_payload = False
|
||||
seen_content = False
|
||||
|
||||
legacy_payload_blockers = {"«", "»", "≔", "§"}
|
||||
new_payload_blockers = ("¶",)
|
||||
|
||||
for line, _start, end in line_spans(text):
|
||||
line_no += 1
|
||||
stripped = line.strip()
|
||||
@@ -456,7 +497,15 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo
|
||||
if header:
|
||||
if needs_payload and not saw_required_payload:
|
||||
break
|
||||
target = (header.group("hl") or header.group("upd") or "").strip()
|
||||
if header.group("hl_new") is not None:
|
||||
target = header.group("hl_new").strip()
|
||||
cur_format = "new"
|
||||
elif header.group("hl_legacy") is not None:
|
||||
target = header.group("hl_legacy").strip()
|
||||
cur_format = "legacy_hl"
|
||||
else:
|
||||
target = (header.group("upd") or "").strip()
|
||||
cur_format = "codex"
|
||||
cur = EditSection(target_file=target)
|
||||
sections.append(cur)
|
||||
parsed_end = end
|
||||
@@ -468,7 +517,49 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo
|
||||
if cur is None:
|
||||
break
|
||||
|
||||
if payload_allowed and line[:1] not in {"«", "»", "≔", "§"}:
|
||||
if cur_format == "new":
|
||||
# New format: payload lines are anything that does not start a new
|
||||
# op or header. `↑`/`↓`/`:` ops may be followed by additional
|
||||
# payload lines; `!` ops are self-contained.
|
||||
if payload_allowed and not line.startswith(new_payload_blockers):
|
||||
# Re-check whether the line itself is an op — an op line ends
|
||||
# the current payload run and starts a new op.
|
||||
if not NEW_INSERT_RE.match(line) and not NEW_RANGE_RE.match(line):
|
||||
cur.payload_lines += 1
|
||||
parsed_end = end
|
||||
continue
|
||||
|
||||
if not stripped:
|
||||
parsed_end = end
|
||||
payload_allowed = False
|
||||
continue
|
||||
|
||||
ins = NEW_INSERT_RE.match(line)
|
||||
if ins:
|
||||
cur.op_count += 1
|
||||
if ins.group("inline"):
|
||||
cur.payload_lines += 1
|
||||
parsed_end = end
|
||||
payload_allowed = True
|
||||
continue
|
||||
|
||||
rng = NEW_RANGE_RE.match(line)
|
||||
if rng:
|
||||
sigil = rng.group("sigil")
|
||||
cur.op_count += 1
|
||||
cur.deleted_lines += new_range_deleted_lines(rng.group("a"), rng.group("b"))
|
||||
if sigil == ":" and rng.group("inline"):
|
||||
cur.payload_lines += 1
|
||||
parsed_end = end
|
||||
# `!` deletes do not accept payload; `:` replaces may carry
|
||||
# subsequent payload lines.
|
||||
payload_allowed = sigil == ":"
|
||||
continue
|
||||
|
||||
break
|
||||
|
||||
# Legacy hashline state machine (preserved verbatim for §«»≔ corpus).
|
||||
if payload_allowed and line[:1] not in legacy_payload_blockers:
|
||||
cur.payload_lines += 1
|
||||
parsed_end = end
|
||||
saw_required_payload = True
|
||||
@@ -487,7 +578,7 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo
|
||||
if needs_payload and not saw_required_payload:
|
||||
break
|
||||
|
||||
ins = INSERT_RE.match(line)
|
||||
ins = LEGACY_INSERT_RE.match(line)
|
||||
if ins:
|
||||
cur.op_count += 1
|
||||
needs_payload = True
|
||||
@@ -496,11 +587,10 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo
|
||||
# Not complete until at least one payload line appears.
|
||||
continue
|
||||
|
||||
|
||||
repl = REPLACE_RE.match(line)
|
||||
repl = LEGACY_REPLACE_RE.match(line)
|
||||
if repl:
|
||||
cur.op_count += 1
|
||||
cur.deleted_lines += range_deleted_lines(repl.group("range"))
|
||||
cur.deleted_lines += legacy_range_deleted_lines(repl.group("range"))
|
||||
parsed_end = end
|
||||
needs_payload = False
|
||||
payload_allowed = True
|
||||
|
||||
+111
-31
@@ -15,7 +15,7 @@ Schema (all tables prefixed `ss_` to avoid collision with packages/stats):
|
||||
ss_assistant_msgs one row per assistant message (text + thinking blobs)
|
||||
ss_user_msgs one row per user message (text blob)
|
||||
ss_edit_calls one row per edit toolCall (success + warnings paired in)
|
||||
ss_edit_sections one row per §PATH section inside an edit toolCall, with
|
||||
ss_edit_sections one row per ¶PATH section inside an edit toolCall, with
|
||||
precomputed detector outputs (longest_repeat_*, dup_anchors)
|
||||
|
||||
Run:
|
||||
@@ -56,7 +56,7 @@ SCHEMA_VERSION = 3
|
||||
# Bump whenever parse_hashline_input / find_longest_repeat / duplicated_anchors
|
||||
# / looks_successful / extract_warnings semantics change. Bump invalidates
|
||||
# previously-stored ss_edit_* rows on next sync.
|
||||
EDIT_PARSER_VERSION = 4
|
||||
EDIT_PARSER_VERSION = 5
|
||||
|
||||
SCHEMA_SQL = """
|
||||
CREATE TABLE IF NOT EXISTS ss_sessions (
|
||||
@@ -227,17 +227,39 @@ def batch_count_tokens(strings: list[str]) -> list[int]:
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# Hashline edit parser (port of cmd_followups.rs::parse_hashline_input).
|
||||
# Hashline edit parser.
|
||||
#
|
||||
# Supports two on-the-wire formats so the analytic tables stay coherent across
|
||||
# the format transition:
|
||||
#
|
||||
# new (current): ¶PATH[#HASH], LINE↑[body], LINE↓[body], A[-B]:[body], A[-B]!
|
||||
# legacy: §PATH, «ANCHOR, »ANCHOR, ≔ANCHOR[..ANCHOR]
|
||||
#
|
||||
# Legacy "anchor" tokens were `<line><2-letter-hash>` (e.g. `4fb`, `12*`);
|
||||
# new ops use bare line numbers and hoist the file hash into the header.
|
||||
|
||||
_LEGACY_RANGE_RE = re.compile(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$")
|
||||
_LEGACY_SINGLE_ANCHOR_RE = re.compile(r"^\s*(\d+)[a-z*]+\s*$")
|
||||
_LEGACY_OP_RE = re.compile(r"^([«»≔])\s*(\S+)\s*$")
|
||||
|
||||
# Header: one or more `¶`, optional whitespace, path (no whitespace/#/¶),
|
||||
# optional `#HASH` (4 lowercase hex).
|
||||
_HEADER_NEW_RE = re.compile(r"^¶+\s*([^\s#¶]+)(?:#([0-9a-f]{4}))?\s*$")
|
||||
# Insert op: LINE↑BODY / LINE↓BODY / BOF↑BODY / EOF↓BODY …
|
||||
_OP_INSERT_NEW_RE = re.compile(
|
||||
r"^\s*(?:[>+\-*]+\s*)?(?P<anchor>[1-9][0-9]*|BOF|EOF)(?P<sigil>[↑↓])(?P<inline>.*)$"
|
||||
)
|
||||
# Replace / delete op: A:BODY / A-B:BODY / A! / A-B!
|
||||
_OP_RANGE_NEW_RE = re.compile(
|
||||
r"^\s*(?:[>+\-*]+\s*)?(?P<a>[1-9][0-9]*)(?:-(?P<b>[1-9][0-9]*))?(?P<sigil>[:!])(?P<inline>.*)$"
|
||||
)
|
||||
|
||||
_RANGE_RE = re.compile(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$")
|
||||
_SINGLE_ANCHOR_RE = re.compile(r"^\s*(\d+)[a-z*]+\s*$")
|
||||
_HASHLINE_OP_RE = re.compile(r"^([«»≔])\s*(\S+)\s*$")
|
||||
_HASHLINE_ENVELOPE_MARKERS = {"*** Begin Patch", "*** End Patch", "*** Abort"}
|
||||
|
||||
|
||||
def _parse_range(raw: str) -> tuple[int, tuple[int, int] | None]:
|
||||
def _parse_legacy_range(raw: str) -> tuple[int, tuple[int, int] | None]:
|
||||
"""Returns (range_size, optional (start_line, end_line)). Size >= 1."""
|
||||
m = _RANGE_RE.match(raw.strip())
|
||||
m = _LEGACY_RANGE_RE.match(raw.strip())
|
||||
if not m:
|
||||
return (1, None)
|
||||
start = int(m.group(1))
|
||||
@@ -248,8 +270,8 @@ def _parse_range(raw: str) -> tuple[int, tuple[int, int] | None]:
|
||||
return (size, lines)
|
||||
|
||||
|
||||
def _parse_anchor_line(raw: str) -> int | None:
|
||||
m = _SINGLE_ANCHOR_RE.match(raw.strip())
|
||||
def _parse_legacy_anchor_line(raw: str) -> int | None:
|
||||
m = _LEGACY_SINGLE_ANCHOR_RE.match(raw.strip())
|
||||
if not m:
|
||||
return None
|
||||
try:
|
||||
@@ -284,6 +306,7 @@ class EditSection:
|
||||
def parse_hashline_input(input_str: str) -> list[EditSection]:
|
||||
sections: list[EditSection] = []
|
||||
cur: EditSection | None = None
|
||||
cur_format: str | None = None # "new" | "legacy"
|
||||
open_idx: int | None = None # current open payload block in cur
|
||||
|
||||
def open_new(s: EditSection) -> int:
|
||||
@@ -299,6 +322,15 @@ def parse_hashline_input(input_str: str) -> list[EditSection]:
|
||||
break
|
||||
continue
|
||||
|
||||
# Headers — new format first, then legacy.
|
||||
new_header = _HEADER_NEW_RE.match(line)
|
||||
if new_header:
|
||||
if cur is not None:
|
||||
sections.append(cur)
|
||||
cur = EditSection(target_file=new_header.group(1))
|
||||
cur_format = "new"
|
||||
open_idx = None
|
||||
continue
|
||||
if line.startswith("§"):
|
||||
if cur is not None:
|
||||
sections.append(cur)
|
||||
@@ -306,38 +338,86 @@ def parse_hashline_input(input_str: str) -> list[EditSection]:
|
||||
while prefix_end < len(line) and line[prefix_end] == "§":
|
||||
prefix_end += 1
|
||||
cur = EditSection(target_file=line[prefix_end:].strip())
|
||||
cur_format = "legacy"
|
||||
open_idx = None
|
||||
continue
|
||||
|
||||
if cur is None:
|
||||
continue
|
||||
|
||||
op_match = _HASHLINE_OP_RE.match(line)
|
||||
if op_match:
|
||||
op = op_match.group(1)
|
||||
body = op_match.group(2)
|
||||
if op in ("«", "»"):
|
||||
anchor_trimmed = body.strip()
|
||||
if anchor_trimmed and anchor_trimmed not in ("BOF", "EOF"):
|
||||
cur.op_anchors.append(anchor_trimmed)
|
||||
line_no = _parse_anchor_line(body)
|
||||
if line_no is not None:
|
||||
cur.touch(line_no)
|
||||
if cur_format == "new":
|
||||
ins = _OP_INSERT_NEW_RE.match(line)
|
||||
if ins:
|
||||
anchor = ins.group("anchor")
|
||||
inline = ins.group("inline")
|
||||
cur.op_anchors.append(anchor)
|
||||
if anchor not in ("BOF", "EOF"):
|
||||
try:
|
||||
cur.touch(int(anchor))
|
||||
except ValueError:
|
||||
pass
|
||||
open_idx = open_new(cur)
|
||||
cur.op_count += 1
|
||||
if inline:
|
||||
cur.payload_blocks[open_idx].append(inline)
|
||||
continue
|
||||
if op == "≔":
|
||||
size, lines = _parse_range(body)
|
||||
|
||||
rng = _OP_RANGE_NEW_RE.match(line)
|
||||
if rng:
|
||||
sigil = rng.group("sigil")
|
||||
a_str = rng.group("a")
|
||||
b_str = rng.group("b") or a_str
|
||||
inline = rng.group("inline")
|
||||
try:
|
||||
a = int(a_str)
|
||||
b = int(b_str)
|
||||
size = max(b - a + 1, 1)
|
||||
cur.touch(a)
|
||||
cur.touch(b)
|
||||
except ValueError:
|
||||
size = 1
|
||||
cur.op_anchors.append(a_str)
|
||||
if b_str != a_str:
|
||||
cur.op_anchors.append(b_str)
|
||||
cur.deleted_lines += size
|
||||
if lines is not None:
|
||||
cur.touch(lines[0])
|
||||
cur.touch(lines[1])
|
||||
for part in body.strip().split(".."):
|
||||
t = part.strip()
|
||||
if t:
|
||||
cur.op_anchors.append(t)
|
||||
cur.op_count += 1
|
||||
if sigil == "!":
|
||||
# Delete op: payload forbidden by the production parser; close.
|
||||
open_idx = None
|
||||
continue
|
||||
# sigil == ":" — replace.
|
||||
open_idx = open_new(cur)
|
||||
if inline:
|
||||
cur.payload_blocks[open_idx].append(inline)
|
||||
continue
|
||||
else:
|
||||
op_match = _LEGACY_OP_RE.match(line)
|
||||
if op_match:
|
||||
op = op_match.group(1)
|
||||
body = op_match.group(2)
|
||||
if op in ("«", "»"):
|
||||
anchor_trimmed = body.strip()
|
||||
if anchor_trimmed and anchor_trimmed not in ("BOF", "EOF"):
|
||||
cur.op_anchors.append(anchor_trimmed)
|
||||
line_no = _parse_legacy_anchor_line(body)
|
||||
if line_no is not None:
|
||||
cur.touch(line_no)
|
||||
open_idx = open_new(cur)
|
||||
cur.op_count += 1
|
||||
continue
|
||||
if op == "≔":
|
||||
size, lines = _parse_legacy_range(body)
|
||||
cur.deleted_lines += size
|
||||
if lines is not None:
|
||||
cur.touch(lines[0])
|
||||
cur.touch(lines[1])
|
||||
for part in body.strip().split(".."):
|
||||
t = part.strip()
|
||||
if t:
|
||||
cur.op_anchors.append(t)
|
||||
cur.op_count += 1
|
||||
open_idx = open_new(cur)
|
||||
continue
|
||||
|
||||
if open_idx is not None:
|
||||
cur.payload_blocks[open_idx].append(line)
|
||||
@@ -669,7 +749,7 @@ def _ingest_edit_call(rec, sf, seq, ts, call_id, arg_obj, arg_json) -> None:
|
||||
(sf, call_id, seq, ts, raw_input_len, EDIT_PARSER_VERSION)
|
||||
)
|
||||
|
||||
if not any(line.startswith("§") for line in input_str.lstrip("\ufeff").splitlines()):
|
||||
if not any(line.startswith(("¶", "§")) for line in input_str.lstrip("\ufeff").splitlines()):
|
||||
# Vim-mode or other shape — no sections to record.
|
||||
return
|
||||
|
||||
|
||||
Reference in New Issue
Block a user