feat: implemented canonical § hashlines with ≔/""/" ops in hashline parser

- Changed hashline format to canonical `§` section headers and `"`/`"`/`≔` operations across grammar, parser, and docs.
- Reworked range and op parsing so single anchors are valid, legacy `-`/`-=` ops error, and empty `≔` payloads now delete ranges.
- Removed legacy `HL_EDIT_SEP`, `$HSEP$`, and `hsep` plumbing, adopting raw payload lines.
- Updated execution, input, renderer, and streaming flows to use `HL_FILE_PREFIX` and `HL_OP_CHARS` helpers.
- Updated session-stats parsing to `§`/`≔`/`"`/`"` format, patch envelopes, and bumped parser versions.
- Removed the hashline-separator benchmark script and its PI_HL_SEP job orchestration.
This commit is contained in:
can1357
2026-05-22 17:45:39 +09:00
committed by Can Bölük
parent f29e733b67
commit ebd2f77770
23 changed files with 493 additions and 888 deletions
+1 -1
View File
@@ -47,7 +47,7 @@ All tables are prefixed `ss_` to avoid collision with `packages/stats`.
|`ss_assistant_msgs`|per assistant message text + thinking blobs and token counts|
|`ss_user_msgs`|per user message text and token count|
|`ss_edit_calls`|per `edit` call: `success`, `warnings`, `raw_input_len`|
|`ss_edit_sections`|per `@PATH` section in an edit; precomputed `longest_repeat_*`, `dup_anchors`|
|`ss_edit_sections`|per `§PATH` section in an edit; precomputed `longest_repeat_*`, `dup_anchors`|
Indexes on `(tool_name, timestamp)` and `(session_file, seq)` make per-tool
aggregations and ordered session walks cheap.
+11 -32
View File
@@ -53,13 +53,12 @@ SCRIPT_RUN_RE = re.compile(
"]{2,}"
)
HEADER_RE = re.compile(r"^(?:@(?P<at>\S.*)|\*\*\* Update File:\s+(?P<upd>\S.*))\s*$")
HEADER_RE = re.compile(r"^(?:§+(?P<hl>.*)|\*\*\* Update File:\s+(?P<upd>\S.*))\s*$")
BEGIN_PATCH_RE = re.compile(r"^\*\*\* Begin Patch\s*$")
END_PATCH_RE = re.compile(r"^\*\*\* End Patch\s*$")
INSERT_RE = re.compile(r"^[+<]\s*(?P<anchor>BOF|EOF|[1-9][0-9]*[A-Za-z]{2})(?:\s*~(?P<inline>.*))?\s*$")
INSERT_RE = re.compile(r"^(?P<op>[«»])\s*(?P<anchor>BOF|EOF|[1-9][0-9]*[A-Za-z]{2})\s*$")
RANGE_RE = re.compile(r"(?P<a>[1-9][0-9]*[A-Za-z]{2})(?:\.\.(?P<b>[1-9][0-9]*[A-Za-z]{2}))?")
DELETE_RE = re.compile(r"^-\s*(?P<range>[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$")
REPLACE_RE = re.compile(r"^=\s*(?P<range>[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$")
REPLACE_RE = re.compile(r"^≔\s*(?P<range>[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$")
JSON_DECODER = json.JSONDecoder()
@@ -457,7 +456,7 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo
if header:
if needs_payload and not saw_required_payload:
break
target = (header.group("at") or header.group("upd") or "").strip()
target = (header.group("hl") or header.group("upd") or "").strip()
cur = EditSection(target_file=target)
sections.append(cur)
parsed_end = end
@@ -469,9 +468,7 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo
if cur is None:
break
if line.startswith("~"):
if not payload_allowed:
break
if payload_allowed and line[:1] not in {"«", "»", "≔", "§"}:
cur.payload_lines += 1
parsed_end = end
saw_required_payload = True
@@ -490,35 +487,17 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo
if needs_payload and not saw_required_payload:
break
trimmed = line.lstrip()
ins = INSERT_RE.match(trimmed)
ins = INSERT_RE.match(line)
if ins:
cur.op_count += 1
inline = ins.group("inline")
if inline is None:
needs_payload = True
payload_allowed = True
saw_required_payload = False
# Not complete until at least one payload line appears.
else:
cur.payload_lines += 1
parsed_end = end
needs_payload = False
payload_allowed = False
saw_required_payload = False
continue
dele = DELETE_RE.match(trimmed)
if dele:
cur.op_count += 1
cur.deleted_lines += range_deleted_lines(dele.group("range"))
parsed_end = end
needs_payload = False
payload_allowed = False
needs_payload = True
payload_allowed = True
saw_required_payload = False
# Not complete until at least one payload line appears.
continue
repl = REPLACE_RE.match(trimmed)
repl = REPLACE_RE.match(line)
if repl:
cur.op_count += 1
cur.deleted_lines += range_deleted_lines(repl.group("range"))
+46 -53
View File
@@ -15,7 +15,7 @@ Schema (all tables prefixed `ss_` to avoid collision with packages/stats):
ss_assistant_msgs one row per assistant message (text + thinking blobs)
ss_user_msgs one row per user message (text blob)
ss_edit_calls one row per edit toolCall (success + warnings paired in)
ss_edit_sections one row per @PATH section inside an edit toolCall, with
ss_edit_sections one row per §PATH section inside an edit toolCall, with
precomputed detector outputs (longest_repeat_*, dup_anchors)
Run:
@@ -52,11 +52,11 @@ except ImportError:
SESSIONS_ROOT = Path.home() / ".omp" / "agent" / "sessions"
DB_PATH = Path.home() / ".omp" / "stats.db"
TOKENIZER_NAME = "o200k_base"
SCHEMA_VERSION = 2
SCHEMA_VERSION = 3
# Bump whenever parse_hashline_input / find_longest_repeat / duplicated_anchors
# / looks_successful / extract_warnings semantics change. Bump invalidates
# previously-stored ss_edit_* rows on next sync.
EDIT_PARSER_VERSION = 1
EDIT_PARSER_VERSION = 4
SCHEMA_SQL = """
CREATE TABLE IF NOT EXISTS ss_sessions (
@@ -231,6 +231,8 @@ def batch_count_tokens(strings: list[str]) -> list[int]:
_RANGE_RE = re.compile(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$")
_SINGLE_ANCHOR_RE = re.compile(r"^\s*(\d+)[a-z*]+\s*$")
_HASHLINE_OP_RE = re.compile(r"^([«»≔])\s*(\S+)\s*$")
_HASHLINE_ENVELOPE_MARKERS = {"*** Begin Patch", "*** End Patch", "*** Abort"}
def _parse_range(raw: str) -> tuple[int, tuple[int, int] | None]:
@@ -290,66 +292,57 @@ def parse_hashline_input(input_str: str) -> list[EditSection]:
for raw_line in input_str.split("\n"):
line = raw_line[:-1] if raw_line.endswith("\r") else raw_line
trimmed_end = line.rstrip()
if line.startswith("@"):
if trimmed_end in _HASHLINE_ENVELOPE_MARKERS:
if trimmed_end != "*** Begin Patch":
break
continue
if line.startswith("§"):
if cur is not None:
sections.append(cur)
cur = EditSection(target_file=line[1:].strip())
prefix_end = 0
while prefix_end < len(line) and line[prefix_end] == "§":
prefix_end += 1
cur = EditSection(target_file=line[prefix_end:].strip())
open_idx = None
continue
if cur is None:
continue
if line.startswith("~"):
payload = line[1:]
if open_idx is None:
op_match = _HASHLINE_OP_RE.match(line)
if op_match:
op = op_match.group(1)
body = op_match.group(2)
if op in ("«", "»"):
anchor_trimmed = body.strip()
if anchor_trimmed and anchor_trimmed not in ("BOF", "EOF"):
cur.op_anchors.append(anchor_trimmed)
line_no = _parse_anchor_line(body)
if line_no is not None:
cur.touch(line_no)
open_idx = open_new(cur)
cur.payload_blocks[open_idx].append(payload)
continue
cur.op_count += 1
continue
if op == "≔":
size, lines = _parse_range(body)
cur.deleted_lines += size
if lines is not None:
cur.touch(lines[0])
cur.touch(lines[1])
for part in body.strip().split(".."):
t = part.strip()
if t:
cur.op_anchors.append(t)
cur.op_count += 1
open_idx = open_new(cur)
continue
trimmed = line.lstrip()
if not trimmed:
if open_idx is not None:
cur.payload_blocks[open_idx].append(line)
elif not line.strip():
continue
op = trimmed[0]
if op in ("+", "<"):
body = trimmed[1:].lstrip()
if "~" in body:
anchor_part, tail = body.split("~", 1)
else:
anchor_part, tail = body, None
anchor_trimmed = anchor_part.strip()
if anchor_trimmed and anchor_trimmed not in ("BOF", "EOF"):
cur.op_anchors.append(anchor_trimmed)
line_no = _parse_anchor_line(anchor_part)
if line_no is not None:
cur.touch(line_no)
if tail is not None:
# Inline `+ ANCHOR~text`: replaces a single line.
if open_idx is None:
open_idx = open_new(cur)
cur.payload_blocks[open_idx].append(tail)
cur.deleted_lines += 1
open_idx = None
else:
open_idx = open_new(cur)
cur.op_count += 1
elif op in ("-", "="):
body = trimmed[1:].lstrip()
size, lines = _parse_range(body)
cur.deleted_lines += size
if lines is not None:
cur.touch(lines[0])
cur.touch(lines[1])
for part in body.strip().split(".."):
t = part.strip()
if t:
cur.op_anchors.append(t)
cur.op_count += 1
if op == "=":
open_idx = open_new(cur)
else:
open_idx = None
# else: blank / unrecognized — keep payload state.
if cur is not None:
sections.append(cur)
@@ -676,7 +669,7 @@ def _ingest_edit_call(rec, sf, seq, ts, call_id, arg_obj, arg_json) -> None:
(sf, call_id, seq, ts, raw_input_len, EDIT_PARSER_VERSION)
)
if not input_str.lstrip().startswith("@"):
if not any(line.startswith("§") for line in input_str.lstrip("\ufeff").splitlines()):
# Vim-mode or other shape — no sections to record.
return