feat: implemented canonical § hashlines with ≔/""/" ops in hashline parser
- Changed hashline format to canonical `§` section headers and `"`/`"`/`≔` operations across grammar, parser, and docs. - Reworked range and op parsing so single anchors are valid, legacy `-`/`-=` ops error, and empty `≔` payloads now delete ranges. - Removed legacy `HL_EDIT_SEP`, `$HSEP$`, and `hsep` plumbing, adopting raw payload lines. - Updated execution, input, renderer, and streaming flows to use `HL_FILE_PREFIX` and `HL_OP_CHARS` helpers. - Updated session-stats parsing to `§`/`≔`/`"`/`"` format, patch envelopes, and bumped parser versions. - Removed the hashline-separator benchmark script and its PI_HL_SEP job orchestration.
This commit is contained in:
@@ -1,191 +0,0 @@
|
||||
#!/usr/bin/env bun
|
||||
/**
|
||||
* Run `bun run bench:edit --edit-variant hashline` across the cartesian product
|
||||
* of `PI_HL_SEP` separator values and models, with a fixed concurrency cap.
|
||||
*
|
||||
* Each invocation writes its markdown report and a captured stdout/stderr log
|
||||
* into `runs/hashline-sep-<timestamp>/`.
|
||||
*
|
||||
* Usage:
|
||||
* bun scripts/bench-edit-hashline-sep.ts
|
||||
*/
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
|
||||
const SEPARATORS = ["~", "%", "÷", ">", ":"] as const;
|
||||
|
||||
const MODELS = [
|
||||
"openrouter/z-ai/glm-4.7:nitro",
|
||||
"openai/gpt-5.4-nano",
|
||||
"anthropic/claude-sonnet-4-6",
|
||||
] as const;
|
||||
|
||||
const CONCURRENCY = 3;
|
||||
const MAX_TASKS = "12";
|
||||
const VARIANT = "hashline";
|
||||
|
||||
const SEP_SLUGS: Record<string, string> = {
|
||||
"~": "tilde",
|
||||
"%": "pct",
|
||||
"÷": "div",
|
||||
">": "gt",
|
||||
":": "colon",
|
||||
};
|
||||
|
||||
function slugifyModel(model: string): string {
|
||||
return model.replace(/[^a-zA-Z0-9._-]+/g, "_");
|
||||
}
|
||||
|
||||
interface Job {
|
||||
sep: string;
|
||||
sepSlug: string;
|
||||
model: string;
|
||||
output: string;
|
||||
logPath: string;
|
||||
tag: string;
|
||||
}
|
||||
|
||||
interface JobResult {
|
||||
job: Job;
|
||||
exitCode: number | null;
|
||||
durationMs: number;
|
||||
}
|
||||
|
||||
const repoRoot = path.resolve(import.meta.dir, "..");
|
||||
const stamp = new Date().toISOString().replace(/[:.]/g, "-").replace(/Z$/, "Z");
|
||||
const runDir = path.join(repoRoot, "runs", `hashline-sep-${stamp}`);
|
||||
await fs.mkdir(runDir, { recursive: true });
|
||||
|
||||
const jobs: Job[] = [];
|
||||
for (const sep of SEPARATORS) {
|
||||
for (const model of MODELS) {
|
||||
const sepSlug = SEP_SLUGS[sep] ?? `sep_${sep.charCodeAt(0).toString(16)}`;
|
||||
const slug = `${sepSlug}__${slugifyModel(model)}`;
|
||||
jobs.push({
|
||||
sep,
|
||||
sepSlug,
|
||||
model,
|
||||
output: path.join(runDir, `${slug}.md`),
|
||||
logPath: path.join(runDir, `${slug}.log`),
|
||||
tag: `[${sepSlug} ${model}]`,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`Total runs: ${jobs.length} concurrency: ${CONCURRENCY}`);
|
||||
console.log(`Output dir: ${runDir}\n`);
|
||||
|
||||
const results: JobResult[] = [];
|
||||
let cursor = 0;
|
||||
let finished = 0;
|
||||
|
||||
async function pipeStream(
|
||||
stream: ReadableStream<Uint8Array>,
|
||||
sink: Bun.FileSink,
|
||||
tag: string,
|
||||
): Promise<void> {
|
||||
const reader = stream.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buf = "";
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
const text = decoder.decode(value, { stream: true });
|
||||
sink.write(text);
|
||||
buf += text;
|
||||
let nl = buf.indexOf("\n");
|
||||
while (nl !== -1) {
|
||||
const line = buf.slice(0, nl);
|
||||
buf = buf.slice(nl + 1);
|
||||
console.log(`${tag} ${line}`);
|
||||
nl = buf.indexOf("\n");
|
||||
}
|
||||
}
|
||||
const tail = decoder.decode();
|
||||
if (tail) {
|
||||
sink.write(tail);
|
||||
buf += tail;
|
||||
}
|
||||
if (buf) console.log(`${tag} ${buf}`);
|
||||
}
|
||||
|
||||
async function runJob(job: Job): Promise<JobResult> {
|
||||
const started = Date.now();
|
||||
const sink = Bun.file(job.logPath).writer();
|
||||
sink.write(`# cmd: PI_HL_SEP=${JSON.stringify(job.sep)} bun run bench:edit \\\n`);
|
||||
sink.write(
|
||||
`# --edit-variant ${VARIANT} --model ${job.model} --max-tasks ${MAX_TASKS} --output ${job.output}\n\n`,
|
||||
);
|
||||
|
||||
const proc = Bun.spawn({
|
||||
cmd: [
|
||||
"bun",
|
||||
"run",
|
||||
"bench:edit",
|
||||
"--edit-variant",
|
||||
VARIANT,
|
||||
"--model",
|
||||
job.model,
|
||||
"--max-tasks",
|
||||
MAX_TASKS,
|
||||
"--output",
|
||||
job.output,
|
||||
],
|
||||
cwd: repoRoot,
|
||||
env: { ...process.env, PI_HL_SEP: job.sep },
|
||||
stdout: "pipe",
|
||||
stderr: "pipe",
|
||||
});
|
||||
|
||||
await Promise.all([
|
||||
pipeStream(proc.stdout as ReadableStream<Uint8Array>, sink, job.tag),
|
||||
pipeStream(proc.stderr as ReadableStream<Uint8Array>, sink, job.tag),
|
||||
]);
|
||||
const exitCode = await proc.exited;
|
||||
await sink.end();
|
||||
return { job, exitCode, durationMs: Date.now() - started };
|
||||
}
|
||||
|
||||
async function worker(workerId: number): Promise<void> {
|
||||
while (true) {
|
||||
const idx = cursor++;
|
||||
if (idx >= jobs.length) return;
|
||||
const job = jobs[idx];
|
||||
console.log(`${job.tag} starting (worker ${workerId}, ${idx + 1}/${jobs.length})`);
|
||||
const result = await runJob(job);
|
||||
results.push(result);
|
||||
finished++;
|
||||
const status = result.exitCode === 0 ? "ok" : `FAIL exit=${result.exitCode}`;
|
||||
console.log(
|
||||
`${job.tag} ${status} in ${(result.durationMs / 1000).toFixed(1)}s [${finished}/${jobs.length}]`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const wallStart = Date.now();
|
||||
await Promise.all(Array.from({ length: CONCURRENCY }, (_, i) => worker(i + 1)));
|
||||
const wallMs = Date.now() - wallStart;
|
||||
|
||||
console.log("\n=== Summary ===");
|
||||
results.sort(
|
||||
(a, b) =>
|
||||
SEPARATORS.indexOf(a.job.sep as (typeof SEPARATORS)[number]) -
|
||||
SEPARATORS.indexOf(b.job.sep as (typeof SEPARATORS)[number]) ||
|
||||
MODELS.indexOf(a.job.model as (typeof MODELS)[number]) -
|
||||
MODELS.indexOf(b.job.model as (typeof MODELS)[number]),
|
||||
);
|
||||
for (const r of results) {
|
||||
const status = r.exitCode === 0 ? "ok " : `FAIL`;
|
||||
console.log(
|
||||
`${status} ${r.job.sepSlug.padEnd(7)} ${r.job.model.padEnd(40)} ${(r.durationMs / 1000)
|
||||
.toFixed(1)
|
||||
.padStart(6)}s ${path.relative(repoRoot, r.job.output)}`,
|
||||
);
|
||||
}
|
||||
console.log(`\nWall time: ${(wallMs / 1000).toFixed(1)}s`);
|
||||
|
||||
const failures = results.filter(r => r.exitCode !== 0);
|
||||
if (failures.length > 0) {
|
||||
console.log(`${failures.length} job(s) failed`);
|
||||
process.exit(1);
|
||||
}
|
||||
@@ -47,7 +47,7 @@ All tables are prefixed `ss_` to avoid collision with `packages/stats`.
|
||||
|`ss_assistant_msgs`|per assistant message text + thinking blobs and token counts|
|
||||
|`ss_user_msgs`|per user message text and token count|
|
||||
|`ss_edit_calls`|per `edit` call: `success`, `warnings`, `raw_input_len`|
|
||||
|`ss_edit_sections`|per `@PATH` section in an edit; precomputed `longest_repeat_*`, `dup_anchors`|
|
||||
|`ss_edit_sections`|per `§PATH` section in an edit; precomputed `longest_repeat_*`, `dup_anchors`|
|
||||
|
||||
Indexes on `(tool_name, timestamp)` and `(session_file, seq)` make per-tool
|
||||
aggregations and ordered session walks cheap.
|
||||
|
||||
@@ -53,13 +53,12 @@ SCRIPT_RUN_RE = re.compile(
|
||||
"]{2,}"
|
||||
)
|
||||
|
||||
HEADER_RE = re.compile(r"^(?:@(?P<at>\S.*)|\*\*\* Update File:\s+(?P<upd>\S.*))\s*$")
|
||||
HEADER_RE = re.compile(r"^(?:§+(?P<hl>.*)|\*\*\* Update File:\s+(?P<upd>\S.*))\s*$")
|
||||
BEGIN_PATCH_RE = re.compile(r"^\*\*\* Begin Patch\s*$")
|
||||
END_PATCH_RE = re.compile(r"^\*\*\* End Patch\s*$")
|
||||
INSERT_RE = re.compile(r"^[+<]\s*(?P<anchor>BOF|EOF|[1-9][0-9]*[A-Za-z]{2})(?:\s*~(?P<inline>.*))?\s*$")
|
||||
INSERT_RE = re.compile(r"^(?P<op>[«»])\s*(?P<anchor>BOF|EOF|[1-9][0-9]*[A-Za-z]{2})\s*$")
|
||||
RANGE_RE = re.compile(r"(?P<a>[1-9][0-9]*[A-Za-z]{2})(?:\.\.(?P<b>[1-9][0-9]*[A-Za-z]{2}))?")
|
||||
DELETE_RE = re.compile(r"^-\s*(?P<range>[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$")
|
||||
REPLACE_RE = re.compile(r"^=\s*(?P<range>[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$")
|
||||
REPLACE_RE = re.compile(r"^≔\s*(?P<range>[1-9][0-9]*[A-Za-z]{2}(?:\.\.[1-9][0-9]*[A-Za-z]{2})?)\s*$")
|
||||
|
||||
JSON_DECODER = json.JSONDecoder()
|
||||
|
||||
@@ -457,7 +456,7 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo
|
||||
if header:
|
||||
if needs_payload and not saw_required_payload:
|
||||
break
|
||||
target = (header.group("at") or header.group("upd") or "").strip()
|
||||
target = (header.group("hl") or header.group("upd") or "").strip()
|
||||
cur = EditSection(target_file=target)
|
||||
sections.append(cur)
|
||||
parsed_end = end
|
||||
@@ -469,9 +468,7 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo
|
||||
if cur is None:
|
||||
break
|
||||
|
||||
if line.startswith("~"):
|
||||
if not payload_allowed:
|
||||
break
|
||||
if payload_allowed and line[:1] not in {"«", "»", "≔", "§"}:
|
||||
cur.payload_lines += 1
|
||||
parsed_end = end
|
||||
saw_required_payload = True
|
||||
@@ -490,35 +487,17 @@ def parse_edit_boundary(text: str, *, legacy_loose_tail: bool = False) -> EditBo
|
||||
if needs_payload and not saw_required_payload:
|
||||
break
|
||||
|
||||
trimmed = line.lstrip()
|
||||
ins = INSERT_RE.match(trimmed)
|
||||
ins = INSERT_RE.match(line)
|
||||
if ins:
|
||||
cur.op_count += 1
|
||||
inline = ins.group("inline")
|
||||
if inline is None:
|
||||
needs_payload = True
|
||||
payload_allowed = True
|
||||
saw_required_payload = False
|
||||
# Not complete until at least one payload line appears.
|
||||
else:
|
||||
cur.payload_lines += 1
|
||||
parsed_end = end
|
||||
needs_payload = False
|
||||
payload_allowed = False
|
||||
saw_required_payload = False
|
||||
continue
|
||||
|
||||
dele = DELETE_RE.match(trimmed)
|
||||
if dele:
|
||||
cur.op_count += 1
|
||||
cur.deleted_lines += range_deleted_lines(dele.group("range"))
|
||||
parsed_end = end
|
||||
needs_payload = False
|
||||
payload_allowed = False
|
||||
needs_payload = True
|
||||
payload_allowed = True
|
||||
saw_required_payload = False
|
||||
# Not complete until at least one payload line appears.
|
||||
continue
|
||||
|
||||
repl = REPLACE_RE.match(trimmed)
|
||||
|
||||
repl = REPLACE_RE.match(line)
|
||||
if repl:
|
||||
cur.op_count += 1
|
||||
cur.deleted_lines += range_deleted_lines(repl.group("range"))
|
||||
|
||||
@@ -15,7 +15,7 @@ Schema (all tables prefixed `ss_` to avoid collision with packages/stats):
|
||||
ss_assistant_msgs one row per assistant message (text + thinking blobs)
|
||||
ss_user_msgs one row per user message (text blob)
|
||||
ss_edit_calls one row per edit toolCall (success + warnings paired in)
|
||||
ss_edit_sections one row per @PATH section inside an edit toolCall, with
|
||||
ss_edit_sections one row per §PATH section inside an edit toolCall, with
|
||||
precomputed detector outputs (longest_repeat_*, dup_anchors)
|
||||
|
||||
Run:
|
||||
@@ -52,11 +52,11 @@ except ImportError:
|
||||
SESSIONS_ROOT = Path.home() / ".omp" / "agent" / "sessions"
|
||||
DB_PATH = Path.home() / ".omp" / "stats.db"
|
||||
TOKENIZER_NAME = "o200k_base"
|
||||
SCHEMA_VERSION = 2
|
||||
SCHEMA_VERSION = 3
|
||||
# Bump whenever parse_hashline_input / find_longest_repeat / duplicated_anchors
|
||||
# / looks_successful / extract_warnings semantics change. Bump invalidates
|
||||
# previously-stored ss_edit_* rows on next sync.
|
||||
EDIT_PARSER_VERSION = 1
|
||||
EDIT_PARSER_VERSION = 4
|
||||
|
||||
SCHEMA_SQL = """
|
||||
CREATE TABLE IF NOT EXISTS ss_sessions (
|
||||
@@ -231,6 +231,8 @@ def batch_count_tokens(strings: list[str]) -> list[int]:
|
||||
|
||||
_RANGE_RE = re.compile(r"^\s*(\d+)[a-z*]+(?:\.\.(\d+)[a-z*]+)?\s*$")
|
||||
_SINGLE_ANCHOR_RE = re.compile(r"^\s*(\d+)[a-z*]+\s*$")
|
||||
_HASHLINE_OP_RE = re.compile(r"^([«»≔])\s*(\S+)\s*$")
|
||||
_HASHLINE_ENVELOPE_MARKERS = {"*** Begin Patch", "*** End Patch", "*** Abort"}
|
||||
|
||||
|
||||
def _parse_range(raw: str) -> tuple[int, tuple[int, int] | None]:
|
||||
@@ -290,66 +292,57 @@ def parse_hashline_input(input_str: str) -> list[EditSection]:
|
||||
|
||||
for raw_line in input_str.split("\n"):
|
||||
line = raw_line[:-1] if raw_line.endswith("\r") else raw_line
|
||||
trimmed_end = line.rstrip()
|
||||
|
||||
if line.startswith("@"):
|
||||
if trimmed_end in _HASHLINE_ENVELOPE_MARKERS:
|
||||
if trimmed_end != "*** Begin Patch":
|
||||
break
|
||||
continue
|
||||
|
||||
if line.startswith("§"):
|
||||
if cur is not None:
|
||||
sections.append(cur)
|
||||
cur = EditSection(target_file=line[1:].strip())
|
||||
prefix_end = 0
|
||||
while prefix_end < len(line) and line[prefix_end] == "§":
|
||||
prefix_end += 1
|
||||
cur = EditSection(target_file=line[prefix_end:].strip())
|
||||
open_idx = None
|
||||
continue
|
||||
if cur is None:
|
||||
continue
|
||||
|
||||
if line.startswith("~"):
|
||||
payload = line[1:]
|
||||
if open_idx is None:
|
||||
op_match = _HASHLINE_OP_RE.match(line)
|
||||
if op_match:
|
||||
op = op_match.group(1)
|
||||
body = op_match.group(2)
|
||||
if op in ("«", "»"):
|
||||
anchor_trimmed = body.strip()
|
||||
if anchor_trimmed and anchor_trimmed not in ("BOF", "EOF"):
|
||||
cur.op_anchors.append(anchor_trimmed)
|
||||
line_no = _parse_anchor_line(body)
|
||||
if line_no is not None:
|
||||
cur.touch(line_no)
|
||||
open_idx = open_new(cur)
|
||||
cur.payload_blocks[open_idx].append(payload)
|
||||
continue
|
||||
cur.op_count += 1
|
||||
continue
|
||||
if op == "≔":
|
||||
size, lines = _parse_range(body)
|
||||
cur.deleted_lines += size
|
||||
if lines is not None:
|
||||
cur.touch(lines[0])
|
||||
cur.touch(lines[1])
|
||||
for part in body.strip().split(".."):
|
||||
t = part.strip()
|
||||
if t:
|
||||
cur.op_anchors.append(t)
|
||||
cur.op_count += 1
|
||||
open_idx = open_new(cur)
|
||||
continue
|
||||
|
||||
trimmed = line.lstrip()
|
||||
if not trimmed:
|
||||
if open_idx is not None:
|
||||
cur.payload_blocks[open_idx].append(line)
|
||||
elif not line.strip():
|
||||
continue
|
||||
op = trimmed[0]
|
||||
if op in ("+", "<"):
|
||||
body = trimmed[1:].lstrip()
|
||||
if "~" in body:
|
||||
anchor_part, tail = body.split("~", 1)
|
||||
else:
|
||||
anchor_part, tail = body, None
|
||||
anchor_trimmed = anchor_part.strip()
|
||||
if anchor_trimmed and anchor_trimmed not in ("BOF", "EOF"):
|
||||
cur.op_anchors.append(anchor_trimmed)
|
||||
line_no = _parse_anchor_line(anchor_part)
|
||||
if line_no is not None:
|
||||
cur.touch(line_no)
|
||||
if tail is not None:
|
||||
# Inline `+ ANCHOR~text`: replaces a single line.
|
||||
if open_idx is None:
|
||||
open_idx = open_new(cur)
|
||||
cur.payload_blocks[open_idx].append(tail)
|
||||
cur.deleted_lines += 1
|
||||
open_idx = None
|
||||
else:
|
||||
open_idx = open_new(cur)
|
||||
cur.op_count += 1
|
||||
elif op in ("-", "="):
|
||||
body = trimmed[1:].lstrip()
|
||||
size, lines = _parse_range(body)
|
||||
cur.deleted_lines += size
|
||||
if lines is not None:
|
||||
cur.touch(lines[0])
|
||||
cur.touch(lines[1])
|
||||
for part in body.strip().split(".."):
|
||||
t = part.strip()
|
||||
if t:
|
||||
cur.op_anchors.append(t)
|
||||
cur.op_count += 1
|
||||
if op == "=":
|
||||
open_idx = open_new(cur)
|
||||
else:
|
||||
open_idx = None
|
||||
# else: blank / unrecognized — keep payload state.
|
||||
|
||||
if cur is not None:
|
||||
sections.append(cur)
|
||||
@@ -676,7 +669,7 @@ def _ingest_edit_call(rec, sf, seq, ts, call_id, arg_obj, arg_json) -> None:
|
||||
(sf, call_id, seq, ts, raw_input_len, EDIT_PARSER_VERSION)
|
||||
)
|
||||
|
||||
if not input_str.lstrip().startswith("@"):
|
||||
if not any(line.startswith("§") for line in input_str.lstrip("\ufeff").splitlines()):
|
||||
# Vim-mode or other shape — no sections to record.
|
||||
return
|
||||
|
||||
|
||||
Reference in New Issue
Block a user