fix: chatgpt really doesnt like sandwitching lark

This commit is contained in:
can1357
2026-05-02 04:52:02 +02:00
parent f0e9a830ff
commit d064170c56
5 changed files with 229 additions and 263 deletions
+24 -30
View File
@@ -1,43 +1,37 @@
%import common.LF
%import common.WS_INLINE
// Strict canonical surface for the eval tool. Callers MUST emit exactly this
// form. The runtime parser accepts additional lenient shapes (positional
// title/duration, alias keys, long-form lang tokens, mixed casing, fence
// runs of any length ≥ 3, etc.) but those are fallback only and MUST NOT
// be relied on.
// Canonical Eval input. Each cell is introduced by a header line:
//
// Each cell is a fenced code block opened and closed by exactly three
// (or exactly five) backticks or tildes — five lets callers nest a 3-char
// fence inside a cell verbatim. The opening fence carries an optional info
// string with up to four parts, IN THIS ORDER:
// ===== <info> =====
//
// lang? id_attr? t_attr? rst_attr?
// where each side is at least 5 equal signs. The info between the bars is
// a list of space-separated tokens, all optional, in any order:
//
// where:
// lang = "py" | "js" | "ts"
// id_attr = id="..." (double-quoted cell id)
// t_attr = t=<duration> (bare integer with optional ms/s/m unit)
// rst_attr= rst=0|1 (per-language kernel reset for this cell)
// py | js | ts language for this cell
// py:"..." | js:"..." | ts:"..." language plus title shorthand
// id:"..." cell title (when language unchanged)
// t:<digits>(ms|s|m)? per-cell timeout (default 30s)
// rst reset this language's kernel before running
//
// Everything between one header line and the next (or end of input) is
// the cell's code, verbatim. The runtime additionally accepts content
// before the first header as an implicit default-language cell, but that
// is lenient fallback only and MUST NOT be relied on.
start: cell+
cell: header LF code_line*
cell: backtick_cell | tilde_cell
header: BAR (WS_INLINE attr)+ WS_INLINE BAR
| BAR WS_INLINE? BAR
backtick_cell: BACKTICKS info? LF code_line* BACKTICKS LF
tilde_cell: TILDES info? LF code_line* TILDES LF
info: lang (WS_INLINE id_attr)? (WS_INLINE t_attr)? (WS_INLINE rst_attr)?
| id_attr (WS_INLINE t_attr)? (WS_INLINE rst_attr)?
| t_attr (WS_INLINE rst_attr)?
| rst_attr
lang: "py" | "js" | "ts"
id_attr: "id=" /"[^"\r\n]*"/
t_attr: "t=" /\d+(ms|s|m)?/
rst_attr: "rst=" /[01]/
attr: LANG_TITLE | LANG | ID_ATTR | T_ATTR | RST_FLAG
code_line: /[^\r\n]*/ LF
BACKTICKS: "```" | "`````"
TILDES: "~~~" | "~~~~~"
BAR: /={5,}/
LANG: "py" | "js" | "ts"
LANG_TITLE: ("py" | "js" | "ts") ":\"" /[^"\r\n]*/ "\""
ID_ATTR: "id:\"" /[^"\r\n]*/ "\""
T_ATTR: "t:" /\d+(ms|s|m)?/
RST_FLAG: "rst"
+129 -129
View File
@@ -1,6 +1,6 @@
import type { EvalLanguage } from "./types";
export type EvalLanguageOrigin = "default" | "fence";
export type EvalLanguageOrigin = "default" | "header";
export interface ParsedEvalCell {
index: number;
@@ -19,11 +19,11 @@ export interface ParsedEvalInput {
const DEFAULT_TIMEOUT_MS = 30_000;
/**
* Canonical fenced-language tokens we map onto our two backends. Matched
* case-insensitively. Anything else found in a fence info string is treated as
* a title fragment rather than a language; this is intentional fallback
* behaviour and MUST NOT be advertised in the tool's prompt — the lark grammar
* describes the canonical surface we encourage callers to emit.
* Canonical language tokens we map onto our two backends. Matched
* case-insensitively. Unknown tokens are treated as title fragments rather
* than languages; this is intentional fallback behaviour and MUST NOT be
* advertised in the tool's prompt — the lark grammar describes the
* canonical surface we encourage callers to emit.
*/
const LANGUAGE_ALIASES: Record<string, EvalLanguage> = {
py: "python",
@@ -41,8 +41,8 @@ function resolveLanguageAlias(token: string): EvalLanguage | undefined {
}
/**
* Map an attribute key (from `key=value` in a fence info string) to one of
* the three canonical roles. Canonical keys: `id`, `t`, `rst`. Fallback
* Map an attribute key (from `key:value` or bare `key` in a header) to one
* of the three canonical roles. Canonical keys: `id`, `t`, `rst`. Fallback
* aliases — accepted but not advertised in the prompt — cover common
* synonyms the LLM is likely to reach for instead of the short canonical.
*/
@@ -57,29 +57,22 @@ function classifyAttrKey(key: string): "id" | "t" | "rst" | null {
return null;
}
interface RawBlock {
type: "raw";
lines: string[];
startLine: number;
}
interface FencedBlock {
type: "fenced";
info: string;
codeLines: string[];
startLine: number;
}
type Block = RawBlock | FencedBlock;
interface FenceInfo {
interface HeaderInfo {
language?: EvalLanguage;
title?: string;
timeoutMs?: number;
reset?: boolean;
}
const ATTR_TOKEN_RE = /^([a-zA-Z][\w-]*)=(?:"([^"]*)"|'([^']*)'|(.*))$/;
/**
* Match a header line: `={5,} <info>? ={5,}`. Both bars MUST be on the
* same line and each MUST be at least five equal signs (lengths need not
* match — a 5/6 split is fine).
*/
const HEADER_RE = /^={5,}([^=].*?)?={5,}\s*$/;
const EMPTY_HEADER_RE = /^={5,}\s*$/;
const ATTR_TOKEN_RE = /^([a-zA-Z][\w-]*)(?::(?:"([^"]*)"|'([^']*)'|(.*)))?$/;
const DURATION_TOKEN_RE = /^\d+(?:ms|s|m)?$/;
function parseDurationMs(raw: string, lineNumber: number): number {
@@ -111,24 +104,26 @@ function trimOuterBlankLines(lines: string[]): string[] {
return lines.slice(start, end);
}
function parseFenceOpener(line: string): { char: "`" | "~"; count: number; info: string } | null {
const opener = /^(`{3,}|~{3,})(.*)$/.exec(line);
if (!opener) return null;
const run = opener[1];
return { char: run[0] as "`" | "~", count: run.length, info: opener[2].trim() };
}
function isFenceCloser(line: string, char: "`" | "~", minCount: number): boolean {
let count = 0;
while (count < line.length && line[count] === char) count++;
if (count < minCount) return false;
return line.slice(count).trim() === "";
/**
* Detect whether a line is a cell header. Returns the info string between
* the two bar runs (trimmed) when it is, or `null` otherwise. An empty
* header (`===== =====` or just `=====`) yields an empty info string.
*
* A line that contains text but only one bar (e.g. `===== title`) is NOT
* a header — it's normal code that happens to start with equal signs.
*/
function parseHeaderLine(line: string): string | null {
if (EMPTY_HEADER_RE.test(line)) return "";
const match = HEADER_RE.exec(line);
if (!match) return null;
return (match[1] ?? "").trim();
}
/**
* Tokenize a fence info string while preserving content inside matching
* single or double quotes as a single token. The opening and closing quote
* characters are kept verbatim so attribute parsing can strip them later.
* Tokenize a header info string while preserving content inside matching
* single or double quotes as a single token. The opening and closing
* quote characters are kept verbatim so attribute parsing can strip them
* later.
*/
function tokenizeInfoString(info: string): string[] {
const tokens: string[] = [];
@@ -161,71 +156,83 @@ function tokenizeInfoString(info: string): string[] {
}
/**
* Decode a fence info string into language, title, timeout, and reset flag.
* Decode a header info string into language, title, timeout, and reset flag.
*
* Layout (positional → kv, all optional):
* `<lang>? <duration>? <(title-fragment | key=value)>*`
* Token forms (all optional, any order):
* - `py` / `js` / `ts` bare language
* - `py:"..."` / `js:"..."` / `ts:"..."` language + title shorthand
* - `id:"..."` cell title
* - `t:<duration>` per-cell timeout
* - `<duration>` bare positional duration (lenient)
* - `rst` reset flag
* - `rst:true|false` reset flag with explicit value
*
* Canonical attribute keys (the only ones surfaced in the lark grammar):
* - `id` → cell title
* - `t` → per-cell timeout
* - `rst` → boolean reset for this cell's kernel
*
* Lenient fallback aliases (NOT advertised in the prompt; we silently accept
* them when the LLM reaches for a more familiar key):
* Fallback aliases (accepted but not advertised in the prompt):
* - id: title, name, cell, file, label
* - t: timeout, duration, time
* - rst: reset
*
* Truly unknown keys are silently dropped. First occurrence wins when a key
* is repeated (canonical or alias).
*
* - First token is consumed as a language alias when it matches one; otherwise
* it falls through to the title-fragment branch and the cell inherits the
* surrounding language.
* - The first remaining duration-shaped token (e.g. `15s`, `500ms`, `2m`,
* `30`) becomes the positional timeout. The `t=` attribute always wins.
* - Anything else accumulates as positional title fragments joined by spaces.
* Truly unknown keys are silently dropped. First occurrence wins when a
* key is repeated (canonical or alias). Anything that doesn't classify
* accumulates as a positional title fragment joined by spaces.
*/
function parseFenceInfo(info: string, lineNumber: number): FenceInfo {
const tokens = tokenizeInfoString(info.trim());
function parseHeaderInfo(info: string, lineNumber: number): HeaderInfo {
const tokens = tokenizeInfoString(info);
if (tokens.length === 0) return {};
let language: EvalLanguage | undefined;
let titleAttr: string | undefined;
let positionalDurationMs: number | undefined;
const titleParts: string[] = [];
let idAttr: string | undefined;
let tAttr: string | undefined;
let rstAttr: string | undefined;
let bareReset = false;
const titleParts: string[] = [];
for (const token of tokens) {
// Bare reset flag.
if (RST_KEYS.has(token.toLowerCase())) {
bareReset = true;
continue;
}
for (let idx = 0; idx < tokens.length; idx++) {
const token = tokens[idx];
const attrMatch = ATTR_TOKEN_RE.exec(token);
if (attrMatch) {
if (attrMatch && token.includes(":")) {
const key = attrMatch[1].toLowerCase();
const value = attrMatch[2] ?? attrMatch[3] ?? attrMatch[4] ?? "";
// Language-with-title shorthand: `py:"foo"` etc.
const langCandidate = resolveLanguageAlias(key);
if (langCandidate) {
if (language === undefined) language = langCandidate;
if (titleAttr === undefined && value !== "") titleAttr = value;
continue;
}
const role = classifyAttrKey(key);
if (role === "id" && idAttr === undefined) idAttr = value;
if (role === "id" && titleAttr === undefined) titleAttr = value;
else if (role === "t" && tAttr === undefined) tAttr = value;
else if (role === "rst" && rstAttr === undefined) rstAttr = value;
// unknown / repeated keys silently dropped
continue;
}
if (idx === 0) {
const lang = resolveLanguageAlias(token);
if (lang) {
language = lang;
continue;
}
// Bare language token (no colon).
const lang = resolveLanguageAlias(token);
if (lang && language === undefined) {
language = lang;
continue;
}
// Bare positional duration (lenient — `t:` is canonical).
if (positionalDurationMs === undefined && DURATION_TOKEN_RE.test(token)) {
positionalDurationMs = parseDurationMs(token, lineNumber);
continue;
}
titleParts.push(token);
}
const explicitTitle = (idAttr ?? "").trim();
const explicitTitle = (titleAttr ?? "").trim();
const positionalTitle = titleParts.join(" ").trim();
const title = explicitTitle.length > 0 ? explicitTitle : positionalTitle.length > 0 ? positionalTitle : undefined;
@@ -243,55 +250,13 @@ function parseFenceInfo(info: string, lineNumber: number): FenceInfo {
throw new Error(`Eval line ${lineNumber}: invalid rst value \`${rstAttr}\`; use true or false.`);
}
reset = parsed;
} else if (bareReset) {
reset = true;
}
return { language, title, timeoutMs, reset };
}
/**
* Walk normalized lines and split into top-level fenced blocks and raw
* (between/around fences) blocks. Unclosed fences are leniently closed at
* end-of-input. Raw blocks with only blank lines are dropped.
*/
function splitIntoBlocks(lines: string[]): Block[] {
const blocks: Block[] = [];
let i = 0;
while (i < lines.length) {
const line = lines[i];
const opener = parseFenceOpener(line);
if (opener) {
const fenceStart = i + 1; // 1-indexed line number of opener
const codeLines: string[] = [];
let j = i + 1;
let closed = false;
while (j < lines.length) {
if (isFenceCloser(lines[j], opener.char, opener.count)) {
closed = true;
break;
}
codeLines.push(lines[j]);
j++;
}
blocks.push({ type: "fenced", info: opener.info, codeLines, startLine: fenceStart });
i = closed ? j + 1 : j;
} else {
const rawStart = i + 1;
const rawLines: string[] = [line];
let j = i + 1;
while (j < lines.length && !parseFenceOpener(lines[j])) {
rawLines.push(lines[j]);
j++;
}
const trimmed = trimOuterBlankLines(rawLines);
if (trimmed.length > 0) {
blocks.push({ type: "raw", lines: trimmed, startLine: rawStart });
}
i = j;
}
}
return blocks;
}
interface ExpansionState {
language: EvalLanguage;
languageOrigin: EvalLanguageOrigin;
@@ -300,34 +265,69 @@ interface ExpansionState {
export function parseEvalInput(input: string): ParsedEvalInput {
const normalized = input.replace(/\r\n?/g, "\n");
const lines = normalized.split("\n");
const blocks = splitIntoBlocks(lines);
// `split("\n")` produces a trailing empty element when the input ends with
// a newline. Drop it so we don't emit phantom blank trailing code lines.
if (lines.length > 0 && lines[lines.length - 1] === "") lines.pop();
const state: ExpansionState = { language: "python", languageOrigin: "default" };
const cells: ParsedEvalCell[] = [];
for (const block of blocks) {
if (block.type === "raw") {
let i = 0;
// Lenient: leading content before any header forms an implicit
// default-language cell. Drop it if it's only blank lines.
if (i < lines.length && parseHeaderLine(lines[i]) === null) {
const buffer: string[] = [];
while (i < lines.length && parseHeaderLine(lines[i]) === null) {
buffer.push(lines[i]);
i++;
}
const trimmed = trimOuterBlankLines(buffer);
if (trimmed.length > 0) {
cells.push({
index: cells.length,
title: undefined,
code: block.lines.join("\n"),
code: trimmed.join("\n"),
language: state.language,
languageOrigin: state.languageOrigin,
timeoutMs: DEFAULT_TIMEOUT_MS,
reset: false,
});
}
}
while (i < lines.length) {
const headerInfo = parseHeaderLine(lines[i]);
if (headerInfo === null) {
// Loop invariant guarantees this is a header line; guard anyway.
i++;
continue;
}
const fence = parseFenceInfo(block.info, block.startLine);
const language = fence.language ?? state.language;
const languageOrigin: EvalLanguageOrigin = fence.language ? "fence" : state.languageOrigin;
const headerLineNumber = i + 1;
const info = parseHeaderInfo(headerInfo, headerLineNumber);
i++; // consume header line
const codeLines: string[] = [];
while (i < lines.length && parseHeaderLine(lines[i]) === null) {
codeLines.push(lines[i]);
i++;
}
// Strip trailing blank lines so visual spacing between cells doesn't
// leak into the preceding cell's code.
while (codeLines.length > 0 && codeLines[codeLines.length - 1].trim() === "") {
codeLines.pop();
}
const language = info.language ?? state.language;
const languageOrigin: EvalLanguageOrigin = info.language ? "header" : state.languageOrigin;
cells.push({
index: cells.length,
title: fence.title,
code: block.codeLines.join("\n"),
title: info.title,
code: codeLines.join("\n"),
language,
languageOrigin,
timeoutMs: fence.timeoutMs ?? DEFAULT_TIMEOUT_MS,
reset: fence.reset ?? false,
timeoutMs: info.timeoutMs ?? DEFAULT_TIMEOUT_MS,
reset: info.reset ?? false,
});
state.language = language;
state.languageOrigin = languageOrigin;
+13 -19
View File
@@ -1,17 +1,19 @@
Run code in a persistent kernel, using a series of codeblocks acting as cells.
<instruction>
Each cell is a markdown fenced code block. The opening fence's info string carries metadata:
Each cell is introduced by a header line of the form:
```
<lang>? <duration>? (title-fragment | key=value)*
===== <info> =====
```
- **Language**: {{#if py}}`py`/`python` for Python{{/if}}{{#ifAll py js}}, {{/ifAll}}{{#if js}}`js`/`javascript`/`ts`/`typescript` for JavaScript{{/if}}.{{#ifAll py js}} Omitted → inherit the previous cell's language (the first cell defaults to Python, falling back to JavaScript when Python is unavailable).{{else}} Omitted → inherit the previous cell's language.{{/ifAll}}
- **Positional duration**: `15s`, `500ms`, `2m`, or a bare integer (seconds). Default 30s.
where each side is at least 5 equal signs. Everything between one header and the next (or end of input) is the cell's code, verbatim. The info is space-separated tokens, all optional, in any order:
- **Language**: {{#if py}}`py` for Python{{/if}}{{#ifAll py js}}, {{/ifAll}}{{#if js}}`js` / `ts` for JavaScript{{/if}}.{{#ifAll py js}} Omitted → inherit the previous cell's language (the first cell defaults to Python, falling back to JavaScript when Python is unavailable).{{else}} Omitted → inherit the previous cell's language.{{/ifAll}}
- **Title shorthand**: `py:"…"`, `js:"…"`, `ts:"…"` set the language and the cell title together.
- **Attributes**:
- `id="…"` — cell id (shown as the title in the transcript).
- `t=<duration>` — overrides the positional duration.
- `rst=true` — wipe **this cell's own language kernel** before running.{{#ifAll py js}} Other languages are untouched.{{/ifAll}}
- `id:"…"` — cell title (when language is unchanged or already set).
- `t:<duration>` — per-cell timeout. Duration is digits with optional `ms` / `s` / `m` units (e.g. `t:500ms`, `t:15s`, `t:2m`). Default 30s.
- `rst` — wipe **this cell's own language kernel** before running.{{#ifAll py js}} Other languages are untouched.{{/ifAll}}
**Work incrementally:** one logical step per cell (imports, define, test, use). Pass multiple small cells in one call. Define small reusable functions you can debug individually. You **MUST** put workflow explanations in the assistant message or cell title — never inside cell code.
@@ -51,30 +53,22 @@ Cells render like a Jupyter notebook. Pass any value to `display(value)`; non-pr
</output>
<caution>
- In session mode, use `rst=true` on a cell to wipe its language's kernel before running.{{#ifAll py js}} Reset is per-language: a python cell's `rst=true` does not touch the JavaScript kernel and vice versa.{{/ifAll}}
- In session mode, use `rst` on a cell to wipe its language's kernel before running.{{#ifAll py js}} Reset is per-language: a python cell's `rst` does not touch the JavaScript kernel and vice versa.{{/ifAll}}
{{#if js}}- **js**: the VM exposes a selective `process` subset, Web APIs, `Buffer`, `fs/promises`.
{{/if}}</caution>
<example>
{{#if py}}```py id="imports" t="10s"
{{#if py}}===== py:"imports" t:10s =====
import json
from pathlib import Path
```
```py id="load config"
===== py:"load config" =====
data = json.loads(read('package.json'))
display(data)
```
{{/if}}{{#ifAll py js}}
{{/ifAll}}{{#if js}}```js id="js summary" rst=true
{{/ifAll}}{{#if js}}===== js:"js summary" rst =====
const data = JSON.parse(await read('package.json'));
display(data);
return data.name;
```
```
return 'still JavaScript';
```
{{/if}}
</example>
+2 -2
View File
@@ -26,7 +26,7 @@ export const EVAL_DEFAULT_PREVIEW_LINES = 10;
export const evalSchema = Type.Object({
input: Type.String({
description: "atom-style eval input containing CELL sections, fenced code, and optional RESET directive",
description: "eval input as a sequence of `===== <info> =====` cell headers followed by code",
}),
});
export type EvalToolParams = Static<typeof evalSchema>;
@@ -250,7 +250,7 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
let previousRuntimeLanguage: EvalLanguage | undefined;
const cells: ResolvedEvalCell[] = [];
for (const cell of parsedInput.cells) {
const requested = cell.languageOrigin === "fence" ? cell.language : (previousRuntimeLanguage ?? undefined);
const requested = cell.languageOrigin === "header" ? cell.language : (previousRuntimeLanguage ?? undefined);
const resolved = await resolveBackend(session, requested, cell.code);
previousRuntimeLanguage = resolved.backend.id;
cells.push({
+61 -83
View File
@@ -2,10 +2,9 @@ import { describe, expect, it } from "bun:test";
import { parseEvalInput } from "../../src/eval/parse";
describe("parseEvalInput", () => {
it("parses a single fenced cell with positional title and timeout", () => {
const result = parseEvalInput(`\`\`\`py setup 15s
it("parses a single header cell with title shorthand and t timeout", () => {
const result = parseEvalInput(`===== py:"setup" t:15s =====
print("hi")
\`\`\`
`);
expect(result.cells).toEqual([
@@ -14,21 +13,18 @@ print("hi")
title: "setup",
code: 'print("hi")',
language: "python",
languageOrigin: "fence",
languageOrigin: "header",
timeoutMs: 15_000,
reset: false,
},
]);
});
it("treats rst=true as a per-language kernel wipe for that cell", () => {
const result = parseEvalInput(`\`\`\`py rst=true id="bootstrap"
it("treats bare rst as a per-language kernel wipe for that cell", () => {
const result = parseEvalInput(`===== py rst id:"bootstrap" =====
import json
\`\`\`
\`\`\`js rst=true
===== js rst =====
const x = 1;
\`\`\`
`);
expect(result.cells.map(cell => [cell.language, cell.reset, cell.title])).toEqual([
@@ -37,42 +33,35 @@ const x = 1;
]);
});
it("inherits language and runs without reset for empty fence info", () => {
const result = parseEvalInput(`\`\`\`js
it("inherits language across consecutive cells when omitted", () => {
const result = parseEvalInput(`===== js =====
const a = 1;
\`\`\`
\`\`\`
===== =====
const b = a + 1;
\`\`\`
`);
expect(result.cells.map(cell => [cell.language, cell.languageOrigin, cell.code, cell.reset])).toEqual([
["js", "fence", "const a = 1;", false],
["js", "fence", "const b = a + 1;", false],
["js", "header", "const a = 1;", false],
["js", "header", "const b = a + 1;", false],
]);
});
it("supports tilde fences and case-insensitive language tokens including ipython aliases", () => {
const result = parseEvalInput(`~~~TypeScript
it("accepts asymmetric bar runs and case-insensitive language tokens", () => {
const result = parseEvalInput(`===== TypeScript ======
const a = 1;
~~~
\`\`\`IPython
====== IPython =====
print("ipy")
\`\`\`
`);
expect(result.cells.map(cell => [cell.language, cell.languageOrigin])).toEqual([
["js", "fence"],
["python", "fence"],
["js", "header"],
["python", "header"],
]);
});
it("uses canonical id and t attributes, with explicit attrs winning over positional", () => {
const result = parseEvalInput(`\`\`\`py 5s some words t=2m id="explicit win"
const result = parseEvalInput(`===== py 5s some words t:2m id:"explicit win" =====
print(1)
\`\`\`
`);
expect(result.cells[0]).toMatchObject({
@@ -83,39 +72,28 @@ print(1)
});
it("accepts fallback aliases for id, t, and rst keys", () => {
const cases = [
{ key: "title", expectTitle: "alpha" },
{ key: "name", expectTitle: "alpha" },
{ key: "cell", expectTitle: "alpha" },
{ key: "file", expectTitle: "alpha" },
{ key: "label", expectTitle: "alpha" },
];
for (const { key, expectTitle } of cases) {
const result = parseEvalInput(`\`\`\`py ${key}="alpha"\nprint(1)\n\`\`\`\n`);
expect(result.cells[0].title).toBe(expectTitle);
const idAliases = ["title", "name", "cell", "file", "label"];
for (const key of idAliases) {
const result = parseEvalInput(`===== py ${key}:"alpha" =====\nprint(1)\n`);
expect(result.cells[0].title).toBe("alpha");
}
const timeoutAliases = ["timeout", "duration", "time"];
for (const key of timeoutAliases) {
const result = parseEvalInput(`\`\`\`py ${key}=2m\nprint(1)\n\`\`\`\n`);
const result = parseEvalInput(`===== py ${key}:2m =====\nprint(1)\n`);
expect(result.cells[0].timeoutMs).toBe(120_000);
}
const resetAliases = ["reset"];
for (const key of resetAliases) {
const result = parseEvalInput(`\`\`\`py ${key}=true\nprint(1)\n\`\`\`\n`);
expect(result.cells[0].reset).toBe(true);
}
const result = parseEvalInput(`===== py reset:true =====\nprint(1)\n`);
expect(result.cells[0].reset).toBe(true);
});
it("first occurrence wins when canonical and alias collide", () => {
const canonicalFirst = parseEvalInput(`\`\`\`py id="canon" title="alias"
const canonicalFirst = parseEvalInput(`===== py id:"canon" title:"alias" =====
print(1)
\`\`\`
`);
const aliasFirst = parseEvalInput(`\`\`\`py title="alias" id="canon"
const aliasFirst = parseEvalInput(`===== py title:"alias" id:"canon" =====
print(1)
\`\`\`
`);
expect(canonicalFirst.cells[0].title).toBe("canon");
@@ -123,26 +101,20 @@ print(1)
});
it("parses millisecond, second, and minute durations", () => {
const result = parseEvalInput(`\`\`\`py 500ms
const result = parseEvalInput(`===== py t:500ms =====
a = 1
\`\`\`
\`\`\`py 5
===== py t:5 =====
a = 2
\`\`\`
\`\`\`py 2m
===== py t:2m =====
a = 3
\`\`\`
`);
expect(result.cells.map(cell => cell.timeoutMs)).toEqual([500, 5_000, 120_000]);
});
it("treats unrecognized fence info as title and inherits the language", () => {
const result = parseEvalInput(`\`\`\`ruby
it("treats unrecognized header tokens as a title and inherits the language", () => {
const result = parseEvalInput(`===== ruby =====
puts "no"
\`\`\`
`);
expect(result.cells[0]).toMatchObject({
@@ -154,21 +126,18 @@ puts "no"
});
it("joins multiple positional title fragments with spaces", () => {
const result = parseEvalInput(`\`\`\`py compute totals
const result = parseEvalInput(`===== py compute totals =====
print(1)
\`\`\`
`);
expect(result.cells[0].title).toBe("compute totals");
});
it("accepts back-to-back fenced cells without blank separators", () => {
const result = parseEvalInput(`\`\`\`py id=a
it("accepts back-to-back header cells without blank separators", () => {
const result = parseEvalInput(`===== py id:"a" =====
print("a")
\`\`\`
\`\`\`py id=b
===== py id:"b" =====
print("b")
\`\`\`
`);
expect(result.cells.map(cell => [cell.title, cell.code])).toEqual([
@@ -177,7 +146,7 @@ print("b")
]);
});
it("wraps bare code with no fences in a single implicit cell", () => {
it("wraps bare code with no headers in a single implicit cell", () => {
const result = parseEvalInput(`print("hello")
print("world")
`);
@@ -195,37 +164,37 @@ print("world")
]);
});
it("surfaces raw inter-fence content as its own implicit cell that inherits language", () => {
const result = parseEvalInput(`\`\`\`js
it("strips blank lines between cells from the preceding cell's code", () => {
const result = parseEvalInput(`===== js =====
const x = 1;
\`\`\`
inherited tail
===== =====
const y = 2;
`);
expect(result.cells.map(cell => [cell.language, cell.languageOrigin, cell.code])).toEqual([
["js", "fence", "const x = 1;"],
["js", "fence", "inherited tail"],
["js", "header", "const x = 1;"],
["js", "header", "const y = 2;"],
]);
});
it("treats unclosed fences leniently and closes them at end of input", () => {
const result = parseEvalInput(`\`\`\`py
print("still typing")`);
it("accepts an empty header introducing a default cell with no info", () => {
const result = parseEvalInput(`=====
print("still typing")
`);
expect(result.cells).toHaveLength(1);
expect(result.cells[0]).toMatchObject({
code: 'print("still typing")',
language: "python",
languageOrigin: "fence",
languageOrigin: "default",
reset: false,
});
});
it("ignores unknown attribute keys without erroring", () => {
const result = parseEvalInput(`\`\`\`py mystery=123 id=ok
const result = parseEvalInput(`===== py mystery:123 id:"ok" =====
print(1)
\`\`\`
`);
expect(result.cells[0]).toMatchObject({ title: "ok", language: "python" });
@@ -233,19 +202,28 @@ print(1)
it("rejects an invalid rst value", () => {
expect(() =>
parseEvalInput(`\`\`\`py rst=maybe
parseEvalInput(`===== py rst:maybe =====
print(1)
\`\`\`
`),
).toThrow("invalid rst value");
});
it("rejects an invalid t value", () => {
expect(() =>
parseEvalInput(`\`\`\`py t=forever
parseEvalInput(`===== py t:forever =====
print(1)
\`\`\`
`),
).toThrow("invalid duration");
});
it("does not treat lines that start with equals but have no closing bar as a header", () => {
const result = parseEvalInput(`===== py =====
x = 1
===== not a header
y = 2
`);
expect(result.cells).toHaveLength(1);
expect(result.cells[0].code).toBe("x = 1\n===== not a header\ny = 2");
});
});