feat(coding-agent): refactored eval tool to single-step execution
- Transitioned the eval tool from batch multi-cell execution to a single-step input structure with flat parameters. - Updated core agent logic, UI components, and documentation to support state persistence across incremental eval calls. - Restricted bash tool capabilities by requiring explicit use of `read` or `find` instead of `ls` or `find`. - Added support for Ruby and Julia language runtimes to the eval tool and associated web renderers.
This commit is contained in:
@@ -1,9 +1,11 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
- Renamed the eval `agent()` helper parameters `agent_type` → `agent` and `return_handle` → `handle` across every workflow runtime (Python, JavaScript, Ruby, Julia), so the names are identical in every language (no camelCase/snake_case split) and the agent-selection parameter matches the `task` tool's `agent`. The `__agent__` eval bridge wire protocol was renamed to match.
|
||||
- Changed the `eval` tool to take a single cell per call (`{ language, code, title?, timeout?, reset? }`) instead of a `cells` array. State still persists per language across separate eval calls, tool calls, and `task` subagents, so each call is one logical step that reuses everything earlier calls defined — the array only encouraged re-importing/re-declaring the same setup in every batch. The schema, field descriptions, examples, system `eval.md`/`workflowz` helper docs, and the `[i/n]` cell-counter (now hidden for single cells) were updated to match; the renderer, ACP start-text, copy-targets, and collab-web tool view still parse legacy multi-cell transcripts.
|
||||
|
||||
### Added
|
||||
|
||||
@@ -11,6 +13,9 @@
|
||||
|
||||
### Changed
|
||||
|
||||
- Simplified `eval` tool to accept a single logical step (code block) instead of an array of cells
|
||||
- Updated `eval` tool documentation to emphasize incremental, single-step execution
|
||||
- Restricted `bash` tool from using `ls` or `find`, requiring the use of `read` or `find` tools
|
||||
- Simplified `todo` tool interface to accept a single operation directly instead of an array of ops
|
||||
- Reinforced routing of fragile, multi-step shell logic to the `eval` tool over `bash`. The system-prompt tool policy, `bash.md`, and `eval.md` now treat loops, conditionals, heredocs, inline `-e`/`-c` scripts, multi-stage pipelines, and quote/JSON escaping as the signal to write an `eval` cell; bash's "compute a fact" carveout is narrowed to single short pipelines, and `eval.md` now actively claims that territory with runtime-templated examples (only enabled backends are advertised).
|
||||
- Made `eval` an essential built-in tool (`loadMode: "essential"`, added to the default essential tool set) so it stays active under `tools.discoveryMode: "all"` instead of being hidden behind `search_tool_bm25`.
|
||||
|
||||
@@ -59,31 +59,23 @@ export const shellFixtures: Record<string, GalleryFixture> = {
|
||||
eval: {
|
||||
label: "Eval",
|
||||
streamingArgs: {
|
||||
cells: [
|
||||
{
|
||||
language: "py",
|
||||
code: 'import json\nfrom pathlib import Path\n\ndata = json.loads(Path("package.js',
|
||||
title: "load config",
|
||||
},
|
||||
],
|
||||
language: "py",
|
||||
code: 'import json\nfrom pathlib import Path\n\ndata = json.loads(Path("package.js',
|
||||
title: "load config",
|
||||
},
|
||||
args: {
|
||||
cells: [
|
||||
{
|
||||
language: "py",
|
||||
title: "load config",
|
||||
code: [
|
||||
"import json",
|
||||
"from pathlib import Path",
|
||||
"",
|
||||
'data = json.loads(Path("package.json").read_text())',
|
||||
'deps = data.get("dependencies", {})',
|
||||
'print(f"{data[\\"name\\"]} v{data[\\"version\\"]}")',
|
||||
'print(f"{len(deps)} dependencies")',
|
||||
"display(sorted(deps)[:3])",
|
||||
].join("\n"),
|
||||
},
|
||||
],
|
||||
language: "py",
|
||||
title: "load config",
|
||||
code: [
|
||||
"import json",
|
||||
"from pathlib import Path",
|
||||
"",
|
||||
'data = json.loads(Path("package.json").read_text())',
|
||||
'deps = data.get("dependencies", {})',
|
||||
'print(f"{data[\\"name\\"]} v{data[\\"version\\"]}")',
|
||||
'print(f"{len(deps)} dependencies")',
|
||||
"display(sorted(deps)[:3])",
|
||||
].join("\n"),
|
||||
},
|
||||
result: {
|
||||
content: [
|
||||
|
||||
@@ -468,8 +468,13 @@ function buildEvalStartText(args: unknown): string | undefined {
|
||||
if (typeof args !== "object" || args === null || Array.isArray(args)) {
|
||||
return undefined;
|
||||
}
|
||||
const cells = (args as EvalCellContainer).cells;
|
||||
if (!Array.isArray(cells) || cells.length === 0) {
|
||||
const container = args as EvalCellContainer & EvalCellLike;
|
||||
const cells = Array.isArray(container.cells)
|
||||
? container.cells
|
||||
: typeof container.code === "string"
|
||||
? [container]
|
||||
: [];
|
||||
if (cells.length === 0) {
|
||||
return undefined;
|
||||
}
|
||||
const lines: string[] = [];
|
||||
|
||||
@@ -127,8 +127,13 @@ export function extractQuoteBlocks(text: string): QuoteBlock[] {
|
||||
|
||||
function extractEvalCode(args: unknown): { code: string; language: string } | undefined {
|
||||
if (!args || typeof args !== "object") return undefined;
|
||||
const cells = (args as { cells?: unknown }).cells;
|
||||
if (!Array.isArray(cells)) return undefined;
|
||||
const argsObj = args as { cells?: unknown; code?: unknown };
|
||||
const cells = Array.isArray(argsObj.cells)
|
||||
? argsObj.cells
|
||||
: typeof argsObj.code === "string"
|
||||
? [argsObj]
|
||||
: undefined;
|
||||
if (!cells) return undefined;
|
||||
|
||||
const codeBlocks: string[] = [];
|
||||
let language = "python";
|
||||
|
||||
@@ -11,7 +11,7 @@ Worth it when the task benefits from decomposition + parallel coverage, or from
|
||||
</when>
|
||||
|
||||
<helpers>
|
||||
State persists across cells, so scout in one cell and fan out in the next. Every cell has:
|
||||
State persists across eval calls, so scout in one call and fan out in the next. Every eval call has:
|
||||
|
||||
- `agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts.
|
||||
- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one.
|
||||
@@ -20,7 +20,7 @@ State persists across cells, so scout in one cell and fan out in the next. Every
|
||||
- `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it.
|
||||
- `budget` — `budget.total` (output-token ceiling, or `None` when none is set), `budget.spent()` (tokens spent this turn — main loop + eval subagents), `budget.remaining()` (`math.inf` when total is `None`), `budget.hard` (whether it's enforced). A ceiling is set by the user: `+Nk` in their message is advisory (you self-limit via `budget.remaining()`), `+Nk!` (or Goal Mode) is hard — `agent()` refuses to spawn once spent reaches it. Gate loops on `budget.total` first, since it's `None` when the user set no budget.
|
||||
|
||||
Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across cells and turns for multi-phase work, reading each result before you decide the next phase.
|
||||
Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across calls and turns for multi-phase work, reading each result before you decide the next phase.
|
||||
</helpers>
|
||||
|
||||
<structure>
|
||||
|
||||
@@ -30,6 +30,7 @@ Anything below → `eval` cell, not bash:
|
||||
<critical>
|
||||
- Bash invokes real binaries with simple args; it is NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps.
|
||||
- NEVER shell out to search content or files: `grep/rg` → `search`.
|
||||
- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `find` tool (globbing). This is non-negotiable, even for a single quick listing.
|
||||
- Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://<id>`.
|
||||
</critical>
|
||||
|
||||
|
||||
@@ -1,22 +1,23 @@
|
||||
Run code in a persistent kernel using a list of cells.
|
||||
Run one step of code in a persistent kernel.
|
||||
|
||||
<instruction>
|
||||
Cells run in array order. State persists per language across cells, tool calls, and `task` subagents — stage helpers/datasets/clients once, subagents reuse directly, no re-import/serialize.
|
||||
**One eval call = one cell = one logical step.** State persists per language across separate eval calls, tool calls, and `task` subagents — define helpers/datasets/clients in one call, then later calls reuse them directly.
|
||||
|
||||
Cell fields:
|
||||
Work incrementally: imports in one call, define in the next, test, then use — each its own eval call. Re-run setup ONLY after `reset`, a kernel crash, or a `NameError`/`ReferenceError` proving the state is gone. Parallelize work *within* a cell with the `parallel(thunks)` helper, not by batching steps.
|
||||
|
||||
Fields:
|
||||
|
||||
- `language` — {{#if py}}`"py"` IPython kernel{{/if}}{{#ifAll py js}}, {{/ifAll}}{{#if js}}`"js"` persistent JavaScript VM{{/if}}{{#if rb}}{{#ifAny py js}}, {{/ifAny}}`"rb"` persistent Ruby kernel{{/if}}{{#if jl}}{{#ifAny py js rb}}, {{/ifAny}}`"jl"` persistent Julia kernel{{/if}}.
|
||||
- `code` — cell body, verbatim. Newlines/quotes JSON-encoded; no fences, no headers.
|
||||
- `title` (optional) — short transcript label (e.g. `"imports"`).
|
||||
- `timeout` (optional) — per-cell seconds. Raise only for heavy compute or long non-agent tool calls.
|
||||
- `reset` (optional) — wipe this cell's language kernel first.{{#ifAll py js}} Per-language: a `py` reset never touches the JS VM.{{/ifAll}}
|
||||
- `timeout` (optional) — seconds. Raise only for heavy compute or long non-agent tool calls.
|
||||
- `reset` (optional) — wipe this language's kernel first.{{#ifAll py js}} Per-language: a `py` reset never touches the JS VM.{{/ifAll}}
|
||||
|
||||
Work incrementally — one logical step per cell (imports, define, test, use), many small cells per call; workflow notes in the assistant message or `title`, never in cell code.
|
||||
{{#if py}}Live event loop: use top-level `await` directly; `asyncio.run(…)` raises "cannot be called from a running event loop".{{/if}}
|
||||
{{#if js}}JS runs under **Bun**: Bun globals/APIs are available (`Bun.file`, `Bun.write`, `Bun.$`, `fetch`, `Buffer`); top-level `await`/`return` work directly.{{/if}}
|
||||
{{#if rb}}Ruby: synchronous; helper options are keyword args (e.g. `tree(".", max_depth: 2)`); the last expression auto-displays unless it is `nil`, an assignment, or a definition (like IRB).{{/if}}
|
||||
{{#if jl}}Julia: synchronous; helper options are standard keyword args (e.g. `tree(max_depth=2)`); the last expression auto-displays unless it is an assignment or a definition (like the Julia REPL).{{/if}}
|
||||
Errors name the failing cell ("Cell 3 failed") — resubmit the fixed cell + any remaining.
|
||||
On error, fix and re-run only the failing step — prior calls' state survives.
|
||||
</instruction>
|
||||
|
||||
<prelude>
|
||||
@@ -71,3 +72,7 @@ Pipe handles through stage helpers to build a dependency graph — acyclic waves
|
||||
- **Acyclic only.** A node never waits on its own descendant.
|
||||
</dag>
|
||||
{{/if}}
|
||||
|
||||
<critical>
|
||||
Prior top-level names (`data`, `sessions`, helpers, imports) survive into the next eval call — reuse them; NEVER re-import, re-require, or re-declare a helper. Re-read a file only if it may have changed since the last read. Re-run setup only after `reset`, a crash, or a `NameError`/`ReferenceError`.
|
||||
</critical>
|
||||
|
||||
@@ -56,6 +56,9 @@ interface EvalRenderCellArg {
|
||||
}
|
||||
|
||||
interface EvalRenderArgs {
|
||||
language?: string;
|
||||
code?: string;
|
||||
title?: string;
|
||||
cells?: EvalRenderCellArg[];
|
||||
__partialJson?: string;
|
||||
}
|
||||
@@ -81,8 +84,8 @@ function normalizeRenderLanguage(value: string | undefined): EvalLanguage {
|
||||
}
|
||||
|
||||
function getRenderCells(args: EvalRenderArgs | undefined): EvalRenderCell[] {
|
||||
const raw = args?.cells;
|
||||
if (!Array.isArray(raw)) return [];
|
||||
if (!args) return [];
|
||||
const raw = Array.isArray(args.cells) ? args.cells : typeof args.code === "string" ? [args] : [];
|
||||
const out: EvalRenderCell[] = [];
|
||||
for (const cell of raw) {
|
||||
if (!cell || typeof cell !== "object") continue;
|
||||
|
||||
@@ -38,8 +38,6 @@ const EVAL_LANGUAGE_NAME: Record<EvalLanguageToken, string> = {
|
||||
rb: "Ruby",
|
||||
jl: "Julia",
|
||||
};
|
||||
const EVAL_CELLS_DESCRIPTION =
|
||||
"cells executed in order. State persists within each language across cells and tool calls.";
|
||||
|
||||
/** Join names as an English "or" list: ["A"]→"A", ["A","B"]→"A or B", 3+→"A, B, or C". */
|
||||
function joinWithOr(items: readonly string[]): string {
|
||||
@@ -56,12 +54,12 @@ function describeCodeField(langs: readonly EvalLanguageToken[]): string {
|
||||
const replLangs = langs.filter(lang => lang === "rb" || lang === "jl");
|
||||
// No persistent REPL backends → keep the original py/js phrasing verbatim so the
|
||||
// default (rb/jl off) wire schema stays byte-identical to the pre-feature one.
|
||||
if (replLangs.length === 0) return "cell body, verbatim. Use top-level await freely.";
|
||||
if (replLangs.length === 0) return "code to run in this eval call, verbatim. Use top-level await freely.";
|
||||
const awaitLangs = langs.filter(lang => lang === "py" || lang === "js");
|
||||
const clauses: string[] = [];
|
||||
if (awaitLangs.length > 0) clauses.push(`Top-level \`await\` is available in ${awaitLangs.join("/")}`);
|
||||
clauses.push(`${replLangs.join("/")} auto-display the last expression like a REPL`);
|
||||
return `cell body, verbatim. ${clauses.join("; ")}.`;
|
||||
return `code to run in this eval call, verbatim. ${clauses.join("; ")}.`;
|
||||
}
|
||||
|
||||
/** One-line discovery summary listing the runtimes available this session. */
|
||||
@@ -86,29 +84,24 @@ function enabledEvalLanguages(backends: EvalBackendsAllowance): EvalLanguageToke
|
||||
|
||||
const evalCellCommonFields = {
|
||||
"title?": type("string").describe('short label shown in transcript (e.g. "imports", "load config")'),
|
||||
"timeout?": type("number").describe("per-cell timeout in seconds"),
|
||||
"reset?": type("boolean").describe(
|
||||
"wipe this cell's language kernel before running. Other languages are untouched.",
|
||||
),
|
||||
"timeout?": type("number").describe("timeout for this eval call in seconds"),
|
||||
"reset?": type("boolean").describe("wipe this language's kernel before running. Other languages are untouched."),
|
||||
};
|
||||
|
||||
/**
|
||||
* Per-cell input. Each cell runs in order; state persists within a language
|
||||
* across cells and across tool calls. This static schema carries the full
|
||||
* language union for typing; {@link buildEvalSchema} narrows the wire copy per
|
||||
* session so disabled backends are never advertised to the model.
|
||||
* Per-call input: a single cell. State persists within a language across
|
||||
* separate eval calls and across tool calls, so each call is one logical step
|
||||
* and later calls reuse what earlier ones defined. This static schema carries
|
||||
* the full language union for typing; {@link buildEvalSchema} narrows the wire
|
||||
* copy per session so disabled backends are never advertised to the model.
|
||||
*/
|
||||
const evalCellSchema = type({
|
||||
language: type("'py' | 'js' | 'rb' | 'jl'").describe(describeLanguageField(EVAL_LANGUAGE_ORDER)),
|
||||
code: type("string").describe(describeCodeField(EVAL_LANGUAGE_ORDER)),
|
||||
...evalCellCommonFields,
|
||||
});
|
||||
export type EvalCellInput = typeof evalCellSchema.infer;
|
||||
|
||||
export const evalSchema = type({
|
||||
cells: evalCellSchema.array().atLeastLength(1).describe(EVAL_CELLS_DESCRIPTION),
|
||||
language: type("'py' | 'js' | 'rb' | 'jl'").describe(describeLanguageField(EVAL_LANGUAGE_ORDER)),
|
||||
...evalCellCommonFields,
|
||||
code: type("string").describe(describeCodeField(EVAL_LANGUAGE_ORDER)),
|
||||
});
|
||||
export type EvalToolParams = typeof evalSchema.infer;
|
||||
export type EvalCellInput = EvalToolParams;
|
||||
|
||||
/**
|
||||
* Build a session-scoped copy of the eval schema whose `language` enum and field
|
||||
@@ -118,14 +111,11 @@ export type EvalToolParams = typeof evalSchema.infer;
|
||||
* {@link evalSchema} (full union) remains the type-level source of truth.
|
||||
*/
|
||||
function buildEvalSchema(langs: readonly EvalLanguageToken[]): typeof evalSchema {
|
||||
const cellSchema = type({
|
||||
const schema = type({
|
||||
language: type.enumerated(...langs).describe(describeLanguageField(langs)),
|
||||
code: type("string").describe(describeCodeField(langs)),
|
||||
...evalCellCommonFields,
|
||||
});
|
||||
const schema = type({
|
||||
cells: cellSchema.array().atLeastLength(1).describe(EVAL_CELLS_DESCRIPTION),
|
||||
});
|
||||
return schema as unknown as typeof evalSchema;
|
||||
}
|
||||
|
||||
@@ -290,17 +280,10 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
|
||||
readonly approval = "exec" as const;
|
||||
readonly formatApprovalDetails = (args: unknown): string[] => {
|
||||
const params = args as Partial<EvalToolParams>;
|
||||
const cells = Array.isArray(params.cells) ? params.cells : [];
|
||||
const firstCell = cells[0] as Partial<EvalCellInput> | undefined;
|
||||
if (!firstCell) return [];
|
||||
const language =
|
||||
typeof firstCell.language === "string" ? formatEvalInputLanguage(firstCell.language) : "javascript (default)";
|
||||
const code = typeof firstCell.code === "string" ? firstCell.code : "";
|
||||
const lines = [`Language: ${language}`, `Code:\n${truncateForPrompt(code)}`];
|
||||
if (cells.length > 1) {
|
||||
lines.push(`+${cells.length - 1} more cell${cells.length === 2 ? "" : "s"}`);
|
||||
}
|
||||
return lines;
|
||||
typeof params.language === "string" ? formatEvalInputLanguage(params.language) : "javascript (default)";
|
||||
const code = typeof params.code === "string" ? params.code : "";
|
||||
return [`Language: ${language}`, `Code:\n${truncateForPrompt(code)}`];
|
||||
};
|
||||
get summary(): string {
|
||||
return summarizeEvalLanguages(this.#enabledLanguages());
|
||||
@@ -320,25 +303,53 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
|
||||
spawns: spawnsAllowed,
|
||||
});
|
||||
}
|
||||
readonly examples: readonly ToolExample<typeof evalSchema.infer>[] = [
|
||||
/** All reuse-chain examples; the `examples` getter filters by enabled languages. */
|
||||
private static readonly ALL_EXAMPLES: readonly ToolExample<typeof evalSchema.infer>[] = [
|
||||
{
|
||||
caption: "First call — set up once",
|
||||
call: {
|
||||
cells: [
|
||||
{
|
||||
language: "py",
|
||||
title: "imports",
|
||||
timeout: 10,
|
||||
code: "import json\nfrom pathlib import Path",
|
||||
},
|
||||
{
|
||||
language: "py",
|
||||
title: "load config",
|
||||
code: "data = json.loads(read('package.json'))\ndisplay(data)",
|
||||
},
|
||||
],
|
||||
language: "py",
|
||||
title: "imports",
|
||||
code: "import json\nfrom pathlib import Path",
|
||||
},
|
||||
},
|
||||
{
|
||||
caption: "Second call — reuse, do NOT re-import",
|
||||
call: {
|
||||
language: "py",
|
||||
title: "load config",
|
||||
code: "data = json.loads(read('package.json'))\ndisplay(data)",
|
||||
},
|
||||
},
|
||||
{
|
||||
caption: "Third call — reuse the loaded config",
|
||||
call: {
|
||||
language: "py",
|
||||
title: "scan deps",
|
||||
code: "display(sorted(data['dependencies']))",
|
||||
},
|
||||
},
|
||||
{
|
||||
caption: "Ruby first call — set up once",
|
||||
call: {
|
||||
language: "rb",
|
||||
title: "setup",
|
||||
code: "require 'json'\npkg_path = 'package.json'",
|
||||
},
|
||||
},
|
||||
{
|
||||
caption: "Ruby second call — reuse, do NOT re-require",
|
||||
call: {
|
||||
language: "rb",
|
||||
title: "load config",
|
||||
code: "pkg = JSON.parse(read(pkg_path))\ndisplay(pkg.keys.sort)",
|
||||
},
|
||||
},
|
||||
];
|
||||
get examples(): readonly ToolExample<typeof evalSchema.infer>[] {
|
||||
const langs = new Set(this.#enabledLanguages());
|
||||
return EvalTool.ALL_EXAMPLES.filter(ex => "call" in ex && langs.has(ex.call.language as EvalLanguageToken));
|
||||
}
|
||||
get parameters(): typeof evalSchema {
|
||||
const langs = this.#enabledLanguages();
|
||||
if (langs.length === 0 || langs.length === EVAL_LANGUAGE_ORDER.length) return evalSchema;
|
||||
@@ -352,13 +363,9 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
|
||||
readonly concurrency = "exclusive";
|
||||
readonly strict = true;
|
||||
readonly intent = (args: Partial<typeof evalSchema.infer>): string | undefined => {
|
||||
const cells = Array.isArray(args.cells) ? args.cells : [];
|
||||
const first = cells.find(c => c && typeof c === "object");
|
||||
if (!first) return "evaluating";
|
||||
const title = typeof first.title === "string" ? first.title : undefined;
|
||||
const language = typeof first.language === "string" ? formatEvalInputLanguage(first.language) : "javascript";
|
||||
const label = title || `running ${language}`;
|
||||
return cells.length > 1 ? `${label} (+${cells.length - 1})` : label;
|
||||
const title = typeof args.title === "string" ? args.title : undefined;
|
||||
const language = typeof args.language === "string" ? formatEvalInputLanguage(args.language) : "javascript";
|
||||
return title || `running ${language}`;
|
||||
};
|
||||
|
||||
readonly #proxyExecutor?: EvalProxyExecutor;
|
||||
@@ -398,27 +405,25 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
|
||||
const session = this.session;
|
||||
const excludeWebP = webpExclusionForModel(session.getActiveModel?.());
|
||||
|
||||
const cells: ResolvedEvalCell[] = [];
|
||||
for (let i = 0; i < params.cells.length; i++) {
|
||||
const cell = params.cells[i];
|
||||
const language: EvalLanguage =
|
||||
cell.language === "py"
|
||||
? "python"
|
||||
: cell.language === "rb"
|
||||
? "ruby"
|
||||
: cell.language === "jl"
|
||||
? "julia"
|
||||
: "js";
|
||||
const resolved = await resolveBackend(session, language);
|
||||
cells.push({
|
||||
index: i,
|
||||
title: cell.title,
|
||||
code: cell.code,
|
||||
timeoutMs: (cell.timeout ?? 30) * 1000,
|
||||
reset: cell.reset ?? false,
|
||||
const cellLanguage: EvalLanguage =
|
||||
params.language === "py"
|
||||
? "python"
|
||||
: params.language === "rb"
|
||||
? "ruby"
|
||||
: params.language === "jl"
|
||||
? "julia"
|
||||
: "js";
|
||||
const resolved = await resolveBackend(session, cellLanguage);
|
||||
const cells: ResolvedEvalCell[] = [
|
||||
{
|
||||
index: 0,
|
||||
title: params.title,
|
||||
code: params.code,
|
||||
timeoutMs: (params.timeout ?? 30) * 1000,
|
||||
reset: params.reset ?? false,
|
||||
resolved,
|
||||
});
|
||||
}
|
||||
},
|
||||
];
|
||||
const languages = uniqueEvalLanguages(cells);
|
||||
const notice = detailsNotice(cells);
|
||||
const sessionAbortController = new AbortController();
|
||||
@@ -623,24 +628,9 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
|
||||
cellResult.statusEvents = cellStatusEvents.length > 0 ? cellStatusEvents : undefined;
|
||||
cellResult.hasMarkdown = cellHasMarkdown || undefined;
|
||||
|
||||
let combinedCellOutput = "";
|
||||
if (cells.length > 1) {
|
||||
const cellHeader = `[${i + 1}/${cells.length}]`;
|
||||
const cellTitle = cell.title ? ` ${cell.title}` : "";
|
||||
if (cellOutput) {
|
||||
combinedCellOutput = `${cellHeader}${cellTitle}\n${cellOutput}`;
|
||||
} else {
|
||||
combinedCellOutput = `${cellHeader}${cellTitle} (ok)`;
|
||||
}
|
||||
cellOutputs.push(combinedCellOutput);
|
||||
} else if (cellOutput) {
|
||||
combinedCellOutput = cellOutput;
|
||||
cellOutputs.push(combinedCellOutput);
|
||||
}
|
||||
|
||||
if (combinedCellOutput) {
|
||||
const prefix = cellOutputs.length > 1 ? "\n\n" : "";
|
||||
appendTail(`${prefix}${combinedCellOutput}`);
|
||||
if (cellOutput) {
|
||||
cellOutputs.push(cellOutput);
|
||||
appendTail(cellOutput);
|
||||
}
|
||||
|
||||
if (result.cancelled) {
|
||||
@@ -648,10 +638,7 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
|
||||
pushUpdate();
|
||||
const errorMsg = result.output || "Command aborted";
|
||||
const combinedOutput = cellOutputs.join("\n\n");
|
||||
const outputText =
|
||||
cells.length > 1
|
||||
? `${combinedOutput}\n\nCell ${i + 1} aborted: ${errorMsg}`
|
||||
: combinedOutput || errorMsg;
|
||||
const outputText = combinedOutput || errorMsg;
|
||||
|
||||
const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput);
|
||||
const details: EvalToolDetails = {
|
||||
@@ -674,12 +661,9 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
|
||||
cellResult.status = "error";
|
||||
pushUpdate();
|
||||
const combinedOutput = cellOutputs.join("\n\n");
|
||||
const outputText =
|
||||
cells.length > 1
|
||||
? `${combinedOutput}\n\nCell ${i + 1} failed (exit code ${result.exitCode}). Earlier cells succeeded—their state persists. Fix only cell ${i + 1}.`
|
||||
: combinedOutput
|
||||
? `${combinedOutput}\n\nCommand exited with code ${result.exitCode}`
|
||||
: `Command exited with code ${result.exitCode}`;
|
||||
const outputText = combinedOutput
|
||||
? `${combinedOutput}\n\nCommand exited with code ${result.exitCode}`
|
||||
: `Command exited with code ${result.exitCode}`;
|
||||
|
||||
const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput);
|
||||
const details: EvalToolDetails = {
|
||||
|
||||
@@ -80,7 +80,7 @@ function formatHeader(options: CodeCellOptions, theme: Theme): { title: string;
|
||||
parts.push(icon);
|
||||
}
|
||||
}
|
||||
if (index !== undefined && total !== undefined) {
|
||||
if (index !== undefined && total !== undefined && total > 1) {
|
||||
parts.push(theme.fg("accent", `[${index + 1}/${total}]`));
|
||||
}
|
||||
if (title) {
|
||||
|
||||
@@ -141,11 +141,7 @@ export async function resizeImage(img: ImageContent, options?: ImageResizeOption
|
||||
// lagging edge up to the floor via the default fit:"fill" resize.
|
||||
if (targetWidth < minDimension || targetHeight < minDimension) {
|
||||
const shortEdge = Math.min(targetWidth, targetHeight);
|
||||
const upscale = Math.min(
|
||||
minDimension / shortEdge,
|
||||
opts.maxWidth / targetWidth,
|
||||
opts.maxHeight / targetHeight,
|
||||
);
|
||||
const upscale = Math.min(minDimension / shortEdge, opts.maxWidth / targetWidth, opts.maxHeight / targetHeight);
|
||||
if (upscale > 1) {
|
||||
targetWidth = Math.round(targetWidth * upscale);
|
||||
targetHeight = Math.round(targetHeight * upscale);
|
||||
|
||||
@@ -247,7 +247,7 @@ describe("ACP event mapper", () => {
|
||||
type: "tool_execution_start",
|
||||
toolCallId: "tc-eval-start",
|
||||
toolName: "eval",
|
||||
args: { cells: [{ language: "js", title: "sum", code: "return 1 + 1;" }] },
|
||||
args: { language: "js", title: "sum", code: "return 1 + 1;" },
|
||||
intent: "sum",
|
||||
} as AgentSessionEvent,
|
||||
"session-1",
|
||||
@@ -267,7 +267,7 @@ describe("ACP event mapper", () => {
|
||||
expect(update.title).toBe("[js] sum\nreturn 1 + 1;");
|
||||
expect(update.kind).toBe("execute");
|
||||
expect(update.status).toBe("pending");
|
||||
expect(update.rawInput).toEqual({ cells: [{ language: "js", title: "sum", code: "return 1 + 1;" }] });
|
||||
expect(update.rawInput).toEqual({ language: "js", title: "sum", code: "return 1 + 1;" });
|
||||
expect(update.content).toContainEqual({
|
||||
type: "content",
|
||||
content: { type: "text", text: "[js] sum\nreturn 1 + 1;" },
|
||||
@@ -305,7 +305,7 @@ describe("ACP event mapper", () => {
|
||||
type: "tool_execution_start",
|
||||
toolCallId: "tc-eval-long-source",
|
||||
toolName: "eval",
|
||||
args: { cells: [{ language: "js", code: source }] },
|
||||
args: { language: "js", code: source },
|
||||
} as AgentSessionEvent,
|
||||
"session-1",
|
||||
);
|
||||
|
||||
@@ -426,7 +426,7 @@ describe("AgentSession python cleanup", () => {
|
||||
expect(EvalTool).toBeDefined();
|
||||
let toolExecutionSettled = false;
|
||||
const toolExecution = EvalTool!
|
||||
.execute("call-id", { cells: [{ language: "py", code: "print('tool')" }] }, undefined, undefined, undefined)
|
||||
.execute("call-id", { language: "py", code: "print('tool')" }, undefined, undefined, undefined)
|
||||
.finally(() => {
|
||||
toolExecutionSettled = true;
|
||||
});
|
||||
@@ -652,13 +652,7 @@ describe("AgentSession python cleanup", () => {
|
||||
expect(EvalTool).toBeDefined();
|
||||
const disposeSession = session.dispose();
|
||||
await expect(
|
||||
EvalTool!.execute(
|
||||
"call-id",
|
||||
{ cells: [{ language: "py", code: "print('late')" }] },
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
),
|
||||
EvalTool!.execute("call-id", { language: "py", code: "print('late')" }, undefined, undefined, undefined),
|
||||
).rejects.toThrow("Python execution is unavailable while session disposal is in progress");
|
||||
await disposeSession;
|
||||
expect(executeSpy).not.toHaveBeenCalled();
|
||||
@@ -693,7 +687,7 @@ describe("AgentSession python cleanup", () => {
|
||||
expect(EvalTool).toBeDefined();
|
||||
const execution = EvalTool!.execute(
|
||||
"call-id",
|
||||
{ cells: [{ language: "py", code: "print('late after artifact')" }] },
|
||||
{ language: "py", code: "print('late after artifact')" },
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
|
||||
@@ -76,18 +76,25 @@ describe("extractLastCommand", () => {
|
||||
expect(extractLastCommand(messages)).toEqual({ kind: "bash", code: "echo b", language: "bash" });
|
||||
});
|
||||
|
||||
it("joins eval cell code and reports the cell language", () => {
|
||||
it("extracts eval code from flat args and reports the language", () => {
|
||||
const py = [
|
||||
assistantCalls([{ name: "eval", arguments: { language: "py", code: "print(1)" } }]),
|
||||
] as unknown as AgentMessage[];
|
||||
expect(extractLastCommand(py)).toEqual({ kind: "eval", code: "print(1)", language: "python" });
|
||||
|
||||
const js = [
|
||||
assistantCalls([{ name: "eval", arguments: { language: "js", code: "log(1)" } }]),
|
||||
] as unknown as AgentMessage[];
|
||||
expect(extractLastCommand(js)?.language).toBe("javascript");
|
||||
});
|
||||
|
||||
it("still joins legacy multi-cell eval args from older transcripts", () => {
|
||||
const py = [
|
||||
assistantCalls([
|
||||
{ name: "eval", arguments: { cells: [{ language: "py", code: "print(1)" }, { code: "print(2)" }] } },
|
||||
]),
|
||||
] as unknown as AgentMessage[];
|
||||
expect(extractLastCommand(py)).toEqual({ kind: "eval", code: "print(1)\n\nprint(2)", language: "python" });
|
||||
|
||||
const js = [
|
||||
assistantCalls([{ name: "eval", arguments: { cells: [{ language: "js", code: "log(1)" }] } }]),
|
||||
] as unknown as AgentMessage[];
|
||||
expect(extractLastCommand(js)?.language).toBe("javascript");
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -242,7 +242,7 @@ describe("UiHelpers.renderInitialMessages — image replay", () => {
|
||||
await Settings.init({ inMemory: true, overrides: { "terminal.showImages": true } });
|
||||
setTerminalImageProtocol(ImageProtocol.Sixel);
|
||||
const transcript = transcriptWith([
|
||||
assistantToolCall("eval-image", "eval", { cells: [{ language: "py", code: "display(image)" }] }),
|
||||
assistantToolCall("eval-image", "eval", { language: "py", code: "display(image)" }),
|
||||
{
|
||||
role: "toolResult",
|
||||
toolCallId: "eval-image",
|
||||
@@ -279,9 +279,7 @@ describe("UiHelpers.renderInitialMessages — image replay", () => {
|
||||
isError: false,
|
||||
timestamp: 2,
|
||||
});
|
||||
session.appendMessage(
|
||||
assistantToolCall("eval-reopened", "eval", { cells: [{ language: "py", code: "display(image)" }] }),
|
||||
);
|
||||
session.appendMessage(assistantToolCall("eval-reopened", "eval", { language: "py", code: "display(image)" }));
|
||||
session.appendMessage({
|
||||
role: "toolResult",
|
||||
toolCallId: "eval-reopened",
|
||||
|
||||
@@ -435,7 +435,9 @@ describe("streaming tool call preview height (bounded across renderers)", () =>
|
||||
const hidden = total - window;
|
||||
const longLines = Array.from({ length: total }, (_, i) => `line-${i}`);
|
||||
const { lines, text } = renderPending("eval", {
|
||||
cells: [{ language: "js", title: "big", code: longLines.map(line => `const ${line} = 1;`).join("\n") }],
|
||||
language: "js",
|
||||
title: "big",
|
||||
code: longLines.map(line => `const ${line} = 1;`).join("\n"),
|
||||
});
|
||||
|
||||
expect(lines.length, "eval code preview should stay bounded").toBeLessThan(window + 10);
|
||||
|
||||
@@ -61,7 +61,7 @@ describe("eval renderer: viewport tail window for cell code", () => {
|
||||
|
||||
it("bounds the pending preview to the same live tail window", () => {
|
||||
const component = evalToolRenderer.renderCall(
|
||||
{ cells: [{ language: "py", code }] },
|
||||
{ language: "py", code },
|
||||
{ expanded: false, isPartial: true },
|
||||
theme,
|
||||
);
|
||||
|
||||
@@ -17,7 +17,7 @@ function makeSession(opts: { spawns?: string | null; backends?: Record<string, b
|
||||
} as unknown as ToolSession;
|
||||
}
|
||||
|
||||
/** Pull the model-facing cell-schema fields (sorted `language` enum + descriptions) from the wire schema. */
|
||||
/** Pull the model-facing cell-schema fields (sorted `language` enum + descriptions) from the flat wire schema. */
|
||||
function wireCellFields(tool: EvalTool): {
|
||||
languages: string[];
|
||||
languageDescription?: string;
|
||||
@@ -25,17 +25,11 @@ function wireCellFields(tool: EvalTool): {
|
||||
} {
|
||||
const wire = toolWireSchema(tool as unknown as AiTool) as {
|
||||
properties?: {
|
||||
cells?: {
|
||||
items?: {
|
||||
properties?: {
|
||||
language?: { enum?: string[]; const?: string; description?: string };
|
||||
code?: { description?: string };
|
||||
};
|
||||
};
|
||||
};
|
||||
language?: { enum?: string[]; const?: string; description?: string };
|
||||
code?: { description?: string };
|
||||
};
|
||||
};
|
||||
const props = wire.properties?.cells?.items?.properties;
|
||||
const props = wire.properties;
|
||||
const language = props?.language;
|
||||
const languages = Array.isArray(language?.enum)
|
||||
? [...language.enum].sort()
|
||||
@@ -86,15 +80,19 @@ describe("eval tool dynamic schema", () => {
|
||||
}
|
||||
});
|
||||
|
||||
it("hides rb/jl from the wire schema, summary, and description by default", () => {
|
||||
const fields = wireCellFields(new EvalTool(makeSession({})));
|
||||
it("hides rb/jl from the wire schema, summary, description, and examples by default", () => {
|
||||
const tool = new EvalTool(makeSession({}));
|
||||
const fields = wireCellFields(tool);
|
||||
// Default config: rb/jl off → the wire schema is byte-identical to the pre-feature py/js one.
|
||||
expect(fields.languages).toEqual(["js", "py"]);
|
||||
expect(fields.languageDescription).toBe('runtime: "py" for the IPython kernel, "js" for the persistent JS VM');
|
||||
expect(fields.codeDescription).toBe("cell body, verbatim. Use top-level await freely.");
|
||||
const tool = new EvalTool(makeSession({}));
|
||||
expect(fields.codeDescription).toBe("code to run in this eval call, verbatim. Use top-level await freely.");
|
||||
expect(tool.summary).toBe("Execute Python or JavaScript code in an in-process eval backend");
|
||||
expect(tool.description).not.toMatch(/ruby|julia/i);
|
||||
// Examples must not advertise a disabled backend.
|
||||
const exampleLangs = tool.examples.map(ex => ("call" in ex ? ex.call.language : null));
|
||||
expect(exampleLangs).toEqual(["py", "py", "py"]);
|
||||
expect(tool.examples.some(ex => "call" in ex && ex.call.language === "rb")).toBe(false);
|
||||
});
|
||||
|
||||
it("advertises rb/jl across enum, descriptions, summary, and prelude once enabled", () => {
|
||||
@@ -104,12 +102,15 @@ describe("eval tool dynamic schema", () => {
|
||||
expect(fields.languageDescription).toBe(
|
||||
'runtime: "py" for the IPython kernel, "js" for the persistent JS VM, "rb" for the persistent Ruby kernel, "jl" for the persistent Julia kernel',
|
||||
);
|
||||
expect(fields.codeDescription).toBe(
|
||||
"cell body, verbatim. Top-level `await` is available in py/js; rb/jl auto-display the last expression like a REPL.",
|
||||
expect(fields.codeDescription).toContain(
|
||||
"code to run in this eval call, verbatim. Top-level `await` is available in py/js; rb/jl auto-display the last expression like a REPL.",
|
||||
);
|
||||
expect(tool.summary).toBe("Execute Python, JavaScript, Ruby, or Julia code in a persistent eval backend");
|
||||
expect(tool.description).toMatch(/ruby/i);
|
||||
expect(tool.description).toMatch(/julia/i);
|
||||
// Ruby examples appear once rb is enabled.
|
||||
const rbExampleLangs = tool.examples.filter(ex => "call" in ex && ex.call.language === "rb");
|
||||
expect(rbExampleLangs.length).toBe(2);
|
||||
});
|
||||
|
||||
it("advertises only the enabled subset of optional backends", () => {
|
||||
|
||||
@@ -55,7 +55,8 @@ describe("EvalTool display() text surfacing", () => {
|
||||
|
||||
const tool = new EvalTool(makeSession());
|
||||
const result = await tool.execute("call-display-json", {
|
||||
cells: [{ language: "js", code: "```js\ndisplay({ stdout: 'hi', exit_code: 0 });\n```\n" }],
|
||||
language: "js",
|
||||
code: "```js\ndisplay({ stdout: 'hi', exit_code: 0 });\n```\n",
|
||||
});
|
||||
|
||||
const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n");
|
||||
@@ -76,7 +77,8 @@ describe("EvalTool display() text surfacing", () => {
|
||||
|
||||
const tool = new EvalTool(makeSession());
|
||||
const result = await tool.execute("call-mixed", {
|
||||
cells: [{ language: "js", code: "```js\nprint('before'); display([1,2,3]);\n```\n" }],
|
||||
language: "js",
|
||||
code: "```js\nprint('before'); display([1,2,3]);\n```\n",
|
||||
});
|
||||
|
||||
const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n");
|
||||
@@ -96,9 +98,8 @@ describe("EvalTool display() text surfacing", () => {
|
||||
|
||||
const tool = new EvalTool(makeSession());
|
||||
const result = await tool.execute("call-image", {
|
||||
cells: [
|
||||
{ language: "js", code: "```js\ndisplay({ type: 'image', data: '...', mimeType: 'image/png' });\n```\n" },
|
||||
],
|
||||
language: "js",
|
||||
code: "```js\ndisplay({ type: 'image', data: '...', mimeType: 'image/png' });\n```\n",
|
||||
});
|
||||
|
||||
const imageBlocks = result.content.filter(c => c.type === "image");
|
||||
@@ -125,12 +126,8 @@ describe("EvalTool display() text surfacing", () => {
|
||||
|
||||
const tool = new EvalTool(makeSession());
|
||||
const result = await tool.execute("call-large-image", {
|
||||
cells: [
|
||||
{
|
||||
language: "js",
|
||||
code: "```js\ndisplay({ type: 'image', data: largePng, mimeType: 'image/png' });\n```\n",
|
||||
},
|
||||
],
|
||||
language: "js",
|
||||
code: "```js\ndisplay({ type: 'image', data: largePng, mimeType: 'image/png' });\n```\n",
|
||||
});
|
||||
|
||||
const image = result.content.find(c => c.type === "image");
|
||||
@@ -154,7 +151,8 @@ describe("EvalTool display() text surfacing", () => {
|
||||
|
||||
const tool = new EvalTool(makeSession());
|
||||
const result = await tool.execute("call-empty", {
|
||||
cells: [{ language: "js", code: "```js\nconst x = 1;\n```\n" }],
|
||||
language: "js",
|
||||
code: "```js\nconst x = 1;\n```\n",
|
||||
});
|
||||
|
||||
const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n");
|
||||
@@ -172,7 +170,8 @@ describe("EvalTool display() text surfacing", () => {
|
||||
|
||||
const tool = new EvalTool(makeSession());
|
||||
const result = await tool.execute("call-huge", {
|
||||
cells: [{ language: "js", code: "```js\ndisplay({ payload: 'x'.repeat(20000) });\n```\n" }],
|
||||
language: "js",
|
||||
code: "```js\ndisplay({ payload: 'x'.repeat(20000) });\n```\n",
|
||||
});
|
||||
|
||||
const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n");
|
||||
|
||||
@@ -67,7 +67,8 @@ describe("EvalTool language dispatch", () => {
|
||||
|
||||
const tool = new EvalTool(makeSession());
|
||||
await tool.execute("call-js", {
|
||||
cells: [{ language: "js", code: "const x = 1;" }],
|
||||
language: "js",
|
||||
code: "const x = 1;",
|
||||
});
|
||||
|
||||
expect(jsExecuteSpy).toHaveBeenCalledTimes(1);
|
||||
@@ -82,26 +83,23 @@ describe("EvalTool language dispatch", () => {
|
||||
|
||||
const tool = new EvalTool(makeSession());
|
||||
await tool.execute("call-py", {
|
||||
cells: [{ language: "py", code: "print('hi')" }],
|
||||
language: "py",
|
||||
code: "print('hi')",
|
||||
});
|
||||
|
||||
expect(pythonExecuteSpy).toHaveBeenCalledTimes(1);
|
||||
expect(jsExecuteSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("interleaves backends across cells in a single call", async () => {
|
||||
it("dispatches each call to the backend named by its language", async () => {
|
||||
vi.spyOn(pyKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true });
|
||||
vi.spyOn(evalIndex.pythonBackend, "isAvailable").mockResolvedValue(true);
|
||||
const pythonExecuteSpy = vi.spyOn(evalIndex.pythonBackend, "execute").mockResolvedValue(mockResult);
|
||||
const jsExecuteSpy = vi.spyOn(evalIndex.jsBackend, "execute").mockResolvedValue(mockResult);
|
||||
|
||||
const tool = new EvalTool(makeSession());
|
||||
await tool.execute("call-mixed", {
|
||||
cells: [
|
||||
{ language: "py", code: "x = 1" },
|
||||
{ language: "js", code: "const y = 2;" },
|
||||
],
|
||||
});
|
||||
await tool.execute("call-py", { language: "py", code: "x = 1" });
|
||||
await tool.execute("call-js", { language: "js", code: "const y = 2;" });
|
||||
|
||||
expect(pythonExecuteSpy).toHaveBeenCalledTimes(1);
|
||||
expect(jsExecuteSpy).toHaveBeenCalledTimes(1);
|
||||
@@ -113,7 +111,8 @@ describe("EvalTool language dispatch", () => {
|
||||
const tool = new EvalTool(makeSession(settings));
|
||||
await expect(
|
||||
tool.execute("call-py-disabled", {
|
||||
cells: [{ language: "py", code: "print('hi')" }],
|
||||
language: "py",
|
||||
code: "print('hi')",
|
||||
}),
|
||||
).rejects.toThrow(/eval\.py = false/);
|
||||
});
|
||||
@@ -124,7 +123,8 @@ describe("EvalTool language dispatch", () => {
|
||||
const tool = new EvalTool(makeSession(settings));
|
||||
await expect(
|
||||
tool.execute("call-js-disabled", {
|
||||
cells: [{ language: "js", code: "const x = 1;" }],
|
||||
language: "js",
|
||||
code: "const x = 1;",
|
||||
}),
|
||||
).rejects.toThrow(/eval\.js = false/);
|
||||
});
|
||||
@@ -151,7 +151,8 @@ describe("EvalTool language dispatch", () => {
|
||||
|
||||
await expect(
|
||||
tool.execute("call-js-env-disabled", {
|
||||
cells: [{ language: "js", code: "const x = 1;" }],
|
||||
language: "js",
|
||||
code: "const x = 1;",
|
||||
}),
|
||||
).rejects.toThrow(/PI_JS=0/);
|
||||
});
|
||||
|
||||
@@ -31,7 +31,9 @@ describe("EvalTool timeout semantics", () => {
|
||||
// 1s budget; the cell idles for 5s and emits no status, so nothing extends
|
||||
// the budget — it must be cut off at the wall-clock limit.
|
||||
const result = await tool.execute("call-compute-timeout", {
|
||||
cells: [{ language: "js", code: "await Bun.sleep(2000); return 'never';", timeout: 1 }],
|
||||
language: "js",
|
||||
code: "await Bun.sleep(2000); return 'never';",
|
||||
timeout: 1,
|
||||
});
|
||||
|
||||
const text = result.content
|
||||
|
||||
@@ -1,6 +1,15 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added support for Ruby and Julia code cells in the eval tool
|
||||
|
||||
### Changed
|
||||
|
||||
- Updated the eval tool view to render the new single-cell eval args (flat `language`/`code`/`title`/`timeout`/`reset`) and to highlight Ruby (`rb`) and Julia (`jl`) cells with their own syntax instead of collapsing them to Python, while still parsing legacy multi-cell `cells` arrays and framed `input` strings from older transcripts.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Improved compatibility with legacy todo task transcripts
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
/**
|
||||
* `eval` (aliases: js, python, notebook) — code cells executed in the
|
||||
* persistent kernel. Args arrive either as the modern `cells` array or as a
|
||||
* legacy framed `input` string (`*** Cell`, `*** Begin LANG`, `===== info =====`);
|
||||
* both render as discrete highlighted cells. When the result carries typed
|
||||
* per-cell details, each cell's output is interleaved beneath its code.
|
||||
* persistent kernel (py/js/rb/jl). Args arrive either as the modern single-cell
|
||||
* flat shape (`language`/`code`/`title`/`timeout`/`reset`), a legacy `cells`
|
||||
* array, or a legacy framed `input` string (`*** Cell`, `*** Begin LANG`,
|
||||
* `===== info =====`); all render as discrete highlighted cells. When the result
|
||||
* carries typed per-cell details, each cell's output is interleaved beneath its code.
|
||||
*/
|
||||
import type { ReactNode } from "react";
|
||||
import { Badges, CodeBlock, InvalidArg, Note, Output, ResultImages, ResultText } from "../parts";
|
||||
@@ -18,13 +19,22 @@ interface EvalCell {
|
||||
code: string;
|
||||
}
|
||||
|
||||
const HLJS_LANG: Record<string, string> = { py: "python", js: "javascript", ts: "typescript" };
|
||||
const HLJS_LANG: Record<string, string> = {
|
||||
py: "python",
|
||||
js: "javascript",
|
||||
ts: "typescript",
|
||||
rb: "ruby",
|
||||
jl: "julia",
|
||||
};
|
||||
|
||||
/** Map an eval language token to its canonical short id, or null when unknown. */
|
||||
function evalLangAlias(token: string | undefined): string | null {
|
||||
const t = (token ?? "").toUpperCase();
|
||||
if (t === "PY" || t === "PYTHON" || t === "IPY" || t === "IPYTHON") return "py";
|
||||
if (t === "JS" || t === "JAVASCRIPT") return "js";
|
||||
if (t === "TS" || t === "TYPESCRIPT") return "ts";
|
||||
if (t === "RB" || t === "RUBY") return "rb";
|
||||
if (t === "JL" || t === "JULIA") return "jl";
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -215,7 +225,7 @@ function parseEvalCellsLegacy(input: string): EvalCell[] {
|
||||
const info = m[1] ?? "";
|
||||
let lang = inheritedLang;
|
||||
let title = "";
|
||||
const langMatch = info.match(/^(py|js|ts)(?::"([^"]*)")?/);
|
||||
const langMatch = info.match(/^(py|js|ts|rb|jl)(?::"([^"]*)")?/);
|
||||
if (langMatch) {
|
||||
lang = langMatch[1];
|
||||
if (langMatch[2]) title = langMatch[2];
|
||||
@@ -257,7 +267,7 @@ function cellsFromArgs(args: Record<string, unknown>, name: string): EvalCell[]
|
||||
if (timeout !== null) attrs.push(`t=${timeout}s`);
|
||||
if (item.reset === true) attrs.push("rst");
|
||||
out.push({
|
||||
lang: str(item.language) === "js" ? "js" : "py",
|
||||
lang: evalLangAlias(str(item.language) ?? undefined) ?? "py",
|
||||
title: str(item.title) ?? "",
|
||||
attrs,
|
||||
code: str(item.code) ?? "",
|
||||
@@ -268,7 +278,14 @@ function cellsFromArgs(args: Record<string, unknown>, name: string): EvalCell[]
|
||||
const input = str(args.input);
|
||||
if (input !== null) return parseEvalCells(input).filter(c => c.code !== "" || c.title !== "");
|
||||
const code = str(args.code);
|
||||
if (code !== null) return [{ lang: name === "js" ? "js" : "py", title: "", attrs: [], code }];
|
||||
if (code !== null) {
|
||||
const attrs: string[] = [];
|
||||
const timeout = num(args.timeout);
|
||||
if (timeout !== null) attrs.push(`t=${timeout}s`);
|
||||
if (args.reset === true) attrs.push("rst");
|
||||
const lang = evalLangAlias(str(args.language) ?? undefined) ?? (name === "js" ? "js" : "py");
|
||||
return [{ lang, title: str(args.title) ?? "", attrs, code }];
|
||||
}
|
||||
return [];
|
||||
}
|
||||
|
||||
@@ -296,7 +313,7 @@ function detailCellsOf(details: Record<string, unknown> | null): DetailCell[] {
|
||||
index: num(item.index) ?? i,
|
||||
title: str(item.title) ?? "",
|
||||
code: str(item.code) ?? "",
|
||||
lang: language === "js" ? "js" : language !== null ? "py" : null,
|
||||
lang: language !== null ? (evalLangAlias(language) ?? "py") : null,
|
||||
output: str(item.output) ?? "",
|
||||
status: str(item.status) ?? "",
|
||||
durationMs: num(item.durationMs),
|
||||
|
||||
Reference in New Issue
Block a user