feat(coding-agent): refactored eval tool to single-step execution

- Transitioned the eval tool from batch multi-cell execution to a single-step input structure with flat parameters.
- Updated core agent logic, UI components, and documentation to support state persistence across incremental eval calls.
- Restricted bash tool capabilities by requiring explicit use of `read` or `find` instead of `ls` or `find`.
- Added support for Ruby and Julia language runtimes to the eval tool and associated web renderers.
This commit is contained in:
can1357
2026-06-23 00:55:07 +02:00
parent 899c0ef08b
commit 060f4004e7
23 changed files with 248 additions and 222 deletions
+5
View File
@@ -1,9 +1,11 @@
# Changelog
## [Unreleased]
### Breaking Changes
- Renamed the eval `agent()` helper parameters `agent_type` → `agent` and `return_handle` → `handle` across every workflow runtime (Python, JavaScript, Ruby, Julia), so the names are identical in every language (no camelCase/snake_case split) and the agent-selection parameter matches the `task` tool's `agent`. The `__agent__` eval bridge wire protocol was renamed to match.
- Changed the `eval` tool to take a single cell per call (`{ language, code, title?, timeout?, reset? }`) instead of a `cells` array. State still persists per language across separate eval calls, tool calls, and `task` subagents, so each call is one logical step that reuses everything earlier calls defined — the array only encouraged re-importing/re-declaring the same setup in every batch. The schema, field descriptions, examples, system `eval.md`/`workflowz` helper docs, and the `[i/n]` cell-counter (now hidden for single cells) were updated to match; the renderer, ACP start-text, copy-targets, and collab-web tool view still parse legacy multi-cell transcripts.
### Added
@@ -11,6 +13,9 @@
### Changed
- Simplified `eval` tool to accept a single logical step (code block) instead of an array of cells
- Updated `eval` tool documentation to emphasize incremental, single-step execution
- Restricted `bash` tool from using `ls` or `find`, requiring the use of `read` or `find` tools
- Simplified `todo` tool interface to accept a single operation directly instead of an array of ops
- Reinforced routing of fragile, multi-step shell logic to the `eval` tool over `bash`. The system-prompt tool policy, `bash.md`, and `eval.md` now treat loops, conditionals, heredocs, inline `-e`/`-c` scripts, multi-stage pipelines, and quote/JSON escaping as the signal to write an `eval` cell; bash's "compute a fact" carveout is narrowed to single short pipelines, and `eval.md` now actively claims that territory with runtime-templated examples (only enabled backends are advertised).
- Made `eval` an essential built-in tool (`loadMode: "essential"`, added to the default essential tool set) so it stays active under `tools.discoveryMode: "all"` instead of being hidden behind `search_tool_bm25`.
@@ -59,31 +59,23 @@ export const shellFixtures: Record<string, GalleryFixture> = {
eval: {
label: "Eval",
streamingArgs: {
cells: [
{
language: "py",
code: 'import json\nfrom pathlib import Path\n\ndata = json.loads(Path("package.js',
title: "load config",
},
],
language: "py",
code: 'import json\nfrom pathlib import Path\n\ndata = json.loads(Path("package.js',
title: "load config",
},
args: {
cells: [
{
language: "py",
title: "load config",
code: [
"import json",
"from pathlib import Path",
"",
'data = json.loads(Path("package.json").read_text())',
'deps = data.get("dependencies", {})',
'print(f"{data[\\"name\\"]} v{data[\\"version\\"]}")',
'print(f"{len(deps)} dependencies")',
"display(sorted(deps)[:3])",
].join("\n"),
},
],
language: "py",
title: "load config",
code: [
"import json",
"from pathlib import Path",
"",
'data = json.loads(Path("package.json").read_text())',
'deps = data.get("dependencies", {})',
'print(f"{data[\\"name\\"]} v{data[\\"version\\"]}")',
'print(f"{len(deps)} dependencies")',
"display(sorted(deps)[:3])",
].join("\n"),
},
result: {
content: [
@@ -468,8 +468,13 @@ function buildEvalStartText(args: unknown): string | undefined {
if (typeof args !== "object" || args === null || Array.isArray(args)) {
return undefined;
}
const cells = (args as EvalCellContainer).cells;
if (!Array.isArray(cells) || cells.length === 0) {
const container = args as EvalCellContainer & EvalCellLike;
const cells = Array.isArray(container.cells)
? container.cells
: typeof container.code === "string"
? [container]
: [];
if (cells.length === 0) {
return undefined;
}
const lines: string[] = [];
@@ -127,8 +127,13 @@ export function extractQuoteBlocks(text: string): QuoteBlock[] {
function extractEvalCode(args: unknown): { code: string; language: string } | undefined {
if (!args || typeof args !== "object") return undefined;
const cells = (args as { cells?: unknown }).cells;
if (!Array.isArray(cells)) return undefined;
const argsObj = args as { cells?: unknown; code?: unknown };
const cells = Array.isArray(argsObj.cells)
? argsObj.cells
: typeof argsObj.code === "string"
? [argsObj]
: undefined;
if (!cells) return undefined;
const codeBlocks: string[] = [];
let language = "python";
@@ -11,7 +11,7 @@ Worth it when the task benefits from decomposition + parallel coverage, or from
</when>
<helpers>
State persists across cells, so scout in one cell and fan out in the next. Every cell has:
State persists across eval calls, so scout in one call and fan out in the next. Every eval call has:
- `agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts.
- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one.
@@ -20,7 +20,7 @@ State persists across cells, so scout in one cell and fan out in the next. Every
- `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it.
- `budget` — `budget.total` (output-token ceiling, or `None` when none is set), `budget.spent()` (tokens spent this turn — main loop + eval subagents), `budget.remaining()` (`math.inf` when total is `None`), `budget.hard` (whether it's enforced). A ceiling is set by the user: `+Nk` in their message is advisory (you self-limit via `budget.remaining()`), `+Nk!` (or Goal Mode) is hard — `agent()` refuses to spawn once spent reaches it. Gate loops on `budget.total` first, since it's `None` when the user set no budget.
Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across cells and turns for multi-phase work, reading each result before you decide the next phase.
Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across calls and turns for multi-phase work, reading each result before you decide the next phase.
</helpers>
<structure>
@@ -30,6 +30,7 @@ Anything below → `eval` cell, not bash:
<critical>
- Bash invokes real binaries with simple args; it is NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps.
- NEVER shell out to search content or files: `grep/rg` → `search`.
- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `find` tool (globbing). This is non-negotiable, even for a single quick listing.
- Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://<id>`.
</critical>
@@ -1,22 +1,23 @@
Run code in a persistent kernel using a list of cells.
Run one step of code in a persistent kernel.
<instruction>
Cells run in array order. State persists per language across cells, tool calls, and `task` subagents — stage helpers/datasets/clients once, subagents reuse directly, no re-import/serialize.
**One eval call = one cell = one logical step.** State persists per language across separate eval calls, tool calls, and `task` subagents — define helpers/datasets/clients in one call, then later calls reuse them directly.
Cell fields:
Work incrementally: imports in one call, define in the next, test, then use — each its own eval call. Re-run setup ONLY after `reset`, a kernel crash, or a `NameError`/`ReferenceError` proving the state is gone. Parallelize work *within* a cell with the `parallel(thunks)` helper, not by batching steps.
Fields:
- `language` — {{#if py}}`"py"` IPython kernel{{/if}}{{#ifAll py js}}, {{/ifAll}}{{#if js}}`"js"` persistent JavaScript VM{{/if}}{{#if rb}}{{#ifAny py js}}, {{/ifAny}}`"rb"` persistent Ruby kernel{{/if}}{{#if jl}}{{#ifAny py js rb}}, {{/ifAny}}`"jl"` persistent Julia kernel{{/if}}.
- `code` — cell body, verbatim. Newlines/quotes JSON-encoded; no fences, no headers.
- `title` (optional) — short transcript label (e.g. `"imports"`).
- `timeout` (optional) — per-cell seconds. Raise only for heavy compute or long non-agent tool calls.
- `reset` (optional) — wipe this cell's language kernel first.{{#ifAll py js}} Per-language: a `py` reset never touches the JS VM.{{/ifAll}}
- `timeout` (optional) — seconds. Raise only for heavy compute or long non-agent tool calls.
- `reset` (optional) — wipe this language's kernel first.{{#ifAll py js}} Per-language: a `py` reset never touches the JS VM.{{/ifAll}}
Work incrementally — one logical step per cell (imports, define, test, use), many small cells per call; workflow notes in the assistant message or `title`, never in cell code.
{{#if py}}Live event loop: use top-level `await` directly; `asyncio.run(…)` raises "cannot be called from a running event loop".{{/if}}
{{#if js}}JS runs under **Bun**: Bun globals/APIs are available (`Bun.file`, `Bun.write`, `Bun.$`, `fetch`, `Buffer`); top-level `await`/`return` work directly.{{/if}}
{{#if rb}}Ruby: synchronous; helper options are keyword args (e.g. `tree(".", max_depth: 2)`); the last expression auto-displays unless it is `nil`, an assignment, or a definition (like IRB).{{/if}}
{{#if jl}}Julia: synchronous; helper options are standard keyword args (e.g. `tree(max_depth=2)`); the last expression auto-displays unless it is an assignment or a definition (like the Julia REPL).{{/if}}
Errors name the failing cell ("Cell 3 failed") — resubmit the fixed cell + any remaining.
On error, fix and re-run only the failing step — prior calls' state survives.
</instruction>
<prelude>
@@ -71,3 +72,7 @@ Pipe handles through stage helpers to build a dependency graph — acyclic waves
- **Acyclic only.** A node never waits on its own descendant.
</dag>
{{/if}}
<critical>
Prior top-level names (`data`, `sessions`, helpers, imports) survive into the next eval call — reuse them; NEVER re-import, re-require, or re-declare a helper. Re-read a file only if it may have changed since the last read. Re-run setup only after `reset`, a crash, or a `NameError`/`ReferenceError`.
</critical>
@@ -56,6 +56,9 @@ interface EvalRenderCellArg {
}
interface EvalRenderArgs {
language?: string;
code?: string;
title?: string;
cells?: EvalRenderCellArg[];
__partialJson?: string;
}
@@ -81,8 +84,8 @@ function normalizeRenderLanguage(value: string | undefined): EvalLanguage {
}
function getRenderCells(args: EvalRenderArgs | undefined): EvalRenderCell[] {
const raw = args?.cells;
if (!Array.isArray(raw)) return [];
if (!args) return [];
const raw = Array.isArray(args.cells) ? args.cells : typeof args.code === "string" ? [args] : [];
const out: EvalRenderCell[] = [];
for (const cell of raw) {
if (!cell || typeof cell !== "object") continue;
+87 -103
View File
@@ -38,8 +38,6 @@ const EVAL_LANGUAGE_NAME: Record<EvalLanguageToken, string> = {
rb: "Ruby",
jl: "Julia",
};
const EVAL_CELLS_DESCRIPTION =
"cells executed in order. State persists within each language across cells and tool calls.";
/** Join names as an English "or" list: ["A"]→"A", ["A","B"]→"A or B", 3+→"A, B, or C". */
function joinWithOr(items: readonly string[]): string {
@@ -56,12 +54,12 @@ function describeCodeField(langs: readonly EvalLanguageToken[]): string {
const replLangs = langs.filter(lang => lang === "rb" || lang === "jl");
// No persistent REPL backends → keep the original py/js phrasing verbatim so the
// default (rb/jl off) wire schema stays byte-identical to the pre-feature one.
if (replLangs.length === 0) return "cell body, verbatim. Use top-level await freely.";
if (replLangs.length === 0) return "code to run in this eval call, verbatim. Use top-level await freely.";
const awaitLangs = langs.filter(lang => lang === "py" || lang === "js");
const clauses: string[] = [];
if (awaitLangs.length > 0) clauses.push(`Top-level \`await\` is available in ${awaitLangs.join("/")}`);
clauses.push(`${replLangs.join("/")} auto-display the last expression like a REPL`);
return `cell body, verbatim. ${clauses.join("; ")}.`;
return `code to run in this eval call, verbatim. ${clauses.join("; ")}.`;
}
/** One-line discovery summary listing the runtimes available this session. */
@@ -86,29 +84,24 @@ function enabledEvalLanguages(backends: EvalBackendsAllowance): EvalLanguageToke
const evalCellCommonFields = {
"title?": type("string").describe('short label shown in transcript (e.g. "imports", "load config")'),
"timeout?": type("number").describe("per-cell timeout in seconds"),
"reset?": type("boolean").describe(
"wipe this cell's language kernel before running. Other languages are untouched.",
),
"timeout?": type("number").describe("timeout for this eval call in seconds"),
"reset?": type("boolean").describe("wipe this language's kernel before running. Other languages are untouched."),
};
/**
* Per-cell input. Each cell runs in order; state persists within a language
* across cells and across tool calls. This static schema carries the full
* language union for typing; {@link buildEvalSchema} narrows the wire copy per
* session so disabled backends are never advertised to the model.
* Per-call input: a single cell. State persists within a language across
* separate eval calls and across tool calls, so each call is one logical step
* and later calls reuse what earlier ones defined. This static schema carries
* the full language union for typing; {@link buildEvalSchema} narrows the wire
* copy per session so disabled backends are never advertised to the model.
*/
const evalCellSchema = type({
language: type("'py' | 'js' | 'rb' | 'jl'").describe(describeLanguageField(EVAL_LANGUAGE_ORDER)),
code: type("string").describe(describeCodeField(EVAL_LANGUAGE_ORDER)),
...evalCellCommonFields,
});
export type EvalCellInput = typeof evalCellSchema.infer;
export const evalSchema = type({
cells: evalCellSchema.array().atLeastLength(1).describe(EVAL_CELLS_DESCRIPTION),
language: type("'py' | 'js' | 'rb' | 'jl'").describe(describeLanguageField(EVAL_LANGUAGE_ORDER)),
...evalCellCommonFields,
code: type("string").describe(describeCodeField(EVAL_LANGUAGE_ORDER)),
});
export type EvalToolParams = typeof evalSchema.infer;
export type EvalCellInput = EvalToolParams;
/**
* Build a session-scoped copy of the eval schema whose `language` enum and field
@@ -118,14 +111,11 @@ export type EvalToolParams = typeof evalSchema.infer;
* {@link evalSchema} (full union) remains the type-level source of truth.
*/
function buildEvalSchema(langs: readonly EvalLanguageToken[]): typeof evalSchema {
const cellSchema = type({
const schema = type({
language: type.enumerated(...langs).describe(describeLanguageField(langs)),
code: type("string").describe(describeCodeField(langs)),
...evalCellCommonFields,
});
const schema = type({
cells: cellSchema.array().atLeastLength(1).describe(EVAL_CELLS_DESCRIPTION),
});
return schema as unknown as typeof evalSchema;
}
@@ -290,17 +280,10 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
readonly approval = "exec" as const;
readonly formatApprovalDetails = (args: unknown): string[] => {
const params = args as Partial<EvalToolParams>;
const cells = Array.isArray(params.cells) ? params.cells : [];
const firstCell = cells[0] as Partial<EvalCellInput> | undefined;
if (!firstCell) return [];
const language =
typeof firstCell.language === "string" ? formatEvalInputLanguage(firstCell.language) : "javascript (default)";
const code = typeof firstCell.code === "string" ? firstCell.code : "";
const lines = [`Language: ${language}`, `Code:\n${truncateForPrompt(code)}`];
if (cells.length > 1) {
lines.push(`+${cells.length - 1} more cell${cells.length === 2 ? "" : "s"}`);
}
return lines;
typeof params.language === "string" ? formatEvalInputLanguage(params.language) : "javascript (default)";
const code = typeof params.code === "string" ? params.code : "";
return [`Language: ${language}`, `Code:\n${truncateForPrompt(code)}`];
};
get summary(): string {
return summarizeEvalLanguages(this.#enabledLanguages());
@@ -320,25 +303,53 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
spawns: spawnsAllowed,
});
}
readonly examples: readonly ToolExample<typeof evalSchema.infer>[] = [
/** All reuse-chain examples; the `examples` getter filters by enabled languages. */
private static readonly ALL_EXAMPLES: readonly ToolExample<typeof evalSchema.infer>[] = [
{
caption: "First call — set up once",
call: {
cells: [
{
language: "py",
title: "imports",
timeout: 10,
code: "import json\nfrom pathlib import Path",
},
{
language: "py",
title: "load config",
code: "data = json.loads(read('package.json'))\ndisplay(data)",
},
],
language: "py",
title: "imports",
code: "import json\nfrom pathlib import Path",
},
},
{
caption: "Second call — reuse, do NOT re-import",
call: {
language: "py",
title: "load config",
code: "data = json.loads(read('package.json'))\ndisplay(data)",
},
},
{
caption: "Third call — reuse the loaded config",
call: {
language: "py",
title: "scan deps",
code: "display(sorted(data['dependencies']))",
},
},
{
caption: "Ruby first call — set up once",
call: {
language: "rb",
title: "setup",
code: "require 'json'\npkg_path = 'package.json'",
},
},
{
caption: "Ruby second call — reuse, do NOT re-require",
call: {
language: "rb",
title: "load config",
code: "pkg = JSON.parse(read(pkg_path))\ndisplay(pkg.keys.sort)",
},
},
];
get examples(): readonly ToolExample<typeof evalSchema.infer>[] {
const langs = new Set(this.#enabledLanguages());
return EvalTool.ALL_EXAMPLES.filter(ex => "call" in ex && langs.has(ex.call.language as EvalLanguageToken));
}
get parameters(): typeof evalSchema {
const langs = this.#enabledLanguages();
if (langs.length === 0 || langs.length === EVAL_LANGUAGE_ORDER.length) return evalSchema;
@@ -352,13 +363,9 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
readonly concurrency = "exclusive";
readonly strict = true;
readonly intent = (args: Partial<typeof evalSchema.infer>): string | undefined => {
const cells = Array.isArray(args.cells) ? args.cells : [];
const first = cells.find(c => c && typeof c === "object");
if (!first) return "evaluating";
const title = typeof first.title === "string" ? first.title : undefined;
const language = typeof first.language === "string" ? formatEvalInputLanguage(first.language) : "javascript";
const label = title || `running ${language}`;
return cells.length > 1 ? `${label} (+${cells.length - 1})` : label;
const title = typeof args.title === "string" ? args.title : undefined;
const language = typeof args.language === "string" ? formatEvalInputLanguage(args.language) : "javascript";
return title || `running ${language}`;
};
readonly #proxyExecutor?: EvalProxyExecutor;
@@ -398,27 +405,25 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
const session = this.session;
const excludeWebP = webpExclusionForModel(session.getActiveModel?.());
const cells: ResolvedEvalCell[] = [];
for (let i = 0; i < params.cells.length; i++) {
const cell = params.cells[i];
const language: EvalLanguage =
cell.language === "py"
? "python"
: cell.language === "rb"
? "ruby"
: cell.language === "jl"
? "julia"
: "js";
const resolved = await resolveBackend(session, language);
cells.push({
index: i,
title: cell.title,
code: cell.code,
timeoutMs: (cell.timeout ?? 30) * 1000,
reset: cell.reset ?? false,
const cellLanguage: EvalLanguage =
params.language === "py"
? "python"
: params.language === "rb"
? "ruby"
: params.language === "jl"
? "julia"
: "js";
const resolved = await resolveBackend(session, cellLanguage);
const cells: ResolvedEvalCell[] = [
{
index: 0,
title: params.title,
code: params.code,
timeoutMs: (params.timeout ?? 30) * 1000,
reset: params.reset ?? false,
resolved,
});
}
},
];
const languages = uniqueEvalLanguages(cells);
const notice = detailsNotice(cells);
const sessionAbortController = new AbortController();
@@ -623,24 +628,9 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
cellResult.statusEvents = cellStatusEvents.length > 0 ? cellStatusEvents : undefined;
cellResult.hasMarkdown = cellHasMarkdown || undefined;
let combinedCellOutput = "";
if (cells.length > 1) {
const cellHeader = `[${i + 1}/${cells.length}]`;
const cellTitle = cell.title ? ` ${cell.title}` : "";
if (cellOutput) {
combinedCellOutput = `${cellHeader}${cellTitle}\n${cellOutput}`;
} else {
combinedCellOutput = `${cellHeader}${cellTitle} (ok)`;
}
cellOutputs.push(combinedCellOutput);
} else if (cellOutput) {
combinedCellOutput = cellOutput;
cellOutputs.push(combinedCellOutput);
}
if (combinedCellOutput) {
const prefix = cellOutputs.length > 1 ? "\n\n" : "";
appendTail(`${prefix}${combinedCellOutput}`);
if (cellOutput) {
cellOutputs.push(cellOutput);
appendTail(cellOutput);
}
if (result.cancelled) {
@@ -648,10 +638,7 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
pushUpdate();
const errorMsg = result.output || "Command aborted";
const combinedOutput = cellOutputs.join("\n\n");
const outputText =
cells.length > 1
? `${combinedOutput}\n\nCell ${i + 1} aborted: ${errorMsg}`
: combinedOutput || errorMsg;
const outputText = combinedOutput || errorMsg;
const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput);
const details: EvalToolDetails = {
@@ -674,12 +661,9 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
cellResult.status = "error";
pushUpdate();
const combinedOutput = cellOutputs.join("\n\n");
const outputText =
cells.length > 1
? `${combinedOutput}\n\nCell ${i + 1} failed (exit code ${result.exitCode}). Earlier cells succeeded—their state persists. Fix only cell ${i + 1}.`
: combinedOutput
? `${combinedOutput}\n\nCommand exited with code ${result.exitCode}`
: `Command exited with code ${result.exitCode}`;
const outputText = combinedOutput
? `${combinedOutput}\n\nCommand exited with code ${result.exitCode}`
: `Command exited with code ${result.exitCode}`;
const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput);
const details: EvalToolDetails = {
+1 -1
View File
@@ -80,7 +80,7 @@ function formatHeader(options: CodeCellOptions, theme: Theme): { title: string;
parts.push(icon);
}
}
if (index !== undefined && total !== undefined) {
if (index !== undefined && total !== undefined && total > 1) {
parts.push(theme.fg("accent", `[${index + 1}/${total}]`));
}
if (title) {
@@ -141,11 +141,7 @@ export async function resizeImage(img: ImageContent, options?: ImageResizeOption
// lagging edge up to the floor via the default fit:"fill" resize.
if (targetWidth < minDimension || targetHeight < minDimension) {
const shortEdge = Math.min(targetWidth, targetHeight);
const upscale = Math.min(
minDimension / shortEdge,
opts.maxWidth / targetWidth,
opts.maxHeight / targetHeight,
);
const upscale = Math.min(minDimension / shortEdge, opts.maxWidth / targetWidth, opts.maxHeight / targetHeight);
if (upscale > 1) {
targetWidth = Math.round(targetWidth * upscale);
targetHeight = Math.round(targetHeight * upscale);
@@ -247,7 +247,7 @@ describe("ACP event mapper", () => {
type: "tool_execution_start",
toolCallId: "tc-eval-start",
toolName: "eval",
args: { cells: [{ language: "js", title: "sum", code: "return 1 + 1;" }] },
args: { language: "js", title: "sum", code: "return 1 + 1;" },
intent: "sum",
} as AgentSessionEvent,
"session-1",
@@ -267,7 +267,7 @@ describe("ACP event mapper", () => {
expect(update.title).toBe("[js] sum\nreturn 1 + 1;");
expect(update.kind).toBe("execute");
expect(update.status).toBe("pending");
expect(update.rawInput).toEqual({ cells: [{ language: "js", title: "sum", code: "return 1 + 1;" }] });
expect(update.rawInput).toEqual({ language: "js", title: "sum", code: "return 1 + 1;" });
expect(update.content).toContainEqual({
type: "content",
content: { type: "text", text: "[js] sum\nreturn 1 + 1;" },
@@ -305,7 +305,7 @@ describe("ACP event mapper", () => {
type: "tool_execution_start",
toolCallId: "tc-eval-long-source",
toolName: "eval",
args: { cells: [{ language: "js", code: source }] },
args: { language: "js", code: source },
} as AgentSessionEvent,
"session-1",
);
@@ -426,7 +426,7 @@ describe("AgentSession python cleanup", () => {
expect(EvalTool).toBeDefined();
let toolExecutionSettled = false;
const toolExecution = EvalTool!
.execute("call-id", { cells: [{ language: "py", code: "print('tool')" }] }, undefined, undefined, undefined)
.execute("call-id", { language: "py", code: "print('tool')" }, undefined, undefined, undefined)
.finally(() => {
toolExecutionSettled = true;
});
@@ -652,13 +652,7 @@ describe("AgentSession python cleanup", () => {
expect(EvalTool).toBeDefined();
const disposeSession = session.dispose();
await expect(
EvalTool!.execute(
"call-id",
{ cells: [{ language: "py", code: "print('late')" }] },
undefined,
undefined,
undefined,
),
EvalTool!.execute("call-id", { language: "py", code: "print('late')" }, undefined, undefined, undefined),
).rejects.toThrow("Python execution is unavailable while session disposal is in progress");
await disposeSession;
expect(executeSpy).not.toHaveBeenCalled();
@@ -693,7 +687,7 @@ describe("AgentSession python cleanup", () => {
expect(EvalTool).toBeDefined();
const execution = EvalTool!.execute(
"call-id",
{ cells: [{ language: "py", code: "print('late after artifact')" }] },
{ language: "py", code: "print('late after artifact')" },
undefined,
undefined,
undefined,
@@ -76,18 +76,25 @@ describe("extractLastCommand", () => {
expect(extractLastCommand(messages)).toEqual({ kind: "bash", code: "echo b", language: "bash" });
});
it("joins eval cell code and reports the cell language", () => {
it("extracts eval code from flat args and reports the language", () => {
const py = [
assistantCalls([{ name: "eval", arguments: { language: "py", code: "print(1)" } }]),
] as unknown as AgentMessage[];
expect(extractLastCommand(py)).toEqual({ kind: "eval", code: "print(1)", language: "python" });
const js = [
assistantCalls([{ name: "eval", arguments: { language: "js", code: "log(1)" } }]),
] as unknown as AgentMessage[];
expect(extractLastCommand(js)?.language).toBe("javascript");
});
it("still joins legacy multi-cell eval args from older transcripts", () => {
const py = [
assistantCalls([
{ name: "eval", arguments: { cells: [{ language: "py", code: "print(1)" }, { code: "print(2)" }] } },
]),
] as unknown as AgentMessage[];
expect(extractLastCommand(py)).toEqual({ kind: "eval", code: "print(1)\n\nprint(2)", language: "python" });
const js = [
assistantCalls([{ name: "eval", arguments: { cells: [{ language: "js", code: "log(1)" }] } }]),
] as unknown as AgentMessage[];
expect(extractLastCommand(js)?.language).toBe("javascript");
});
});
@@ -242,7 +242,7 @@ describe("UiHelpers.renderInitialMessages — image replay", () => {
await Settings.init({ inMemory: true, overrides: { "terminal.showImages": true } });
setTerminalImageProtocol(ImageProtocol.Sixel);
const transcript = transcriptWith([
assistantToolCall("eval-image", "eval", { cells: [{ language: "py", code: "display(image)" }] }),
assistantToolCall("eval-image", "eval", { language: "py", code: "display(image)" }),
{
role: "toolResult",
toolCallId: "eval-image",
@@ -279,9 +279,7 @@ describe("UiHelpers.renderInitialMessages — image replay", () => {
isError: false,
timestamp: 2,
});
session.appendMessage(
assistantToolCall("eval-reopened", "eval", { cells: [{ language: "py", code: "display(image)" }] }),
);
session.appendMessage(assistantToolCall("eval-reopened", "eval", { language: "py", code: "display(image)" }));
session.appendMessage({
role: "toolResult",
toolCallId: "eval-reopened",
@@ -435,7 +435,9 @@ describe("streaming tool call preview height (bounded across renderers)", () =>
const hidden = total - window;
const longLines = Array.from({ length: total }, (_, i) => `line-${i}`);
const { lines, text } = renderPending("eval", {
cells: [{ language: "js", title: "big", code: longLines.map(line => `const ${line} = 1;`).join("\n") }],
language: "js",
title: "big",
code: longLines.map(line => `const ${line} = 1;`).join("\n"),
});
expect(lines.length, "eval code preview should stay bounded").toBeLessThan(window + 10);
@@ -61,7 +61,7 @@ describe("eval renderer: viewport tail window for cell code", () => {
it("bounds the pending preview to the same live tail window", () => {
const component = evalToolRenderer.renderCall(
{ cells: [{ language: "py", code }] },
{ language: "py", code },
{ expanded: false, isPartial: true },
theme,
);
@@ -17,7 +17,7 @@ function makeSession(opts: { spawns?: string | null; backends?: Record<string, b
} as unknown as ToolSession;
}
/** Pull the model-facing cell-schema fields (sorted `language` enum + descriptions) from the wire schema. */
/** Pull the model-facing cell-schema fields (sorted `language` enum + descriptions) from the flat wire schema. */
function wireCellFields(tool: EvalTool): {
languages: string[];
languageDescription?: string;
@@ -25,17 +25,11 @@ function wireCellFields(tool: EvalTool): {
} {
const wire = toolWireSchema(tool as unknown as AiTool) as {
properties?: {
cells?: {
items?: {
properties?: {
language?: { enum?: string[]; const?: string; description?: string };
code?: { description?: string };
};
};
};
language?: { enum?: string[]; const?: string; description?: string };
code?: { description?: string };
};
};
const props = wire.properties?.cells?.items?.properties;
const props = wire.properties;
const language = props?.language;
const languages = Array.isArray(language?.enum)
? [...language.enum].sort()
@@ -86,15 +80,19 @@ describe("eval tool dynamic schema", () => {
}
});
it("hides rb/jl from the wire schema, summary, and description by default", () => {
const fields = wireCellFields(new EvalTool(makeSession({})));
it("hides rb/jl from the wire schema, summary, description, and examples by default", () => {
const tool = new EvalTool(makeSession({}));
const fields = wireCellFields(tool);
// Default config: rb/jl off → the wire schema is byte-identical to the pre-feature py/js one.
expect(fields.languages).toEqual(["js", "py"]);
expect(fields.languageDescription).toBe('runtime: "py" for the IPython kernel, "js" for the persistent JS VM');
expect(fields.codeDescription).toBe("cell body, verbatim. Use top-level await freely.");
const tool = new EvalTool(makeSession({}));
expect(fields.codeDescription).toBe("code to run in this eval call, verbatim. Use top-level await freely.");
expect(tool.summary).toBe("Execute Python or JavaScript code in an in-process eval backend");
expect(tool.description).not.toMatch(/ruby|julia/i);
// Examples must not advertise a disabled backend.
const exampleLangs = tool.examples.map(ex => ("call" in ex ? ex.call.language : null));
expect(exampleLangs).toEqual(["py", "py", "py"]);
expect(tool.examples.some(ex => "call" in ex && ex.call.language === "rb")).toBe(false);
});
it("advertises rb/jl across enum, descriptions, summary, and prelude once enabled", () => {
@@ -104,12 +102,15 @@ describe("eval tool dynamic schema", () => {
expect(fields.languageDescription).toBe(
'runtime: "py" for the IPython kernel, "js" for the persistent JS VM, "rb" for the persistent Ruby kernel, "jl" for the persistent Julia kernel',
);
expect(fields.codeDescription).toBe(
"cell body, verbatim. Top-level `await` is available in py/js; rb/jl auto-display the last expression like a REPL.",
expect(fields.codeDescription).toContain(
"code to run in this eval call, verbatim. Top-level `await` is available in py/js; rb/jl auto-display the last expression like a REPL.",
);
expect(tool.summary).toBe("Execute Python, JavaScript, Ruby, or Julia code in a persistent eval backend");
expect(tool.description).toMatch(/ruby/i);
expect(tool.description).toMatch(/julia/i);
// Ruby examples appear once rb is enabled.
const rbExampleLangs = tool.examples.filter(ex => "call" in ex && ex.call.language === "rb");
expect(rbExampleLangs.length).toBe(2);
});
it("advertises only the enabled subset of optional backends", () => {
@@ -55,7 +55,8 @@ describe("EvalTool display() text surfacing", () => {
const tool = new EvalTool(makeSession());
const result = await tool.execute("call-display-json", {
cells: [{ language: "js", code: "```js\ndisplay({ stdout: 'hi', exit_code: 0 });\n```\n" }],
language: "js",
code: "```js\ndisplay({ stdout: 'hi', exit_code: 0 });\n```\n",
});
const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n");
@@ -76,7 +77,8 @@ describe("EvalTool display() text surfacing", () => {
const tool = new EvalTool(makeSession());
const result = await tool.execute("call-mixed", {
cells: [{ language: "js", code: "```js\nprint('before'); display([1,2,3]);\n```\n" }],
language: "js",
code: "```js\nprint('before'); display([1,2,3]);\n```\n",
});
const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n");
@@ -96,9 +98,8 @@ describe("EvalTool display() text surfacing", () => {
const tool = new EvalTool(makeSession());
const result = await tool.execute("call-image", {
cells: [
{ language: "js", code: "```js\ndisplay({ type: 'image', data: '...', mimeType: 'image/png' });\n```\n" },
],
language: "js",
code: "```js\ndisplay({ type: 'image', data: '...', mimeType: 'image/png' });\n```\n",
});
const imageBlocks = result.content.filter(c => c.type === "image");
@@ -125,12 +126,8 @@ describe("EvalTool display() text surfacing", () => {
const tool = new EvalTool(makeSession());
const result = await tool.execute("call-large-image", {
cells: [
{
language: "js",
code: "```js\ndisplay({ type: 'image', data: largePng, mimeType: 'image/png' });\n```\n",
},
],
language: "js",
code: "```js\ndisplay({ type: 'image', data: largePng, mimeType: 'image/png' });\n```\n",
});
const image = result.content.find(c => c.type === "image");
@@ -154,7 +151,8 @@ describe("EvalTool display() text surfacing", () => {
const tool = new EvalTool(makeSession());
const result = await tool.execute("call-empty", {
cells: [{ language: "js", code: "```js\nconst x = 1;\n```\n" }],
language: "js",
code: "```js\nconst x = 1;\n```\n",
});
const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n");
@@ -172,7 +170,8 @@ describe("EvalTool display() text surfacing", () => {
const tool = new EvalTool(makeSession());
const result = await tool.execute("call-huge", {
cells: [{ language: "js", code: "```js\ndisplay({ payload: 'x'.repeat(20000) });\n```\n" }],
language: "js",
code: "```js\ndisplay({ payload: 'x'.repeat(20000) });\n```\n",
});
const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n");
@@ -67,7 +67,8 @@ describe("EvalTool language dispatch", () => {
const tool = new EvalTool(makeSession());
await tool.execute("call-js", {
cells: [{ language: "js", code: "const x = 1;" }],
language: "js",
code: "const x = 1;",
});
expect(jsExecuteSpy).toHaveBeenCalledTimes(1);
@@ -82,26 +83,23 @@ describe("EvalTool language dispatch", () => {
const tool = new EvalTool(makeSession());
await tool.execute("call-py", {
cells: [{ language: "py", code: "print('hi')" }],
language: "py",
code: "print('hi')",
});
expect(pythonExecuteSpy).toHaveBeenCalledTimes(1);
expect(jsExecuteSpy).not.toHaveBeenCalled();
});
it("interleaves backends across cells in a single call", async () => {
it("dispatches each call to the backend named by its language", async () => {
vi.spyOn(pyKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true });
vi.spyOn(evalIndex.pythonBackend, "isAvailable").mockResolvedValue(true);
const pythonExecuteSpy = vi.spyOn(evalIndex.pythonBackend, "execute").mockResolvedValue(mockResult);
const jsExecuteSpy = vi.spyOn(evalIndex.jsBackend, "execute").mockResolvedValue(mockResult);
const tool = new EvalTool(makeSession());
await tool.execute("call-mixed", {
cells: [
{ language: "py", code: "x = 1" },
{ language: "js", code: "const y = 2;" },
],
});
await tool.execute("call-py", { language: "py", code: "x = 1" });
await tool.execute("call-js", { language: "js", code: "const y = 2;" });
expect(pythonExecuteSpy).toHaveBeenCalledTimes(1);
expect(jsExecuteSpy).toHaveBeenCalledTimes(1);
@@ -113,7 +111,8 @@ describe("EvalTool language dispatch", () => {
const tool = new EvalTool(makeSession(settings));
await expect(
tool.execute("call-py-disabled", {
cells: [{ language: "py", code: "print('hi')" }],
language: "py",
code: "print('hi')",
}),
).rejects.toThrow(/eval\.py = false/);
});
@@ -124,7 +123,8 @@ describe("EvalTool language dispatch", () => {
const tool = new EvalTool(makeSession(settings));
await expect(
tool.execute("call-js-disabled", {
cells: [{ language: "js", code: "const x = 1;" }],
language: "js",
code: "const x = 1;",
}),
).rejects.toThrow(/eval\.js = false/);
});
@@ -151,7 +151,8 @@ describe("EvalTool language dispatch", () => {
await expect(
tool.execute("call-js-env-disabled", {
cells: [{ language: "js", code: "const x = 1;" }],
language: "js",
code: "const x = 1;",
}),
).rejects.toThrow(/PI_JS=0/);
});
@@ -31,7 +31,9 @@ describe("EvalTool timeout semantics", () => {
// 1s budget; the cell idles for 5s and emits no status, so nothing extends
// the budget — it must be cut off at the wall-clock limit.
const result = await tool.execute("call-compute-timeout", {
cells: [{ language: "js", code: "await Bun.sleep(2000); return 'never';", timeout: 1 }],
language: "js",
code: "await Bun.sleep(2000); return 'never';",
timeout: 1,
});
const text = result.content
+9
View File
@@ -1,6 +1,15 @@
# Changelog
## [Unreleased]
### Added
- Added support for Ruby and Julia code cells in the eval tool
### Changed
- Updated the eval tool view to render the new single-cell eval args (flat `language`/`code`/`title`/`timeout`/`reset`) and to highlight Ruby (`rb`) and Julia (`jl`) cells with their own syntax instead of collapsing them to Python, while still parsing legacy multi-cell `cells` arrays and framed `input` strings from older transcripts.
### Fixed
- Improved compatibility with legacy todo task transcripts
@@ -1,9 +1,10 @@
/**
* `eval` (aliases: js, python, notebook) — code cells executed in the
* persistent kernel. Args arrive either as the modern `cells` array or as a
* legacy framed `input` string (`*** Cell`, `*** Begin LANG`, `===== info =====`);
* both render as discrete highlighted cells. When the result carries typed
* per-cell details, each cell's output is interleaved beneath its code.
* persistent kernel (py/js/rb/jl). Args arrive either as the modern single-cell
* flat shape (`language`/`code`/`title`/`timeout`/`reset`), a legacy `cells`
* array, or a legacy framed `input` string (`*** Cell`, `*** Begin LANG`,
* `===== info =====`); all render as discrete highlighted cells. When the result
* carries typed per-cell details, each cell's output is interleaved beneath its code.
*/
import type { ReactNode } from "react";
import { Badges, CodeBlock, InvalidArg, Note, Output, ResultImages, ResultText } from "../parts";
@@ -18,13 +19,22 @@ interface EvalCell {
code: string;
}
const HLJS_LANG: Record<string, string> = { py: "python", js: "javascript", ts: "typescript" };
const HLJS_LANG: Record<string, string> = {
py: "python",
js: "javascript",
ts: "typescript",
rb: "ruby",
jl: "julia",
};
/** Map an eval language token to its canonical short id, or null when unknown. */
function evalLangAlias(token: string | undefined): string | null {
const t = (token ?? "").toUpperCase();
if (t === "PY" || t === "PYTHON" || t === "IPY" || t === "IPYTHON") return "py";
if (t === "JS" || t === "JAVASCRIPT") return "js";
if (t === "TS" || t === "TYPESCRIPT") return "ts";
if (t === "RB" || t === "RUBY") return "rb";
if (t === "JL" || t === "JULIA") return "jl";
return null;
}
@@ -215,7 +225,7 @@ function parseEvalCellsLegacy(input: string): EvalCell[] {
const info = m[1] ?? "";
let lang = inheritedLang;
let title = "";
const langMatch = info.match(/^(py|js|ts)(?::"([^"]*)")?/);
const langMatch = info.match(/^(py|js|ts|rb|jl)(?::"([^"]*)")?/);
if (langMatch) {
lang = langMatch[1];
if (langMatch[2]) title = langMatch[2];
@@ -257,7 +267,7 @@ function cellsFromArgs(args: Record<string, unknown>, name: string): EvalCell[]
if (timeout !== null) attrs.push(`t=${timeout}s`);
if (item.reset === true) attrs.push("rst");
out.push({
lang: str(item.language) === "js" ? "js" : "py",
lang: evalLangAlias(str(item.language) ?? undefined) ?? "py",
title: str(item.title) ?? "",
attrs,
code: str(item.code) ?? "",
@@ -268,7 +278,14 @@ function cellsFromArgs(args: Record<string, unknown>, name: string): EvalCell[]
const input = str(args.input);
if (input !== null) return parseEvalCells(input).filter(c => c.code !== "" || c.title !== "");
const code = str(args.code);
if (code !== null) return [{ lang: name === "js" ? "js" : "py", title: "", attrs: [], code }];
if (code !== null) {
const attrs: string[] = [];
const timeout = num(args.timeout);
if (timeout !== null) attrs.push(`t=${timeout}s`);
if (args.reset === true) attrs.push("rst");
const lang = evalLangAlias(str(args.language) ?? undefined) ?? (name === "js" ? "js" : "py");
return [{ lang, title: str(args.title) ?? "", attrs, code }];
}
return [];
}
@@ -296,7 +313,7 @@ function detailCellsOf(details: Record<string, unknown> | null): DetailCell[] {
index: num(item.index) ?? i,
title: str(item.title) ?? "",
code: str(item.code) ?? "",
lang: language === "js" ? "js" : language !== null ? "py" : null,
lang: language !== null ? (evalLangAlias(language) ?? "py") : null,
output: str(item.output) ?? "",
status: str(item.status) ?? "",
durationMs: num(item.durationMs),