feat(coding-agent): refactored eval tool to single-step execution

- Transitioned the eval tool from batch multi-cell execution to a single-step input structure with flat parameters.
- Updated core agent logic, UI components, and documentation to support state persistence across incremental eval calls.
- Restricted bash tool capabilities by requiring explicit use of `read` or `find` instead of `ls` or `find`.
- Added support for Ruby and Julia language runtimes to the eval tool and associated web renderers.
This commit is contained in:
can1357
2026-06-23 00:55:07 +02:00
parent 899c0ef08b
commit 060f4004e7
23 changed files with 248 additions and 222 deletions
@@ -426,7 +426,7 @@ describe("AgentSession python cleanup", () => {
expect(EvalTool).toBeDefined();
let toolExecutionSettled = false;
const toolExecution = EvalTool!
.execute("call-id", { cells: [{ language: "py", code: "print('tool')" }] }, undefined, undefined, undefined)
.execute("call-id", { language: "py", code: "print('tool')" }, undefined, undefined, undefined)
.finally(() => {
toolExecutionSettled = true;
});
@@ -652,13 +652,7 @@ describe("AgentSession python cleanup", () => {
expect(EvalTool).toBeDefined();
const disposeSession = session.dispose();
await expect(
EvalTool!.execute(
"call-id",
{ cells: [{ language: "py", code: "print('late')" }] },
undefined,
undefined,
undefined,
),
EvalTool!.execute("call-id", { language: "py", code: "print('late')" }, undefined, undefined, undefined),
).rejects.toThrow("Python execution is unavailable while session disposal is in progress");
await disposeSession;
expect(executeSpy).not.toHaveBeenCalled();
@@ -693,7 +687,7 @@ describe("AgentSession python cleanup", () => {
expect(EvalTool).toBeDefined();
const execution = EvalTool!.execute(
"call-id",
{ cells: [{ language: "py", code: "print('late after artifact')" }] },
{ language: "py", code: "print('late after artifact')" },
undefined,
undefined,
undefined,