cf60e6df51
- Added a unified eval framework with parser grammar, backend interfaces, and JS/Python execution result types. - Added eval tool docs and updated prompts for fenced cells, `eval.py`/`eval.js`, and fallback behavior. - Replaced the built-in `python` tool with `eval` across registry, rendering, interactive modes, and tool settings. - Migrated Python execution runtime from `src/ipy` to `src/eval/py`, renamed state fields, and removed legacy introspection. - Refactored browser tooling from in-process VM helpers to worker-managed tab supervisors and protocol transport. - Added eval parser fallback and JS tool-bridge tests, updated imports, and removed obsolete python-mode suites.
78 lines
2.6 KiB
TypeScript
78 lines
2.6 KiB
TypeScript
import { afterEach, describe, expect, it, vi } from "bun:test";
|
|
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
|
import * as evalIndex from "@oh-my-pi/pi-coding-agent/eval";
|
|
import * as pyKernel from "@oh-my-pi/pi-coding-agent/eval/py/kernel";
|
|
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
|
|
import { EvalTool } from "@oh-my-pi/pi-coding-agent/tools/eval";
|
|
|
|
function makeSession(): ToolSession {
|
|
return {
|
|
cwd: "/tmp/eval-test",
|
|
hasUI: false,
|
|
getSessionFile: () => null,
|
|
getSessionSpawns: () => null,
|
|
settings: Settings.isolated(),
|
|
};
|
|
}
|
|
|
|
const mockResult = {
|
|
output: "ok",
|
|
exitCode: 0,
|
|
cancelled: false,
|
|
truncated: false,
|
|
artifactId: undefined,
|
|
totalLines: 1,
|
|
totalBytes: 2,
|
|
outputLines: 1,
|
|
outputBytes: 2,
|
|
displayOutputs: [],
|
|
};
|
|
|
|
describe("EvalTool language resolution", () => {
|
|
afterEach(() => {
|
|
vi.restoreAllMocks();
|
|
});
|
|
|
|
it("dispatches to js when fenced code declares ```js", async () => {
|
|
vi.spyOn(pyKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true });
|
|
const jsExecuteSpy = vi.spyOn(evalIndex.jsBackend, "execute").mockResolvedValue(mockResult);
|
|
const pythonExecuteSpy = vi.spyOn(evalIndex.pythonBackend, "execute");
|
|
|
|
const tool = new EvalTool(makeSession());
|
|
await tool.execute("call-1", {
|
|
input: "=== CELL one ===\n```js\nconst x = 1;\n```\n",
|
|
});
|
|
|
|
expect(jsExecuteSpy).toHaveBeenCalledTimes(1);
|
|
expect(pythonExecuteSpy).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it("dispatches to python when fenced code declares ```python", async () => {
|
|
const pythonExecuteSpy = vi.spyOn(evalIndex.pythonBackend, "execute").mockResolvedValue(mockResult);
|
|
vi.spyOn(evalIndex.pythonBackend, "isAvailable").mockResolvedValue(true);
|
|
const jsExecuteSpy = vi.spyOn(evalIndex.jsBackend, "execute");
|
|
|
|
const tool = new EvalTool(makeSession());
|
|
await tool.execute("call-2", {
|
|
input: "=== CELL one ===\n```python\nprint('hi')\n```\n",
|
|
});
|
|
|
|
expect(pythonExecuteSpy).toHaveBeenCalledTimes(1);
|
|
expect(jsExecuteSpy).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it("auto-detects python via syntactic markers when fence is bare", async () => {
|
|
const pythonExecuteSpy = vi.spyOn(evalIndex.pythonBackend, "execute").mockResolvedValue(mockResult);
|
|
vi.spyOn(evalIndex.pythonBackend, "isAvailable").mockResolvedValue(true);
|
|
const jsExecuteSpy = vi.spyOn(evalIndex.jsBackend, "execute");
|
|
|
|
const tool = new EvalTool(makeSession());
|
|
await tool.execute("call-3", {
|
|
input: "=== CELL one ===\ndef greet():\n print('hi')\ngreet()\n",
|
|
});
|
|
|
|
expect(pythonExecuteSpy).toHaveBeenCalledTimes(1);
|
|
expect(jsExecuteSpy).not.toHaveBeenCalled();
|
|
});
|
|
});
|