From f74c6d892af4bc30071986bfb94690b441f5ee14 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 8 Jun 2026 12:22:47 +0200 Subject: [PATCH] feat(coding-agent-eval): renamed the eval helper API from llm() to completion() - Renamed eval oneshot helper from llm() to completion() across JS/Python APIs. - Remapped eval bridge internals to completion semantics (__completion__, runEvalCompletion, completion status op). - Updated docs, prompts, and timeout guidance to describe completion() usage and behavior. - Adjusted completion defaults for active-session model preference, fallback parsing, and slow-tier effort handling. --- docs/python-repl.md | 4 +- docs/tools/eval.md | 12 +- packages/coding-agent/CHANGELOG.md | 11 +- ...idge.test.ts => completion-bridge.test.ts} | 114 +++++++++--------- .../coding-agent/src/eval/bridge-timeout.ts | 2 +- .../{llm-bridge.ts => completion-bridge.ts} | 57 ++++----- .../coding-agent/src/eval/idle-timeout.ts | 2 +- .../src/eval/js/shared/prelude.txt | 8 +- .../coding-agent/src/eval/js/tool-bridge.ts | 6 +- packages/coding-agent/src/eval/py/prelude.py | 6 +- .../src/modes/components/tips.txt | 2 +- .../src/prompts/system/workflow-notice.md | 2 +- .../coding-agent/src/prompts/tools/eval.md | 6 +- .../coding-agent/src/tools/eval-render.ts | 4 +- packages/coding-agent/src/tools/eval.ts | 2 +- .../test/tools/eval-timeout.test.ts | 4 +- 16 files changed, 129 insertions(+), 113 deletions(-) rename packages/coding-agent/src/eval/__tests__/{llm-bridge.test.ts => completion-bridge.test.ts} (76%) rename packages/coding-agent/src/eval/{llm-bridge.ts => completion-bridge.ts} (73%) diff --git a/docs/python-repl.md b/docs/python-repl.md index 11a0ad631..40f246a8c 100644 --- a/docs/python-repl.md +++ b/docs/python-repl.md @@ -166,9 +166,9 @@ Python prelude helpers include `agent(prompt, *, agent_type="task", model=None, ### Cell timeout -Each eval cell `timeout` is in seconds, defaults to 30, and is clamped to `1..600`. It is a **wall-clock budget on the cell's own work** that the watchdog (`IdleTimeout`, `src/eval/idle-timeout.ts`) enforces, **but it is paused while a host-side `agent()`/`parallel()`/`llm()` bridge call is in flight**: those calls pump a heartbeat (`withBridgeHeartbeat`, `src/eval/heartbeat.ts`) that re-arms the watchdog, so a long fanout or a slow completion runs to completion instead of being killed mid-stream. +Each eval cell `timeout` is in seconds, defaults to 30, and is clamped to `1..600`. It is a **wall-clock budget on the cell's own work** that the watchdog (`IdleTimeout`, `src/eval/idle-timeout.ts`) enforces, **but it is paused while a host-side `agent()`/`parallel()`/`completion()` bridge call is in flight**: those calls pump a heartbeat (`withBridgeHeartbeat`, `src/eval/heartbeat.ts`) that re-arms the watchdog, so a long fanout or a slow completion runs to completion instead of being killed mid-stream. -The heartbeat is the **sole** signal that extends the budget. Everything else the cell does — compute, `stdout`/`stderr`, `log()`/`phase()`, and ordinary (non-agent) tool calls — counts against `timeout`, so a cell that is not delegating to an agent/llm is bounded by a plain wall-clock timeout. The tool combines the caller abort signal, the session abort signal, and the watchdog's signal with `AbortSignal.any(...)`; no wall-clock deadline is passed to the backend, so neither runtime arms a competing fixed timer. +The heartbeat is the **sole** signal that extends the budget. Everything else the cell does — compute, `stdout`/`stderr`, `log()`/`phase()`, and ordinary (non-agent) tool calls — counts against `timeout`, so a cell that is not delegating to an agent/completion is bounded by a plain wall-clock timeout. The tool combines the caller abort signal, the session abort signal, and the watchdog's signal with `AbortSignal.any(...)`; no wall-clock deadline is passed to the backend, so neither runtime arms a competing fixed timer. ### Kernel execution cancellation diff --git a/docs/tools/eval.md b/docs/tools/eval.md index 835730249..399d4fbb7 100644 --- a/docs/tools/eval.md +++ b/docs/tools/eval.md @@ -131,7 +131,7 @@ Implemented in `packages/coding-agent/src/eval/js/worker-core.ts`, `packages/cod - `display`, `print` - `read`, `write`, `append`, `sort`, `uniq`, `counter`, `diff`, `tree`, `env`, `output` - `tool.(args)` proxy for arbitrary session tool calls - - `llm(prompt, opts?)` for oneshot, stateless LLM calls (see _Oneshot LLM helper_ below) + - `completion(prompt, opts?)` for oneshot, stateless model calls (see _Oneshot completion helper_ below) - `agent(prompt, opts?)` for a single subagent call, plus `parallel()` / `pipeline()` bounded-pool helpers (see _Subagent helper_ below) - JS helpers that touch the host/runtime boundary are async and `await`able; pure text helpers (`sort`, `uniq`, `counter`) return synchronously but may still be safely awaited. - JS helper signatures use a trailing options object rather than Python keyword arguments: @@ -161,7 +161,7 @@ Implemented in `packages/coding-agent/src/eval/py/executor.ts`, `packages/coding - initialize cwd / env / `sys.path` - execute `PYTHON_PRELUDE` - Python cells run in the runner's persistent asyncio event loop, so top-level `await` works; the prompt warns not to use `asyncio.run(...)` -- The Python prelude defines helpers with the same surface as JS where practical, including `tool.(args)`, `llm(...)`, and `agent(...)` through a per-run loopback bridge +- The Python prelude defines helpers with the same surface as JS where practical, including `tool.(args)`, `completion(...)`, and `agent(...)` through a per-run loopback bridge - Synchronous statement blocks run in the default executor with ContextVar state copied in; the GIL still serializes bytecode execution, but awaited regions can interleave with sibling cells - Kernel `display_data` / `execute_result` messages map to: - `application/x-omp-status` → status event @@ -172,13 +172,13 @@ Implemented in `packages/coding-agent/src/eval/py/executor.ts`, `packages/coding - `text/html` → HTML converted to markdown with `htmlToBasicMarkdown()` - Interactive stdin is rejected: `input_request` sends an empty reply, marks `stdinRequested`, and the executor returns exit code `1` -### Oneshot LLM helper (`llm`) +### Oneshot completion helper (`completion`) -Both runtimes expose `llm()` — a single stateless completion against a model tier. It is intentionally minimal: no conversation history, no agent-visible tools, pure text in / text (or object) out. Implemented host-side in `packages/coding-agent/src/eval/llm-bridge.ts` and routed through the existing tool bridge under the reserved name `__llm__`. +Both runtimes expose `completion()` — a single stateless completion against a model tier. It is intentionally minimal: no conversation history, no agent-visible tools, pure text in / text (or object) out. Implemented host-side in `packages/coding-agent/src/eval/completion-bridge.ts` and routed through the existing tool bridge under the reserved name `__completion__`. - Signatures: - - JS: `await llm(prompt, { model?, system?, schema? })` - - Python: `llm(prompt, *, model="default", system=None, schema=None)` + - JS: `await completion(prompt, { model?, system?, schema? })` + - Python: `completion(prompt, *, model="default", system=None, schema=None)` - `model` selects a tier (default `"default"`): - `"smol"` → `pi/smol` role (fast / cheap) - `"default"` → the session's active model, falling back to the `pi/default` role diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index acad171b2..3fc69a0ca 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,7 +1,6 @@ # Changelog ## [Unreleased] - ### Added - macOS release binaries are now signed with a Developer ID Application identity (hardened runtime + secure timestamp + JIT/library-validation entitlements) and notarized in CI when the `APPLE_*` signing secrets are configured; releases auto-fall back to ad-hoc signing until then. This makes the shipped binaries Gatekeeper-acceptable, unblocking an official Homebrew submission ([#776](https://github.com/can1357/oh-my-pi/issues/776)). See `docs/macos-signing-notarization.md`. @@ -9,7 +8,15 @@ ### Changed +- Adjusted `completion()` model resolution so the `default` tier now prefers the session’s active model and falls back to the configured default role when needed - Rewrote the session auto-title prompt (`prompts/system/title-system.md`) and the `set_title` tool description to ask for a concise, sentence-case title (3-7 words) that captures the session's topic/goal, with good/bad examples and explicit guidance to treat the first message as data (no following embedded links/instructions, no refusals, describe URL/reference asks). The local on-device title prompt (`tiny-title-system.md`) was aligned to the same 3-7 word, sentence-case convention. The deterministic greeting/low-signal filter and the `none` deferral sentinel are unchanged. +- Renamed the eval oneshot helper from `llm()` to `completion()` in both JavaScript and Python preludes, including status events, prompt docs, and runtime tests. + +### Fixed + +- Fixed `completion()` to always send a non-empty default system prompt when `system` is omitted so providers that require instructions no longer reject requests +- Fixed structured `completion()` mode to return parsed JSON from plain text output when the model skips the forced `respond` tool call +- Fixed slow-tier `completion()` reasoning requests to avoid unsupported effort settings by only enabling reasoning on reasoning-capable models and capping effort to supported levels ## [15.10.3] - 2026-06-08 @@ -9692,4 +9699,4 @@ Initial public release. - Git branch display in footer - Message queueing during streaming responses - OAuth integration for Gmail and Google Calendar access -- HTML export with syntax highlighting and collapsible sections +- HTML export with syntax highlighting and collapsible sections \ No newline at end of file diff --git a/packages/coding-agent/src/eval/__tests__/llm-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/completion-bridge.test.ts similarity index 76% rename from packages/coding-agent/src/eval/__tests__/llm-bridge.test.ts rename to packages/coding-agent/src/eval/__tests__/completion-bridge.test.ts index 2ce98a02d..89b5ff7d2 100644 --- a/packages/coding-agent/src/eval/__tests__/llm-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/completion-bridge.test.ts @@ -10,10 +10,10 @@ import { Settings } from "../../config/settings"; import type { ToolSession } from "../../tools"; import { ToolError } from "../../tools/tool-errors"; import { EVAL_TIMEOUT_PAUSE_OP, EVAL_TIMEOUT_RESUME_OP } from "../bridge-timeout"; +import { runEvalCompletion } from "../completion-bridge"; import { IdleTimeout } from "../idle-timeout"; import { disposeAllVmContexts } from "../js/context-manager"; import { executeJs } from "../js/executor"; -import { runEvalLlm } from "../llm-bridge"; import { disposeAllKernelSessions, type PythonResult } from "../py/executor"; function makeModel(provider: string, id: string, extra: Partial> = {}): Model { @@ -98,16 +98,19 @@ function assistant(opts: { }; } -async function runPythonLlmInSubprocess(options: { structured: boolean; tempDir: TempDir }): Promise { +async function runPythonCompletionInSubprocess(options: { + structured: boolean; + tempDir: TempDir; +}): Promise { const repoRoot = path.resolve(import.meta.dir, "../../../.."); - const scriptPath = path.join(options.tempDir.path(), "run-python-llm.ts"); - const resultPath = path.join(options.tempDir.path(), "python-llm-result.json"); + const scriptPath = path.join(options.tempDir.path(), "run-python-completion.ts"); + const resultPath = path.join(options.tempDir.path(), "python-completion-result.json"); const aiPath = path.resolve(import.meta.dir, "../../../../ai/src/index.ts"); const executorPath = path.resolve(import.meta.dir, "../py/executor.ts"); const settingsPath = path.resolve(import.meta.dir, "../../config/settings.ts"); const code = options.structured - ? 'import json\nprint(json.dumps(llm("hi", schema={"type": "object"})))' - : 'print(llm("hi", model="smol"))'; + ? 'import json\nprint(json.dumps(completion("hi", schema={"type": "object"})))' + : 'print(completion("hi", model="smol"))'; const responseContent = options.structured ? '[{ type: "toolCall", id: "tc-1", name: "respond", arguments: { ok: true } }]' : '[{ type: "text", text: "hello from python" }]'; @@ -153,7 +156,7 @@ vi.spyOn(ai, "completeSimple").mockResolvedValue({ }); const result = await executePython(${JSON.stringify(code)}, { cwd: ${JSON.stringify(options.tempDir.path())}, - sessionId: ${JSON.stringify(`py-llm:${options.structured ? "struct" : "plain"}`)}, + sessionId: ${JSON.stringify(`py-completion:${options.structured ? "struct" : "plain"}`)}, sessionFile: ${JSON.stringify(path.join(options.tempDir.path(), "session.jsonl"))}, toolSession: session, kernelMode: "per-call", @@ -165,11 +168,12 @@ process.exit(0); const child = await $`bun ${scriptPath}`.cwd(repoRoot).quiet().nothrow(); const stdout = child.stdout.toString(); const stderr = child.stderr.toString(); - if (child.exitCode !== 0) throw new Error(stderr || stdout || `Python llm subprocess exited with ${child.exitCode}`); + if (child.exitCode !== 0) + throw new Error(stderr || stdout || `Python completion subprocess exited with ${child.exitCode}`); return (await Bun.file(resultPath).json()) as PythonResult; } -describe("runEvalLlm", () => { +describe("runEvalCompletion", () => { afterEach(() => { vi.restoreAllMocks(); }); @@ -178,9 +182,9 @@ describe("runEvalLlm", () => { const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); const session = makeSession(); - await runEvalLlm({ prompt: "q", model: "smol" }, { session }); - await runEvalLlm({ prompt: "q", model: "default" }, { session }); - await runEvalLlm({ prompt: "q", model: "slow" }, { session }); + await runEvalCompletion({ prompt: "q", model: "smol" }, { session }); + await runEvalCompletion({ prompt: "q", model: "default" }, { session }); + await runEvalCompletion({ prompt: "q", model: "slow" }, { session }); const resolved = spy.mock.calls.map(call => { const model = call[0] as Model; @@ -193,7 +197,7 @@ describe("runEvalLlm", () => { const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); const session = makeSession({ available: [SMOL, DEFAULT, SLOW], activeModel: "p/slow" }); - await runEvalLlm({ prompt: "q", model: "default" }, { session }); + await runEvalCompletion({ prompt: "q", model: "default" }, { session }); const model = spy.mock.calls[0]?.[0] as Model; expect(`${model.provider}/${model.id}`).toBe("p/slow"); @@ -201,7 +205,7 @@ describe("runEvalLlm", () => { it("returns the completion text in plain mode", async () => { vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "the answer" })); - const result = await runEvalLlm({ prompt: "q", model: "smol" }, { session: makeSession() }); + const result = await runEvalCompletion({ prompt: "q", model: "smol" }, { session: makeSession() }); expect(result.text).toBe("the answer"); expect(result.details).toEqual({ model: "p/smol", tier: "smol", structured: false }); }); @@ -209,10 +213,10 @@ describe("runEvalLlm", () => { it("supplies a non-empty systemPrompt when system is omitted (codex 'Instructions are required' guard)", async () => { // The openai-codex Responses transformer drops `instructions` when no // system prompt is provided, and the remote endpoint then 400s with - // "Instructions are required". runEvalLlm must always carry a non-empty - // systemPrompt so `llm("…")` without a `system` argument works. + // "Instructions are required". runEvalCompletion must always carry a non-empty + // systemPrompt so `completion("…")` without a `system` argument works. const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); - await runEvalLlm({ prompt: "q", model: "smol" }, { session: makeSession() }); + await runEvalCompletion({ prompt: "q", model: "smol" }, { session: makeSession() }); const ctx = spy.mock.calls[0]?.[1] as { systemPrompt?: string[] }; expect(ctx.systemPrompt).toBeDefined(); expect(ctx.systemPrompt?.length).toBeGreaterThan(0); @@ -221,7 +225,7 @@ describe("runEvalLlm", () => { it("honors an explicit system prompt instead of overriding it", async () => { const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); - await runEvalLlm({ prompt: "q", model: "smol", system: "Be terse." }, { session: makeSession() }); + await runEvalCompletion({ prompt: "q", model: "smol", system: "Be terse." }, { session: makeSession() }); const ctx = spy.mock.calls[0]?.[1] as { systemPrompt?: string[] }; expect(ctx.systemPrompt).toEqual(["Be terse."]); }); @@ -230,7 +234,7 @@ describe("runEvalLlm", () => { const spy = vi .spyOn(ai, "completeSimple") .mockResolvedValue(assistant({ toolCall: { name: "respond", arguments: { answer: 42 } } })); - const result = await runEvalLlm( + const result = await runEvalCompletion( { prompt: "q", model: "smol", schema: { type: "object", properties: { answer: { type: "number" } } } }, { session: makeSession() }, ); @@ -246,7 +250,7 @@ describe("runEvalLlm", () => { it("falls back to JSON embedded in text when the model skips the respond tool", async () => { vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: 'here: {"answer": 7}' })); - const result = await runEvalLlm( + const result = await runEvalCompletion( { prompt: "q", model: "smol", schema: { type: "object" } }, { session: makeSession() }, ); @@ -257,8 +261,8 @@ describe("runEvalLlm", () => { const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); const session = makeSession({ available: [SMOL, DEFAULT, REASONING_SLOW] }); - await runEvalLlm({ prompt: "q", model: "smol" }, { session }); - await runEvalLlm({ prompt: "q", model: "slow" }, { session }); + await runEvalCompletion({ prompt: "q", model: "smol" }, { session }); + await runEvalCompletion({ prompt: "q", model: "slow" }, { session }); const smolOpts = spy.mock.calls[0]?.[2] as { reasoning?: unknown }; const slowOpts = spy.mock.calls[1]?.[2] as { reasoning?: unknown }; @@ -269,47 +273,49 @@ describe("runEvalLlm", () => { it("does not request reasoning for the slow tier on a non-reasoning model", async () => { const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); // SLOW is reasoning:false — must not trip requireSupportedEffort downstream. - const result = await runEvalLlm({ prompt: "q", model: "slow" }, { session: makeSession() }); + const result = await runEvalCompletion({ prompt: "q", model: "slow" }, { session: makeSession() }); expect(result.text).toBe("ok"); const opts = spy.mock.calls[0]?.[2] as { reasoning?: unknown }; expect(opts.reasoning).toBeUndefined(); }); it("throws ToolError on invalid arguments", async () => { - await expect(runEvalLlm({ prompt: "" }, { session: makeSession() })).rejects.toBeInstanceOf(ToolError); - await expect(runEvalLlm({ prompt: "q", model: "huge" }, { session: makeSession() })).rejects.toBeInstanceOf( - ToolError, - ); + await expect(runEvalCompletion({ prompt: "" }, { session: makeSession() })).rejects.toBeInstanceOf(ToolError); + await expect( + runEvalCompletion({ prompt: "q", model: "huge" }, { session: makeSession() }), + ).rejects.toBeInstanceOf(ToolError); }); it("throws ToolError when no model resolves for the tier", async () => { const session = makeSession({ available: [DEFAULT], roles: { smol: "missing/model" } }); - await expect(runEvalLlm({ prompt: "q", model: "smol" }, { session })).rejects.toBeInstanceOf(ToolError); + await expect(runEvalCompletion({ prompt: "q", model: "smol" }, { session })).rejects.toBeInstanceOf(ToolError); }); it("throws ToolError when the resolved model has no API key", async () => { const session = makeSession({ apiKey: null }); - await expect(runEvalLlm({ prompt: "q", model: "smol" }, { session })).rejects.toBeInstanceOf(ToolError); + await expect(runEvalCompletion({ prompt: "q", model: "smol" }, { session })).rejects.toBeInstanceOf(ToolError); }); it("maps error and aborted stop reasons to ToolError", async () => { vi.spyOn(ai, "completeSimple").mockResolvedValueOnce(assistant({ stopReason: "error", errorMessage: "boom" })); - await expect(runEvalLlm({ prompt: "q", model: "smol" }, { session: makeSession() })).rejects.toThrow("boom"); + await expect(runEvalCompletion({ prompt: "q", model: "smol" }, { session: makeSession() })).rejects.toThrow( + "boom", + ); vi.spyOn(ai, "completeSimple").mockResolvedValueOnce(assistant({ stopReason: "aborted" })); - await expect(runEvalLlm({ prompt: "q", model: "smol" }, { session: makeSession() })).rejects.toBeInstanceOf( - ToolError, - ); + await expect( + runEvalCompletion({ prompt: "q", model: "smol" }, { session: makeSession() }), + ).rejects.toBeInstanceOf(ToolError); }); it("throws ToolError when plain mode produces no text", async () => { vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "" })); - await expect(runEvalLlm({ prompt: "q", model: "smol" }, { session: makeSession() })).rejects.toBeInstanceOf( - ToolError, - ); + await expect( + runEvalCompletion({ prompt: "q", model: "smol" }, { session: makeSession() }), + ).rejects.toBeInstanceOf(ToolError); }); - it("pauses the idle watchdog while a slow llm() request is in flight", async () => { + it("pauses the idle watchdog while a slow completion() request is in flight", async () => { // A oneshot completion emits no status until it returns; delegated model // time must be invisible to the eval timeout budget. vi.spyOn(ai, "completeSimple").mockImplementation(async () => { @@ -319,7 +325,7 @@ describe("runEvalLlm", () => { const ops: string[] = []; using idle = new IdleTimeout(60); - const result = await runEvalLlm( + const result = await runEvalCompletion( { prompt: "q", model: "smol" }, { session: makeSession(), @@ -333,12 +339,12 @@ describe("runEvalLlm", () => { ); expect(result.text).toBe("the answer"); - expect(ops).toEqual([EVAL_TIMEOUT_PAUSE_OP, EVAL_TIMEOUT_RESUME_OP, "llm"]); + expect(ops).toEqual([EVAL_TIMEOUT_PAUSE_OP, EVAL_TIMEOUT_RESUME_OP, "completion"]); expect(idle.signal.aborted).toBe(false); }); }); -describe("llm() through eval runtimes", () => { +describe("completion() through eval runtimes", () => { afterEach(() => { vi.restoreAllMocks(); }); @@ -348,13 +354,13 @@ describe("llm() through eval runtimes", () => { await disposeAllKernelSessions(); }); - it("exposes llm() in the JavaScript runtime", async () => { - using tempDir = TempDir.createSync("@omp-eval-llm-js-"); + it("exposes completion() in the JavaScript runtime", async () => { + using tempDir = TempDir.createSync("@omp-eval-completion-js-"); const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `js-llm:${crypto.randomUUID()}`; + const sessionId = `js-completion:${crypto.randomUUID()}`; vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "hello from smol" })); - const result = await executeJs('return await llm("hi", { model: "smol" });', { + const result = await executeJs('return await completion("hi", { model: "smol" });', { cwd: tempDir.path(), sessionId, session: makeSession(), @@ -365,16 +371,16 @@ describe("llm() through eval runtimes", () => { expect(result.output.trim()).toBe("hello from smol"); }); - it("parses structured llm() output in the JavaScript runtime", async () => { - using tempDir = TempDir.createSync("@omp-eval-llm-js-struct-"); + it("parses structured completion() output in the JavaScript runtime", async () => { + using tempDir = TempDir.createSync("@omp-eval-completion-js-struct-"); const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `js-llm-struct:${crypto.randomUUID()}`; + const sessionId = `js-completion-struct:${crypto.randomUUID()}`; vi.spyOn(ai, "completeSimple").mockResolvedValue( assistant({ toolCall: { name: "respond", arguments: { ok: true, n: 3 } } }), ); const result = await executeJs( - 'const r = await llm("hi", { schema: { type: "object" } }); return JSON.stringify(r);', + 'const r = await completion("hi", { schema: { type: "object" } }); return JSON.stringify(r);', { cwd: tempDir.path(), sessionId, session: makeSession(), sessionFile }, ); @@ -382,10 +388,10 @@ describe("llm() through eval runtimes", () => { expect(JSON.parse(result.output.trim())).toEqual({ ok: true, n: 3 }); }); - it("exposes llm() in the Python runtime", async () => { - const tempDir = TempDir.createSync("@omp-eval-llm-py-"); + it("exposes completion() in the Python runtime", async () => { + const tempDir = TempDir.createSync("@omp-eval-completion-py-"); try { - const result = await runPythonLlmInSubprocess({ structured: false, tempDir }); + const result = await runPythonCompletionInSubprocess({ structured: false, tempDir }); expect(result.exitCode).toBe(0); expect(result.output.trim()).toBe("hello from python"); } finally { @@ -393,10 +399,10 @@ describe("llm() through eval runtimes", () => { } }); - it("parses structured llm() output in the Python runtime", async () => { - const tempDir = TempDir.createSync("@omp-eval-llm-py-struct-"); + it("parses structured completion() output in the Python runtime", async () => { + const tempDir = TempDir.createSync("@omp-eval-completion-py-struct-"); try { - const result = await runPythonLlmInSubprocess({ structured: true, tempDir }); + const result = await runPythonCompletionInSubprocess({ structured: true, tempDir }); expect(result.exitCode).toBe(0); expect(JSON.parse(result.output.trim())).toEqual({ ok: true }); } finally { diff --git a/packages/coding-agent/src/eval/bridge-timeout.ts b/packages/coding-agent/src/eval/bridge-timeout.ts index bef0798cc..90907b0e1 100644 --- a/packages/coding-agent/src/eval/bridge-timeout.ts +++ b/packages/coding-agent/src/eval/bridge-timeout.ts @@ -2,7 +2,7 @@ * Timeout suspension for in-flight host-side eval bridge calls. * * The eval watchdog caps a cell's `timeout` as a budget on the cell runtime's - * own work. Host-side `agent()` / `parallel()` / `llm()` bridge calls hand + * own work. Host-side `agent()` / `parallel()` / `completion()` bridge calls hand * control to the outer TypeScript process, where the Python kernel or JS VM is * only waiting for a result. While that delegated work is in flight, the cell * timeout must be ignored completely; once the bridge returns and the runtime is diff --git a/packages/coding-agent/src/eval/llm-bridge.ts b/packages/coding-agent/src/eval/completion-bridge.ts similarity index 73% rename from packages/coding-agent/src/eval/llm-bridge.ts rename to packages/coding-agent/src/eval/completion-bridge.ts index ccba720b2..848ca8504 100644 --- a/packages/coding-agent/src/eval/llm-bridge.ts +++ b/packages/coding-agent/src/eval/completion-bridge.ts @@ -1,11 +1,11 @@ /** - * Host-side handler for the eval `llm()` helper. + * Host-side handler for the eval `completion()` helper. * * Both eval runtimes (JS worker + Python kernel) route helper→host calls * through {@link callSessionTool}. Reserving the synthetic tool name - * {@link EVAL_LLM_BRIDGE_NAME} lets a single host handler serve both + * {@link EVAL_COMPLETION_BRIDGE_NAME} lets a single host handler serve both * transports without registering an agent-visible tool: cell code calls - * `llm(prompt, opts)`, the prelude forwards `{ prompt, model, system?, schema? }` + * `completion(prompt, opts)`, the prelude forwards `{ prompt, model, system?, schema? }` * through the bridge, and this module performs one stateless completion. * * The call is oneshot and toolless from the model's perspective — pure text @@ -27,36 +27,36 @@ import { ToolError } from "../tools/tool-errors"; import { withBridgeTimeoutPause } from "./bridge-timeout"; import type { JsStatusEvent } from "./js/shared/types"; -/** Synthetic bridge name reserved for the `llm()` helper across both runtimes. */ -export const EVAL_LLM_BRIDGE_NAME = "__llm__"; +/** Synthetic bridge name reserved for the `completion()` helper across both runtimes. */ +export const EVAL_COMPLETION_BRIDGE_NAME = "__completion__"; /** Synthetic tool the model is forced to call when a `schema` is supplied. */ const STRUCTURED_TOOL_NAME = "respond"; -type LlmTier = "smol" | "default" | "slow"; +type CompletionTier = "smol" | "default" | "slow"; -const TIER_TO_PATTERN: Record = { +const TIER_TO_PATTERN: Record = { smol: "pi/smol", default: "pi/default", slow: "pi/slow", }; -const llmArgsSchema = z.object({ +const completionArgsSchema = z.object({ prompt: z.string().min(1, "prompt must be a non-empty string"), model: z.enum(["smol", "default", "slow"]).default("default"), system: z.string().optional(), schema: z.record(z.string(), z.unknown()).optional(), }); -export interface EvalLlmBridgeOptions { +export interface EvalCompletionBridgeOptions { session: ToolSession; signal?: AbortSignal; emitStatus?: (event: JsStatusEvent) => void; } -export interface EvalLlmResult { +export interface EvalCompletionResult { text: string; - details: { model: string; tier: LlmTier; structured: boolean }; + details: { model: string; tier: CompletionTier; structured: boolean }; } /** @@ -64,7 +64,7 @@ export interface EvalLlmResult { * active model and falls back to the `pi/default` role; `smol`/`slow` resolve * their respective role patterns. Returns `undefined` when nothing matches. */ -function resolveTierModel(tier: LlmTier, session: ToolSession): Model | undefined { +function resolveTierModel(tier: CompletionTier, session: ToolSession): Model | undefined { const modelRegistry = session.modelRegistry; if (!modelRegistry) return undefined; const available = modelRegistry.getAvailable(); @@ -90,7 +90,7 @@ function resolveTierModel(tier: LlmTier, session: ToolSession): Model | und * throwing downstream on models that cannot reason. Clamps to the highest * supported effort so a reasoning model without `high` does not 400. */ -function reasoningForTier(tier: LlmTier, model: Model): Effort | undefined { +function reasoningForTier(tier: CompletionTier, model: Model): Effort | undefined { if (tier !== "slow" || !model.reasoning) return undefined; const efforts = getSupportedEfforts(model); if (efforts.length === 0) return undefined; @@ -98,23 +98,26 @@ function reasoningForTier(tier: LlmTier, model: Model): Effort | undefined } /** - * Run a single stateless completion on behalf of an eval cell's `llm()` call. + * Run a single stateless completion on behalf of an eval cell's `completion()` call. * Returns a `{ text, details }` value shaped like a {@link callSessionTool} * result so the existing bridge transport carries it to either runtime. */ -export async function runEvalLlm(args: unknown, options: EvalLlmBridgeOptions): Promise { - const parsed = llmArgsSchema.safeParse(args); +export async function runEvalCompletion( + args: unknown, + options: EvalCompletionBridgeOptions, +): Promise { + const parsed = completionArgsSchema.safeParse(args); if (!parsed.success) { const issue = parsed.error.issues[0]; const where = issue?.path.length ? `${issue.path.join(".")}: ` : ""; - throw new ToolError(`llm() received invalid arguments: ${where}${issue?.message ?? "bad input"}`); + throw new ToolError(`completion() received invalid arguments: ${where}${issue?.message ?? "bad input"}`); } const { prompt, model: tier, system, schema } = parsed.data; const model = resolveTierModel(tier, options.session); if (!model) { throw new ToolError( - `llm() could not resolve a model for the "${tier}" tier. Configure modelRoles.${tier === "default" ? "default" : tier} or ensure a provider is available.`, + `completion() could not resolve a model for the "${tier}" tier. Configure modelRoles.${tier === "default" ? "default" : tier} or ensure a provider is available.`, ); } @@ -122,7 +125,7 @@ export async function runEvalLlm(args: unknown, options: EvalLlmBridgeOptions): const apiKey = await registry?.getApiKey(model); if (!registry || !apiKey) { throw new ToolError( - `llm() has no API key for ${formatModelString(model)}. Configure credentials for this provider or choose another tier.`, + `completion() has no API key for ${formatModelString(model)}. Configure credentials for this provider or choose another tier.`, ); } @@ -141,7 +144,7 @@ export async function runEvalLlm(args: unknown, options: EvalLlmBridgeOptions): // Some providers (notably openai-codex) require a non-empty `instructions` // field on every Responses request and 400 with "Instructions are required" - // when it is missing. Fall back to a minimal default so `llm(prompt)` works + // when it is missing. Fall back to a minimal default so `completion(prompt)` works // without forcing every caller to pass a `system` prompt. const systemPrompt = system ? [system] : ["You are a helpful assistant."]; @@ -164,15 +167,15 @@ export async function runEvalLlm(args: unknown, options: EvalLlmBridgeOptions): reasoning: reasoningForTier(tier, model), toolChoice: schema ? { type: "tool", name: STRUCTURED_TOOL_NAME } : undefined, }, - { telemetry, oneshotKind: "eval_llm" }, + { telemetry, oneshotKind: "eval_completion" }, ), ); if (response.stopReason === "error") { - throw new ToolError(response.errorMessage ?? "llm() request failed."); + throw new ToolError(response.errorMessage ?? "completion() request failed."); } if (response.stopReason === "aborted") { - throw new ToolError("llm() request aborted."); + throw new ToolError("completion() request aborted."); } let resultText: string; @@ -183,20 +186,20 @@ export async function runEvalLlm(args: unknown, options: EvalLlmBridgeOptions): value = call.arguments; } else { const text = extractTextContent(response); - if (!text) throw new ToolError("llm() returned no structured response."); + if (!text) throw new ToolError("completion() returned no structured response."); try { value = parseJsonPayload(text); } catch { - throw new ToolError("llm() did not return a structured response matching the schema."); + throw new ToolError("completion() did not return a structured response matching the schema."); } } resultText = JSON.stringify(value); } else { resultText = extractTextContent(response); - if (!resultText) throw new ToolError("llm() returned no text output."); + if (!resultText) throw new ToolError("completion() returned no text output."); } - options.emitStatus?.({ op: "llm", model: formatModelString(model), tier, chars: resultText.length }); + options.emitStatus?.({ op: "completion", model: formatModelString(model), tier, chars: resultText.length }); return { text: resultText, details: { model: formatModelString(model), tier, structured: Boolean(schema) } }; } diff --git a/packages/coding-agent/src/eval/idle-timeout.ts b/packages/coding-agent/src/eval/idle-timeout.ts index 44c438a65..a5fd40405 100644 --- a/packages/coding-agent/src/eval/idle-timeout.ts +++ b/packages/coding-agent/src/eval/idle-timeout.ts @@ -3,7 +3,7 @@ * * A cell's `timeout` bounds time while the Python kernel or JS VM is in control. * Host-side bridge calls can {@link pause} the watchdog so delegated - * `agent()`/`parallel()`/`llm()` work is ignored completely, then {@link resume} + * `agent()`/`parallel()`/`completion()` work is ignored completely, then {@link resume} * starts a fresh timeout window once the runtime gets control back. * * The active timer self-reschedules instead of being torn down on every diff --git a/packages/coding-agent/src/eval/js/shared/prelude.txt b/packages/coding-agent/src/eval/js/shared/prelude.txt index 235b0229d..c2e369263 100644 --- a/packages/coding-agent/src/eval/js/shared/prelude.txt +++ b/packages/coding-agent/src/eval/js/shared/prelude.txt @@ -57,9 +57,9 @@ if (!globalThis.__omp_js_prelude_loaded__) { const hasOwn = (object, key) => Object.prototype.hasOwnProperty.call(object, key); - const llm = async (prompt, opts, ...rest) => { - const o = optionsArg("llm", opts, rest, "{ model, system, schema }"); - const res = await globalThis.__omp_call_tool__("__llm__", { prompt, ...o }); + const completion = async (prompt, opts, ...rest) => { + const o = optionsArg("completion", opts, rest, "{ model, system, schema }"); + const res = await globalThis.__omp_call_tool__("__completion__", { prompt, ...o }); const text = res && typeof res === "object" ? res.text : res; return hasOwn(o, "schema") ? JSON.parse(text) : text; }; @@ -164,7 +164,7 @@ if (!globalThis.__omp_js_prelude_loaded__) { globalThis.print = consoleBridge.log; globalThis.display = display; globalThis.tool = tool; - globalThis.llm = llm; + globalThis.completion = completion; globalThis.output = output; globalThis.agent = agent; globalThis.parallel = parallel; diff --git a/packages/coding-agent/src/eval/js/tool-bridge.ts b/packages/coding-agent/src/eval/js/tool-bridge.ts index 97caec9df..7b3745450 100644 --- a/packages/coding-agent/src/eval/js/tool-bridge.ts +++ b/packages/coding-agent/src/eval/js/tool-bridge.ts @@ -3,8 +3,8 @@ import type { ToolSession } from "../../tools"; import { ToolError } from "../../tools/tool-errors"; import { EVAL_AGENT_BRIDGE_NAME, runEvalAgent } from "../agent-bridge"; import { EVAL_BUDGET_BRIDGE_NAME, type EvalBudgetResult, runEvalBudget } from "../budget-bridge"; +import { EVAL_COMPLETION_BRIDGE_NAME, runEvalCompletion } from "../completion-bridge"; import { EVAL_CONCURRENCY_BRIDGE_NAME, type EvalConcurrencyResult, runEvalConcurrency } from "../concurrency-bridge"; -import { EVAL_LLM_BRIDGE_NAME, runEvalLlm } from "../llm-bridge"; import type { JsStatusEvent } from "./shared/types"; export type { JsStatusEvent } from "./shared/types"; @@ -107,8 +107,8 @@ function summarizeToolResult( } export async function callSessionTool(name: string, args: unknown, options: ToolBridgeOptions): Promise { - if (name === EVAL_LLM_BRIDGE_NAME) { - return await runEvalLlm(args, options); + if (name === EVAL_COMPLETION_BRIDGE_NAME) { + return await runEvalCompletion(args, options); } if (name === EVAL_AGENT_BRIDGE_NAME) { return await runEvalAgent(args, options); diff --git a/packages/coding-agent/src/eval/py/prelude.py b/packages/coding-agent/src/eval/py/prelude.py index d167533aa..744ef453c 100644 --- a/packages/coding-agent/src/eval/py/prelude.py +++ b/packages/coding-agent/src/eval/py/prelude.py @@ -463,8 +463,8 @@ if "__omp_prelude_loaded__" not in globals(): tool = _ToolProxy() - def llm(prompt, *, model="default", system=None, schema=None): - """Oneshot, stateless LLM call against a model tier. + def completion(prompt, *, model="default", system=None, schema=None): + """Oneshot, stateless completion against a model tier. `model` selects a tier: "smol", "default" (the session's active model), or "slow". Pass `system` for a system prompt. Pass a JSON-Schema dict @@ -476,7 +476,7 @@ if "__omp_prelude_loaded__" not in globals(): args["system"] = system if schema is not None: args["schema"] = schema - res = _bridge_call("__llm__", args) + res = _bridge_call("__completion__", args) text = res.get("text") if isinstance(res, dict) else res return json.loads(text) if schema is not None else text diff --git a/packages/coding-agent/src/modes/components/tips.txt b/packages/coding-agent/src/modes/components/tips.txt index f5f42bf7c..f606541c8 100644 --- a/packages/coding-agent/src/modes/components/tips.txt +++ b/packages/coding-agent/src/modes/components/tips.txt @@ -4,7 +4,7 @@ Use /tan to fork the current conversation into a background agent Ctrl+D can be used to exit, but with your draft saved! Find out which model you emotionally abuse the most with `omp stats` Try task isolation to create CoW worktrees -Your LLM can call an LLM using `llm(x...)`. Have a big batch of tasks? Ask clanker to use it! +Need a cheap nested model call? Use `completion(x...)`. Have a big batch of tasks? Ask clanker to use it! Spaghetti code? Try complaining with /omfg Did you know? Each kitty/tmux/cmux split keeps its own session — `omp -c` resumes the right one Drop the word `ultrathink` in your message for harder multi-step reasoning — watch it glow rainbow as you type diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index 8715a5f67..5d2fd7099 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -16,7 +16,7 @@ State persists across cells, so scout in one cell and fan out in the next. Every - `agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent_type` picks a discovered agent ("explore", "reviewer", "oracle", …); `context` is shared background; `label` names the artifact. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. - `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool runs as wide as a `task` tool batch (the `task.maxConcurrency` setting; don't hand-tune it — fan out as wide as the work divides). A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one. - `pipeline(items, *stages)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. Same pool width as `parallel()`. -- `llm(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out. +- `completion(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out. - `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it. - `budget` — `budget.total` (output-token ceiling, or `None` when none is set), `budget.spent()` (tokens spent this turn — main loop + eval subagents), `budget.remaining()` (`math.inf` when total is `None`), `budget.hard` (whether it's enforced). A ceiling is set by the user: `+Nk` in their message is advisory (you self-limit via `budget.remaining()`), `+Nk!` (or Goal Mode) is hard — `agent()` refuses to spawn once spent reaches it. Gate loops on `budget.total` first, since it's `None` when the user set no budget. diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index 94ff3cb0c..35d216690 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -8,7 +8,7 @@ Cell fields: - `language` — {{#if py}}`"py"` for the IPython kernel{{/if}}{{#ifAll py js}}, {{/ifAll}}{{#if js}}`"js"` for the persistent JavaScript VM{{/if}}. - `code` — cell body, verbatim. Newlines, quotes, and indentation are JSON-encoded; no fences, no headers. - `title` (optional) — short label shown in the transcript (e.g. `"imports"`, `"load config"`). -- `timeout` (optional) — per-cell wall-clock budget in seconds (1-600). Default 30. It bounds the cell's **own** work, but is paused while an `agent()`/`parallel()`/`llm()` call is in flight — so a long fanout or a slow completion runs to completion, while the cell itself is still bounded. Compute, `print`/stdout, `log()`/`phase()`, and ordinary tool calls all count against the budget; raise `timeout` for a cell that does heavy local work or long non-agent tool calls. +- `timeout` (optional) — per-cell wall-clock budget in seconds (1-600). Default 30. It bounds the cell's **own** work, but is paused while an `agent()`/`parallel()`/`completion()` call is in flight — so a long fanout or a slow completion runs to completion, while the cell itself is still bounded. Compute, `print`/stdout, `log()`/`phase()`, and ordinary tool calls all count against the budget; raise `timeout` for a cell that does heavy local work or long non-agent tool calls. - `reset` (optional) — wipe this cell's language kernel before running.{{#ifAll py js}} Reset is per-language: a `py` cell's reset does not touch the JavaScript VM and vice versa.{{/ifAll}} **Work incrementally:** @@ -44,8 +44,8 @@ output(*ids, format?="raw", query?=None, offset?=None, limit?=None) → str | di Read task/agent output by ID. Single id returns text/dict; multiple ids return a list. tool.(args) → unknown Invoke any session tool by name. `args` is the tool's parameter object. -llm(prompt, model?="default", system?=None, schema?=None) → str | dict - Oneshot, stateless LLM call (no history, no tools). `model` picks a tier: "smol" (fast), "default" (this session's model), "slow" (most capable). Pass `system` for a system prompt. Pass a JSON-Schema `schema` to force structured output and get the parsed object back; otherwise returns the completion text. +completion(prompt, model?="default", system?=None, schema?=None) → str | dict + Oneshot, stateless completion (no history, no tools). `model` picks a tier: "smol" (fast), "default" (this session's model), "slow" (most capable). Pass `system` for a system prompt. Pass a JSON-Schema `schema` to force structured output and get the parsed object back; otherwise returns the completion text. {{#if spawns}}agent(prompt, agent_type?="task", model?=None, context?=None, label?=None, schema?=None) → str | dict Run a subagent and return its final output. Defaults to the bundled "task" agent; pass `agent_type`/`agentType` for another discovered agent. Pass a JSON-Schema `schema` to force structured output and get the parsed object back. {{#if js}} In JS, pass options as one trailing object — never positional: agent(prompt, { agentType, context, schema }). diff --git a/packages/coding-agent/src/tools/eval-render.ts b/packages/coding-agent/src/tools/eval-render.ts index c797bda0a..71730469b 100644 --- a/packages/coding-agent/src/tools/eval-render.ts +++ b/packages/coding-agent/src/tools/eval-render.ts @@ -246,7 +246,7 @@ function formatStatusEvent(event: EvalStatusEvent, theme: Theme): string { sh: "icon.package", env: "icon.package", batch: "icon.package", - llm: "icon.package", + completion: "icon.package", log: "icon.package", phase: "icon.package", }; @@ -315,7 +315,7 @@ function formatStatusEvent(event: EvalStatusEvent, theme: Theme): string { case "batch": parts.push(`${data.files} file${(data.files as number) !== 1 ? "s" : ""} processed`); break; - case "llm": + case "completion": if (data.model) parts.push(String(data.model)); if (data.tier && data.tier !== data.model) parts.push(`(${data.tier})`); parts.push(`${data.chars ?? 0} chars`); diff --git a/packages/coding-agent/src/tools/eval.ts b/packages/coding-agent/src/tools/eval.ts index eeb457a10..5f0faf0ef 100644 --- a/packages/coding-agent/src/tools/eval.ts +++ b/packages/coding-agent/src/tools/eval.ts @@ -326,7 +326,7 @@ export class EvalTool implements AgentTool { const cell = cells[i]; const backend = cell.resolved.backend; // The per-cell `timeout` is a budget on the cell runtime's *own* - // work. Host-side `agent()`/`parallel()`/`llm()` bridge calls suspend + // work. Host-side `agent()`/`parallel()`/`completion()` bridge calls suspend // that budget entirely and restart a fresh timeout window when control // returns to Python/JS. Compute, stdout, `log()`/`phase()`, and // ordinary tool calls all count against the budget. The watchdog drives diff --git a/packages/coding-agent/test/tools/eval-timeout.test.ts b/packages/coding-agent/test/tools/eval-timeout.test.ts index cd792dddd..2f3dd7fcc 100644 --- a/packages/coding-agent/test/tools/eval-timeout.test.ts +++ b/packages/coding-agent/test/tools/eval-timeout.test.ts @@ -16,7 +16,7 @@ function makeSession(): ToolSession { /** * Defends the contract that a cell which does not delegate to an `agent()`/ - * `llm()` bridge call is bounded by a *plain wall-clock* timeout — not the + * `completion()` bridge call is bounded by a *plain wall-clock* timeout — not the * activity watchdog, which now only extends the budget while a bridge call is in * flight. Regression guard for the watchdog killing ordinary compute cells and * surfacing a misleading "of inactivity" message. @@ -26,7 +26,7 @@ describe("EvalTool timeout semantics", () => { await disposeAllVmContexts(); }); - it("bounds a compute cell (no agent/llm) by a plain wall-clock timeout", async () => { + it("bounds a compute cell (no agent/completion) by a plain wall-clock timeout", async () => { const tool = new EvalTool(makeSession()); // 1s budget; the cell idles for 5s and emits no status, so nothing extends // the budget — it must be cut off at the wall-clock limit.