From 8715ed207c1d4be35a29082a0453d85711d12d1d Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 30 May 2026 00:24:08 +0200 Subject: [PATCH] feat(coding-agent-eval): added oneshot llm helper and __llm__ bridge - Added one-shot `llm(prompt, opts)` helpers in JS and Python eval runtimes. - Added `__llm__` eval bridge wiring for synthetic LLM tool dispatch and status/event output. - Added `runEvalLlm` with tier-to-model resolution, effort handling, and oneshot completion execution. - Added structured schema output handling via `respond` tool and JSON fallback parsing. - Documented new llm behavior in eval docs/changelog and added tests for tier mapping and error cases. --- docs/tools/eval.md | 16 + packages/coding-agent/CHANGELOG.md | 4 + .../src/eval/__tests__/llm-bridge.test.ts | 297 ++++++++++++++++++ .../src/eval/js/shared/prelude.txt | 8 + .../coding-agent/src/eval/js/tool-bridge.ts | 4 + packages/coding-agent/src/eval/llm-bridge.ts | 181 +++++++++++ packages/coding-agent/src/eval/py/prelude.py | 83 +++-- .../coding-agent/src/prompts/tools/eval.md | 2 + packages/coding-agent/src/tools/eval.ts | 6 + 9 files changed, 570 insertions(+), 31 deletions(-) create mode 100644 packages/coding-agent/src/eval/__tests__/llm-bridge.test.ts create mode 100644 packages/coding-agent/src/eval/llm-bridge.ts diff --git a/docs/tools/eval.md b/docs/tools/eval.md index 56753e4f5..2143db6b3 100644 --- a/docs/tools/eval.md +++ b/docs/tools/eval.md @@ -131,6 +131,7 @@ Implemented in `packages/coding-agent/src/eval/js/context-manager.ts` and `packa - `display`, `print` - `read`, `write`, `append`, `sort`, `uniq`, `counter`, `diff`, `tree`, `env`, `output` - `tool.(args)` proxy for arbitrary session tool calls + - `llm(prompt, opts)` for oneshot, stateless LLM calls (see _Oneshot LLM helper_ below) - JS helpers are async because they cross the VM/tool boundary - `display(value)` behavior: - plain objects/arrays become JSON outputs @@ -163,6 +164,21 @@ Implemented in `packages/coding-agent/src/eval/py/executor.ts`, `packages/coding - `text/html` → HTML converted to markdown with `htmlToBasicMarkdown()` - Interactive stdin is rejected: `input_request` sends an empty reply, marks `stdinRequested`, and the executor returns exit code `1` +### Oneshot LLM helper (`llm`) + +Both runtimes expose `llm()` — a single stateless completion against a model tier. It is intentionally minimal: no conversation history, no agent-visible tools, pure text in / text (or object) out. Implemented host-side in `packages/coding-agent/src/eval/llm-bridge.ts` and routed through the existing tool bridge under the reserved name `__llm__`. + +- Signatures: + - JS: `await llm(prompt, { model?, system?, schema? })` + - Python: `llm(prompt, *, model="default", system=None, schema=None)` +- `model` selects a tier (default `"default"`): + - `"smol"` → `pi/smol` role (fast / cheap) + - `"default"` → the session's active model, falling back to the `pi/default` role + - `"slow"` → `pi/slow` role; requests high reasoning effort only on reasoning-capable models +- `system` (optional) supplies a system prompt. +- `schema` (optional) is a plain JSON-Schema object. When present, the model is forced to call a single synthetic `respond` tool with that schema (loose, non-strict), and the helper returns the parsed object. When absent, the helper returns the completion string. +- Errors surface as exceptions: unresolved tier, missing API key, an `error`/`aborted` stop reason, or empty output each raise. + ### Multi-language call behavior A single tool call can mix Python and JS cells. Persistence is per language runtime: diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index e93eaeaff..87a3545f0 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,10 @@ # Changelog ## [Unreleased] +### Added + +- Added progress status output for `llm()` calls in `eval`, including the resolved model, tier, and returned character count +- Added an `llm(prompt, opts)` helper to both `eval` runtimes (JavaScript and Python) for oneshot, stateless LLM calls. `opts.model` selects a tier — `"smol"` (`pi/smol`), `"default"` (the session's active model, falling back to `pi/default`), or `"slow"` (`pi/slow`, with high reasoning effort on reasoning-capable models). Pass `system` for a system prompt and a plain JSON-Schema `schema` to force a structured response (the helper returns the parsed object instead of the completion string). Calls carry no conversation history and expose no agent-visible tools; they route host-side through the existing tool bridge under the reserved name `__llm__` (`packages/coding-agent/src/eval/llm-bridge.ts`). ### Fixed diff --git a/packages/coding-agent/src/eval/__tests__/llm-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/llm-bridge.test.ts new file mode 100644 index 000000000..c5d2ce6be --- /dev/null +++ b/packages/coding-agent/src/eval/__tests__/llm-bridge.test.ts @@ -0,0 +1,297 @@ +import { afterAll, afterEach, describe, expect, it, vi } from "bun:test"; +import * as path from "node:path"; +import type { Api, AssistantMessage, Model } from "@oh-my-pi/pi-ai"; +import * as ai from "@oh-my-pi/pi-ai"; +import { Effort } from "@oh-my-pi/pi-ai"; +import { TempDir } from "@oh-my-pi/pi-utils"; +import type { ModelRegistry } from "../../config/model-registry"; +import { Settings } from "../../config/settings"; +import type { ToolSession } from "../../tools"; +import { ToolError } from "../../tools/tool-errors"; +import { disposeAllVmContexts } from "../js/context-manager"; +import { executeJs } from "../js/executor"; +import { runEvalLlm } from "../llm-bridge"; +import { disposeAllKernelSessions, executePython } from "../py/executor"; + +function makeModel(provider: string, id: string, extra: Partial> = {}): Model { + return { + id, + name: id, + api: "openai-responses", + provider, + baseUrl: "https://example.test/v1", + reasoning: false, + input: ["text"], + cost: { input: 1, output: 1, cacheRead: 0, cacheWrite: 1 }, + contextWindow: 128000, + maxTokens: 4096, + ...extra, + } as Model; +} + +const SMOL = makeModel("p", "smol"); +const DEFAULT = makeModel("p", "default"); +const SLOW = makeModel("p", "slow"); +const REASONING_SLOW = makeModel("p", "slow", { + api: "anthropic-messages", + reasoning: true, + thinking: { minLevel: Effort.Low, maxLevel: Effort.High, mode: "anthropic-adaptive" }, +}); + +interface SessionOptions { + available?: Model[]; + apiKey?: string | null; + activeModel?: string; + roles?: Partial>; +} + +function makeSession(opts: SessionOptions = {}): ToolSession { + const settings = Settings.isolated({ "async.enabled": false, "task.isolation.mode": "none" }); + const roles = opts.roles ?? { smol: "p/smol", slow: "p/slow" }; + for (const role in roles) { + const value = roles[role as keyof typeof roles]; + if (value) settings.setModelRole(role, value); + } + const modelRegistry = { + getAvailable: () => opts.available ?? [SMOL, DEFAULT, SLOW], + getApiKey: async () => (opts.apiKey === undefined ? "test-key" : opts.apiKey), + } as unknown as ModelRegistry; + return { + settings, + modelRegistry, + getActiveModelString: () => opts.activeModel ?? "p/default", + } as unknown as ToolSession; +} + +function assistant(opts: { + text?: string; + toolCall?: { name: string; arguments: Record }; + stopReason?: AssistantMessage["stopReason"]; + errorMessage?: string; +}): AssistantMessage { + const content: AssistantMessage["content"] = []; + if (opts.text) content.push({ type: "text", text: opts.text }); + if (opts.toolCall) { + content.push({ type: "toolCall", id: "tc-1", name: opts.toolCall.name, arguments: opts.toolCall.arguments }); + } + return { + role: "assistant", + content, + api: "openai-responses", + provider: "p", + model: "default", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: opts.stopReason ?? "stop", + errorMessage: opts.errorMessage, + timestamp: Date.now(), + }; +} + +describe("runEvalLlm", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("resolves each tier to its expected model", async () => { + const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); + const session = makeSession(); + + await runEvalLlm({ prompt: "q", model: "smol" }, { session }); + await runEvalLlm({ prompt: "q", model: "default" }, { session }); + await runEvalLlm({ prompt: "q", model: "slow" }, { session }); + + const resolved = spy.mock.calls.map(call => { + const model = call[0] as Model; + return `${model.provider}/${model.id}`; + }); + expect(resolved).toEqual(["p/smol", "p/default", "p/slow"]); + }); + + it("prefers the session active model for the default tier, falling back to pi/default", async () => { + const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); + const session = makeSession({ available: [SMOL, DEFAULT, SLOW], activeModel: "p/slow" }); + + await runEvalLlm({ prompt: "q", model: "default" }, { session }); + + const model = spy.mock.calls[0]?.[0] as Model; + expect(`${model.provider}/${model.id}`).toBe("p/slow"); + }); + + it("returns the completion text in plain mode", async () => { + vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "the answer" })); + const result = await runEvalLlm({ prompt: "q", model: "smol" }, { session: makeSession() }); + expect(result.text).toBe("the answer"); + expect(result.details).toEqual({ model: "p/smol", tier: "smol", structured: false }); + }); + + it("forces a respond tool call and returns its arguments in structured mode", async () => { + const spy = vi + .spyOn(ai, "completeSimple") + .mockResolvedValue(assistant({ toolCall: { name: "respond", arguments: { answer: 42 } } })); + const result = await runEvalLlm( + { prompt: "q", model: "smol", schema: { type: "object", properties: { answer: { type: "number" } } } }, + { session: makeSession() }, + ); + + expect(JSON.parse(result.text)).toEqual({ answer: 42 }); + expect(result.details.structured).toBe(true); + + const ctx = spy.mock.calls[0]?.[1] as { tools?: Array<{ name: string }> }; + const opts = spy.mock.calls[0]?.[2] as { toolChoice?: unknown }; + expect(ctx.tools?.[0]?.name).toBe("respond"); + expect(opts.toolChoice).toEqual({ type: "tool", name: "respond" }); + }); + + it("falls back to JSON embedded in text when the model skips the respond tool", async () => { + vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: 'here: {"answer": 7}' })); + const result = await runEvalLlm( + { prompt: "q", model: "smol", schema: { type: "object" } }, + { session: makeSession() }, + ); + expect(JSON.parse(result.text)).toEqual({ answer: 7 }); + }); + + it("requests reasoning only for the slow tier on a reasoning-capable model", async () => { + const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); + const session = makeSession({ available: [SMOL, DEFAULT, REASONING_SLOW] }); + + await runEvalLlm({ prompt: "q", model: "smol" }, { session }); + await runEvalLlm({ prompt: "q", model: "slow" }, { session }); + + const smolOpts = spy.mock.calls[0]?.[2] as { reasoning?: unknown }; + const slowOpts = spy.mock.calls[1]?.[2] as { reasoning?: unknown }; + expect(smolOpts.reasoning).toBeUndefined(); + expect(slowOpts.reasoning).toBe(Effort.High); + }); + + it("does not request reasoning for the slow tier on a non-reasoning model", async () => { + const spy = vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "ok" })); + // SLOW is reasoning:false — must not trip requireSupportedEffort downstream. + const result = await runEvalLlm({ prompt: "q", model: "slow" }, { session: makeSession() }); + expect(result.text).toBe("ok"); + const opts = spy.mock.calls[0]?.[2] as { reasoning?: unknown }; + expect(opts.reasoning).toBeUndefined(); + }); + + it("throws ToolError on invalid arguments", async () => { + await expect(runEvalLlm({ prompt: "" }, { session: makeSession() })).rejects.toBeInstanceOf(ToolError); + await expect(runEvalLlm({ prompt: "q", model: "huge" }, { session: makeSession() })).rejects.toBeInstanceOf( + ToolError, + ); + }); + + it("throws ToolError when no model resolves for the tier", async () => { + const session = makeSession({ available: [DEFAULT], roles: { smol: "missing/model" } }); + await expect(runEvalLlm({ prompt: "q", model: "smol" }, { session })).rejects.toBeInstanceOf(ToolError); + }); + + it("throws ToolError when the resolved model has no API key", async () => { + const session = makeSession({ apiKey: null }); + await expect(runEvalLlm({ prompt: "q", model: "smol" }, { session })).rejects.toBeInstanceOf(ToolError); + }); + + it("maps error and aborted stop reasons to ToolError", async () => { + vi.spyOn(ai, "completeSimple").mockResolvedValueOnce(assistant({ stopReason: "error", errorMessage: "boom" })); + await expect(runEvalLlm({ prompt: "q", model: "smol" }, { session: makeSession() })).rejects.toThrow("boom"); + + vi.spyOn(ai, "completeSimple").mockResolvedValueOnce(assistant({ stopReason: "aborted" })); + await expect(runEvalLlm({ prompt: "q", model: "smol" }, { session: makeSession() })).rejects.toBeInstanceOf( + ToolError, + ); + }); + + it("throws ToolError when plain mode produces no text", async () => { + vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "" })); + await expect(runEvalLlm({ prompt: "q", model: "smol" }, { session: makeSession() })).rejects.toBeInstanceOf( + ToolError, + ); + }); +}); + +describe("llm() through eval runtimes", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + afterAll(async () => { + await disposeAllVmContexts(); + await disposeAllKernelSessions(); + }); + + it("exposes llm() in the JavaScript runtime", async () => { + using tempDir = TempDir.createSync("@omp-eval-llm-js-"); + const sessionFile = path.join(tempDir.path(), "session.jsonl"); + const sessionId = `js-llm:${crypto.randomUUID()}`; + vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "hello from smol" })); + + const result = await executeJs('return await llm("hi", { model: "smol" });', { + cwd: tempDir.path(), + sessionId, + session: makeSession(), + sessionFile, + }); + + expect(result.exitCode).toBe(0); + expect(result.output.trim()).toBe("hello from smol"); + }); + + it("parses structured llm() output in the JavaScript runtime", async () => { + using tempDir = TempDir.createSync("@omp-eval-llm-js-struct-"); + const sessionFile = path.join(tempDir.path(), "session.jsonl"); + const sessionId = `js-llm-struct:${crypto.randomUUID()}`; + vi.spyOn(ai, "completeSimple").mockResolvedValue( + assistant({ toolCall: { name: "respond", arguments: { ok: true, n: 3 } } }), + ); + + const result = await executeJs( + 'const r = await llm("hi", { schema: { type: "object" } }); return JSON.stringify(r);', + { cwd: tempDir.path(), sessionId, session: makeSession(), sessionFile }, + ); + + expect(result.exitCode).toBe(0); + expect(JSON.parse(result.output.trim())).toEqual({ ok: true, n: 3 }); + }); + + it("exposes llm() in the Python runtime", async () => { + using tempDir = TempDir.createSync("@omp-eval-llm-py-"); + const sessionFile = path.join(tempDir.path(), "session.jsonl"); + const sessionId = `py-llm:${crypto.randomUUID()}`; + vi.spyOn(ai, "completeSimple").mockResolvedValue(assistant({ text: "hello from python" })); + + const result = await executePython('print(llm("hi", model="smol"))', { + cwd: tempDir.path(), + sessionId, + sessionFile, + toolSession: makeSession(), + }); + + expect(result.exitCode).toBe(0); + expect(result.output.trim()).toBe("hello from python"); + }); + + it("parses structured llm() output in the Python runtime", async () => { + using tempDir = TempDir.createSync("@omp-eval-llm-py-struct-"); + const sessionFile = path.join(tempDir.path(), "session.jsonl"); + const sessionId = `py-llm-struct:${crypto.randomUUID()}`; + vi.spyOn(ai, "completeSimple").mockResolvedValue( + assistant({ toolCall: { name: "respond", arguments: { ok: true } } }), + ); + + const result = await executePython('import json\nprint(json.dumps(llm("hi", schema={"type": "object"})))', { + cwd: tempDir.path(), + sessionId, + sessionFile, + toolSession: makeSession(), + }); + + expect(result.exitCode).toBe(0); + expect(JSON.parse(result.output.trim())).toEqual({ ok: true }); + }); +}); diff --git a/packages/coding-agent/src/eval/js/shared/prelude.txt b/packages/coding-agent/src/eval/js/shared/prelude.txt index 0a7ce544d..89f5a76a1 100644 --- a/packages/coding-agent/src/eval/js/shared/prelude.txt +++ b/packages/coding-agent/src/eval/js/shared/prelude.txt @@ -39,6 +39,13 @@ if (!globalThis.__omp_js_prelude_loaded__) { return values.length === 1 ? values[0] : values; }; + const llm = async (prompt, opts = {}) => { + const o = toOptions(opts); + const res = await globalThis.__omp_call_tool__("__llm__", { prompt, ...o }); + const text = res && typeof res === "object" ? res.text : res; + return o.schema ? JSON.parse(text) : text; + }; + const display = value => { globalThis.__omp_display__(value); }; @@ -61,6 +68,7 @@ if (!globalThis.__omp_js_prelude_loaded__) { globalThis.print = consoleBridge.log; globalThis.display = display; globalThis.tool = tool; + globalThis.llm = llm; globalThis.output = output; globalThis.read = read; globalThis.write = write; diff --git a/packages/coding-agent/src/eval/js/tool-bridge.ts b/packages/coding-agent/src/eval/js/tool-bridge.ts index 205b9d383..d587a3477 100644 --- a/packages/coding-agent/src/eval/js/tool-bridge.ts +++ b/packages/coding-agent/src/eval/js/tool-bridge.ts @@ -1,6 +1,7 @@ import type { AgentTool, AgentToolResult } from "@oh-my-pi/pi-agent-core"; import type { ToolSession } from "../../tools"; import { ToolError } from "../../tools/tool-errors"; +import { EVAL_LLM_BRIDGE_NAME, runEvalLlm } from "../llm-bridge"; import type { JsStatusEvent } from "./shared/types"; export type { JsStatusEvent } from "./shared/types"; @@ -101,6 +102,9 @@ function summarizeToolResult( } export async function callSessionTool(name: string, args: unknown, options: ToolBridgeOptions): Promise { + if (name === EVAL_LLM_BRIDGE_NAME) { + return await runEvalLlm(args, options); + } const tool = getTool(options.session, name); const normalizedArgs = normalizeArgs(args); const toolCallId = `js-${name}-${crypto.randomUUID()}`; diff --git a/packages/coding-agent/src/eval/llm-bridge.ts b/packages/coding-agent/src/eval/llm-bridge.ts new file mode 100644 index 000000000..301d553bf --- /dev/null +++ b/packages/coding-agent/src/eval/llm-bridge.ts @@ -0,0 +1,181 @@ +/** + * Host-side handler for the eval `llm()` helper. + * + * Both eval runtimes (JS worker + Python kernel) route helper→host calls + * through {@link callSessionTool}. Reserving the synthetic tool name + * {@link EVAL_LLM_BRIDGE_NAME} lets a single host handler serve both + * transports without registering an agent-visible tool: cell code calls + * `llm(prompt, opts)`, the prelude forwards `{ prompt, model, system?, schema? }` + * through the bridge, and this module performs one stateless completion. + * + * The call is oneshot and toolless from the model's perspective — pure text + * in, text (or, with `schema`, a structured object) out. + */ +import { instrumentedCompleteSimple, resolveTelemetry } from "@oh-my-pi/pi-agent-core"; +import { type Api, Effort, getSupportedEfforts, type Model, type Tool } from "@oh-my-pi/pi-ai"; +import * as z from "zod/v4"; +import { extractTextContent, extractToolCall, parseJsonPayload } from "../commit/utils"; +import { expandRoleAlias, formatModelString, resolveModelFromString } from "../config/model-resolver"; +import type { ToolSession } from "../tools"; +import { ToolError } from "../tools/tool-errors"; +import type { JsStatusEvent } from "./js/shared/types"; + +/** Synthetic bridge name reserved for the `llm()` helper across both runtimes. */ +export const EVAL_LLM_BRIDGE_NAME = "__llm__"; + +/** Synthetic tool the model is forced to call when a `schema` is supplied. */ +const STRUCTURED_TOOL_NAME = "respond"; + +type LlmTier = "smol" | "default" | "slow"; + +const TIER_TO_PATTERN: Record = { + smol: "pi/smol", + default: "pi/default", + slow: "pi/slow", +}; + +const llmArgsSchema = z.object({ + prompt: z.string().min(1, "prompt must be a non-empty string"), + model: z.enum(["smol", "default", "slow"]).default("default"), + system: z.string().optional(), + schema: z.record(z.string(), z.unknown()).optional(), +}); + +export interface EvalLlmBridgeOptions { + session: ToolSession; + signal?: AbortSignal; + emitStatus?: (event: JsStatusEvent) => void; +} + +export interface EvalLlmResult { + text: string; + details: { model: string; tier: LlmTier; structured: boolean }; +} + +/** + * Resolve a tier to a concrete {@link Model}. `default` prefers the session's + * active model and falls back to the `pi/default` role; `smol`/`slow` resolve + * their respective role patterns. Returns `undefined` when nothing matches. + */ +function resolveTierModel(tier: LlmTier, session: ToolSession): Model | undefined { + const modelRegistry = session.modelRegistry; + if (!modelRegistry) return undefined; + const available = modelRegistry.getAvailable(); + if (available.length === 0) return undefined; + + const matchPreferences = { usageOrder: session.settings.getStorage()?.getModelUsageOrder() }; + const resolve = (pattern: string | undefined): Model | undefined => { + if (!pattern) return undefined; + const expanded = expandRoleAlias(pattern, session.settings); + return resolveModelFromString(expanded, available, matchPreferences, modelRegistry); + }; + + if (tier === "default") { + const activePattern = session.getActiveModelString?.() ?? session.getModelString?.(); + return resolve(activePattern) ?? resolve(TIER_TO_PATTERN.default); + } + return resolve(TIER_TO_PATTERN[tier]); +} + +/** + * Choose the reasoning effort for a tier. Only `slow` opts into thinking, and + * only on reasoning-capable models — guarding against `requireSupportedEffort` + * throwing downstream on models that cannot reason. Clamps to the highest + * supported effort so a reasoning model without `high` does not 400. + */ +function reasoningForTier(tier: LlmTier, model: Model): Effort | undefined { + if (tier !== "slow" || !model.reasoning) return undefined; + const efforts = getSupportedEfforts(model); + if (efforts.length === 0) return undefined; + return efforts.includes(Effort.High) ? Effort.High : efforts[efforts.length - 1]; +} + +/** + * Run a single stateless completion on behalf of an eval cell's `llm()` call. + * Returns a `{ text, details }` value shaped like a {@link callSessionTool} + * result so the existing bridge transport carries it to either runtime. + */ +export async function runEvalLlm(args: unknown, options: EvalLlmBridgeOptions): Promise { + const parsed = llmArgsSchema.safeParse(args); + if (!parsed.success) { + const issue = parsed.error.issues[0]; + const where = issue?.path.length ? `${issue.path.join(".")}: ` : ""; + throw new ToolError(`llm() received invalid arguments: ${where}${issue?.message ?? "bad input"}`); + } + const { prompt, model: tier, system, schema } = parsed.data; + + const model = resolveTierModel(tier, options.session); + if (!model) { + throw new ToolError( + `llm() could not resolve a model for the "${tier}" tier. Configure modelRoles.${tier === "default" ? "default" : tier} or ensure a provider is available.`, + ); + } + + const apiKey = await options.session.modelRegistry?.getApiKey(model); + if (!apiKey) { + throw new ToolError( + `llm() has no API key for ${formatModelString(model)}. Configure credentials for this provider or choose another tier.`, + ); + } + + const tools: Tool[] | undefined = schema + ? [ + { + name: STRUCTURED_TOOL_NAME, + description: "Return your answer by calling this tool with the requested structured fields.", + parameters: schema, + strict: false, + }, + ] + : undefined; + + const telemetry = resolveTelemetry(options.session.getTelemetry?.(), options.session.getSessionId?.() ?? undefined); + + const response = await instrumentedCompleteSimple( + model, + { + systemPrompt: system ? [system] : undefined, + messages: [{ role: "user", content: [{ type: "text", text: prompt }], timestamp: Date.now() }], + tools, + }, + { + apiKey, + signal: options.signal, + reasoning: reasoningForTier(tier, model), + toolChoice: schema ? { type: "tool", name: STRUCTURED_TOOL_NAME } : undefined, + }, + { telemetry, oneshotKind: "eval_llm" }, + ); + + if (response.stopReason === "error") { + throw new ToolError(response.errorMessage ?? "llm() request failed."); + } + if (response.stopReason === "aborted") { + throw new ToolError("llm() request aborted."); + } + + let resultText: string; + if (schema) { + const call = extractToolCall(response, STRUCTURED_TOOL_NAME); + let value: unknown; + if (call) { + value = call.arguments; + } else { + const text = extractTextContent(response); + if (!text) throw new ToolError("llm() returned no structured response."); + try { + value = parseJsonPayload(text); + } catch { + throw new ToolError("llm() did not return a structured response matching the schema."); + } + } + resultText = JSON.stringify(value); + } else { + resultText = extractTextContent(response); + if (!resultText) throw new ToolError("llm() returned no text output."); + } + + options.emitStatus?.({ op: "llm", model: formatModelString(model), tier, chars: resultText.length }); + + return { text: resultText, details: { model: formatModelString(model), tier, structured: Boolean(schema) } }; +} diff --git a/packages/coding-agent/src/eval/py/prelude.py b/packages/coding-agent/src/eval/py/prelude.py index 7858c3286..24e761ab8 100644 --- a/packages/coding-agent/src/eval/py/prelude.py +++ b/packages/coding-agent/src/eval/py/prelude.py @@ -385,6 +385,40 @@ if "__omp_prelude_loaded__" not in globals(): raise RuntimeError("tool bridge is unavailable in this kernel") return (base.rstrip("/"), token, session) + def _bridge_call(name: str, args: dict): + """POST one request to the host tool bridge and return its `value`.""" + import urllib.request, urllib.error + base, token, session = _tool_proxy_from_env() + _run_id_getter = globals().get("__omp_current_run_id__") + _run_id = _run_id_getter() if callable(_run_id_getter) else globals().get("__omp_run_id__") + payload = json.dumps( + {"session": session, "run": _run_id, "name": name, "args": args} + ).encode("utf-8") + req = urllib.request.Request( + f"{base}/v1/tool", + data=payload, + method="POST", + headers={ + "Content-Type": "application/json", + "Authorization": f"Bearer {token}", + }, + ) + try: + with urllib.request.urlopen(req) as resp: + body = resp.read() + except urllib.error.HTTPError as exc: + body = exc.read() + try: + data = json.loads(body) + except json.JSONDecodeError: + raise RuntimeError( + f"bridge call {name!r}: non-JSON response: {body[:200]!r}" + ) from None + if not isinstance(data, dict) or not data.get("ok"): + msg = (data or {}).get("error") if isinstance(data, dict) else None + raise RuntimeError(msg or f"bridge call {name!r} failed") + return data.get("value") + class _ToolCallable: """Invokes one host-side tool via the loopback HTTP bridge.""" @@ -397,7 +431,6 @@ if "__omp_prelude_loaded__" not in globals(): return f"" def __call__(self, args=None, /, **kwargs): - import urllib.request, urllib.error if args is None: merged: dict = {} elif isinstance(args, dict): @@ -409,36 +442,7 @@ if "__omp_prelude_loaded__" not in globals(): merged.update(kwargs) if "_i" not in merged: merged["_i"] = "py prelude" - base, token, session = _tool_proxy_from_env() - _run_id_getter = globals().get("__omp_current_run_id__") - _run_id = _run_id_getter() if callable(_run_id_getter) else globals().get("__omp_run_id__") - payload = json.dumps( - {"session": session, "run": _run_id, "name": self._name, "args": merged} - ).encode("utf-8") - req = urllib.request.Request( - f"{base}/v1/tool", - data=payload, - method="POST", - headers={ - "Content-Type": "application/json", - "Authorization": f"Bearer {token}", - }, - ) - try: - with urllib.request.urlopen(req) as resp: - body = resp.read() - except urllib.error.HTTPError as exc: - body = exc.read() - try: - data = json.loads(body) - except json.JSONDecodeError: - raise RuntimeError( - f"tool.{self._name}: bridge returned non-JSON response: {body[:200]!r}" - ) from None - if not isinstance(data, dict) or not data.get("ok"): - msg = (data or {}).get("error") if isinstance(data, dict) else None - raise RuntimeError(msg or f"tool.{self._name} failed") - return data.get("value") + return _bridge_call(self._name, merged) class _ToolProxy: """`tool.(args)` proxy mirroring the JS runtime bridge.""" @@ -458,3 +462,20 @@ if "__omp_prelude_loaded__" not in globals(): return f"" if session else "" tool = _ToolProxy() + + def llm(prompt, *, model="default", system=None, schema=None): + """Oneshot, stateless LLM call against a model tier. + + `model` selects a tier: "smol", "default" (the session's active model), + or "slow". Pass `system` for a system prompt. Pass a JSON-Schema dict + as `schema` to force a structured response; the parsed object is then + returned instead of the completion text. + """ + args = {"prompt": prompt, "model": model} + if system is not None: + args["system"] = system + if schema is not None: + args["schema"] = schema + res = _bridge_call("__llm__", args) + text = res.get("text") if isinstance(res, dict) else res + return json.loads(text) if schema is not None else text diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index b95b09677..2814b1e17 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -44,6 +44,8 @@ output(*ids, format?="raw", query?=None, offset?=None, limit?=None) → str | di Read task/agent output by ID. Single id returns text/dict; multiple ids return a list. tool.(args) → unknown Invoke any session tool by name. `args` is the tool's parameter object. +llm(prompt, model?="default", system?=None, schema?=None) → str | dict + Oneshot, stateless LLM call (no history, no tools). `model` picks a tier: "smol" (fast), "default" (this session's model), "slow" (most capable). Pass `system` for a system prompt. Pass a JSON-Schema `schema` to force structured output and get the parsed object back; otherwise returns the completion text. ``` diff --git a/packages/coding-agent/src/tools/eval.ts b/packages/coding-agent/src/tools/eval.ts index 428eb1f2b..a9043754d 100644 --- a/packages/coding-agent/src/tools/eval.ts +++ b/packages/coding-agent/src/tools/eval.ts @@ -634,6 +634,7 @@ function formatStatusEvent(event: EvalStatusEvent, theme: Theme): string { sh: "icon.package", env: "icon.package", batch: "icon.package", + llm: "icon.package", }; const iconKey = opIcons[op] ?? "icon.file"; @@ -700,6 +701,11 @@ function formatStatusEvent(event: EvalStatusEvent, theme: Theme): string { case "batch": parts.push(`${data.files} file${(data.files as number) !== 1 ? "s" : ""} processed`); break; + case "llm": + if (data.model) parts.push(String(data.model)); + if (data.tier && data.tier !== data.model) parts.push(`(${data.tier})`); + parts.push(`${data.chars ?? 0} chars`); + break; case "wc": parts.push(`${data.lines}L ${data.words}W ${data.chars}C`); break;