diff --git a/docs/tools/eval.md b/docs/tools/eval.md index 0fccef25b..054816a0f 100644 --- a/docs/tools/eval.md +++ b/docs/tools/eval.md @@ -138,7 +138,7 @@ Implemented in `packages/coding-agent/src/eval/js/worker-core.ts`, `packages/cod - `await read(path, { offset?, limit? })` - `await tree(path = ".", { maxDepth?, hidden? })` - `sort(text, { reverse?, unique? })`, `uniq(text, { count? })`, `counter(items, { limit?, reverse? })` - - `await agent(prompt, { agentType?, model?, context?, label?, schema? })` + - `await agent(prompt, { agentType?, model?, label?, schema? })` - `await parallel([() => agent("a"), () => agent("b")])` - `await pipeline(items, stage1, stage2)` - `display(value)` behavior: @@ -192,11 +192,11 @@ Both runtimes expose `completion()` — a single stateless completion against a Both runtimes expose `agent()` — a single subagent invocation routed through `packages/coding-agent/src/eval/agent-bridge.ts` into the same `runSubprocess(...)` path used by the `task` tool. It uses the current eval session's spawn policy and inherits the parent eval executor id, so parent and subagent code share JS/Python runtime state. - Signatures: - - JS: `await agent(prompt, { agentType?, model?, context?, label?, schema? })` - - Python: `agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None)` + - JS: `await agent(prompt, { agentType?, model?, label?, schema? })` + - Python: `agent(prompt, *, agent_type="task", model=None, label=None, schema=None)` - `agentType` / `agent_type` defaults to the bundled `task` agent and resolves through normal agent discovery, so project and user agents work. - `model` overrides the selected agent's model. Without it, normal per-agent settings and the agent frontmatter model apply. -- `context` supplies shared background; `label` controls the `agent://` output label prefix. +- Shared background is passed via files: write a `local://` file and reference it in the prompt. `label` controls the `agent://` output label prefix. - `schema` passes a JSON Schema to the subagent structured-output path. When present, the helper parses the final JSON text and returns an object. - Spawn restrictions use `session.getSessionSpawns()` exactly like the `task` tool. Eval-driven subagent recursion is capped at depth 3. - JS and Python both expose `parallel(thunks)` and `pipeline(items, ...stages)`; both use a bounded async/threaded pool whose width tracks the `task.maxConcurrency` setting (the same ceiling the `task` tool uses; `0` = run every item at once), preserve item order, and propagate rejections. The width is fetched live from the host via the `__concurrency__` bridge, so the helpers no longer take a `concurrency` argument. diff --git a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts index a5e263cf8..588fd1b5a 100644 --- a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts @@ -178,7 +178,7 @@ describe("runEvalAgent", () => { expect(runSpy).not.toHaveBeenCalled(); }); - it("passes the parent execution context and only sets outputSchema when schema is supplied", async () => { + it("passes parent execution options and only sets outputSchema when schema is supplied", async () => { mockAgents(); const runSpy = vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => singleResult(options)); const abortController = new AbortController(); @@ -186,7 +186,7 @@ describe("runEvalAgent", () => { const session = makeSession({ depth: 2, activeModel: "p/current", modelString: "p/fallback" }); await runEvalAgent( - { prompt: " hello ", context: " context ", label: "My Agent", model: "p/override", schema }, + { prompt: " hello ", label: "My Agent", model: "p/override", schema }, { session, signal: abortController.signal }, ); await runEvalAgent({ prompt: "plain" }, { session }); @@ -199,7 +199,6 @@ describe("runEvalAgent", () => { expect(firstOptions.parentActiveModelPattern).toBe("p/current"); expect(firstOptions.outputSchema).toBe(schema); expect(firstOptions.assignment).toBe("hello"); - expect(firstOptions.context).toBe("context"); expect(firstOptions.description).toBe("My Agent"); expect(firstOptions.modelOverride).toEqual(["p/override"]); expect(secondOptions.outputSchema).toBeUndefined(); diff --git a/packages/coding-agent/src/eval/agent-bridge.ts b/packages/coding-agent/src/eval/agent-bridge.ts index 23a4ecff0..d18bb48b4 100644 --- a/packages/coding-agent/src/eval/agent-bridge.ts +++ b/packages/coding-agent/src/eval/agent-bridge.ts @@ -34,7 +34,6 @@ const agentArgsSchema = z.object({ prompt: z.string().min(1, "prompt must be a non-empty string"), agentType: z.string().min(1).optional(), model: z.union([z.string().min(1), z.array(z.string().min(1)).min(1)]).optional(), - context: z.string().optional(), label: z.string().optional(), schema: z.unknown().optional(), }); @@ -43,7 +42,6 @@ interface EvalAgentArgs { prompt: string; agentType?: string; model?: string | string[]; - context?: string; label?: string; schema?: unknown; } @@ -135,20 +133,12 @@ function getOutputManager(session: ToolSession): AgentOutputManager { async function getArtifacts(session: ToolSession): Promise<{ sessionFile: string | null; artifactsDir: string; - contextFile?: string; }> { const sessionFile = session.getSessionFile(); const sessionArtifactsDir = sessionFile ? sessionFile.slice(0, -6) : null; const artifactsDir = sessionArtifactsDir ?? path.join(os.tmpdir(), `omp-eval-agent-${Snowflake.next()}`); await fs.mkdir(artifactsDir, { recursive: true }); - - const shouldWriteConversationContext = session.settings.get("irc.enabled") !== true; - const compactContext = shouldWriteConversationContext ? session.getCompactContext?.() : undefined; - if (!compactContext) return { sessionFile, artifactsDir }; - - const contextFile = path.join(artifactsDir, "context.md"); - await Bun.write(contextFile, compactContext); - return { sessionFile, artifactsDir, contextFile }; + return { sessionFile, artifactsDir }; } function emitProgressStatus(emitStatus: ((event: JsStatusEvent) => void) | undefined, progress: AgentProgress): void { @@ -246,11 +236,10 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption }; const parentArtifactManager = options.session.getArtifactManager?.() ?? undefined; const mcpManager = options.session.mcpManager ?? MCPManager.instance(); - const { sessionFile, artifactsDir, contextFile } = await getArtifacts(options.session); + const { sessionFile, artifactsDir } = await getArtifacts(options.session); const outputManager = getOutputManager(options.session); const id = await outputManager.allocate(outputIdBase(parsed.label, agentName)); const assignment = parsed.prompt.trim(); - const context = trimToUndefined(parsed.context); // Suspend eval timeout accounting while the subagent owns control. The // timeout clock restarts once the bridge returns to the cell runtime. const result = await withBridgeTimeoutPause(options.emitStatus, () => @@ -259,7 +248,6 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption agent: effectiveAgent, task: renderSubagentPrompt(assignment), assignment, - context, description: trimToUndefined(parsed.label), index: 0, id, @@ -271,7 +259,6 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption sessionFile, persistArtifacts: Boolean(sessionFile), artifactsDir, - contextFile, // Eval `agent()` subagents are short-lived programmatic helpers (data // collection, structured output, parallel() fan-out). LSP server // cold-start costs tens of seconds and is pure overhead here, so it is diff --git a/packages/coding-agent/src/eval/js/shared/prelude.txt b/packages/coding-agent/src/eval/js/shared/prelude.txt index 36b61c5ab..095b219a2 100644 --- a/packages/coding-agent/src/eval/js/shared/prelude.txt +++ b/packages/coding-agent/src/eval/js/shared/prelude.txt @@ -65,7 +65,7 @@ if (!globalThis.__omp_js_prelude_loaded__) { }; const agent = async (prompt, opts, ...rest) => { - const o = optionsArg("agent", opts, rest, "{ agentType, model, context, label, schema }"); + const o = optionsArg("agent", opts, rest, "{ agentType, model, label, schema }"); const res = await globalThis.__omp_call_tool__("__agent__", { prompt, ...o }); const text = res && typeof res === "object" ? res.text : res; return hasOwn(o, "schema") ? JSON.parse(text) : text; diff --git a/packages/coding-agent/src/eval/py/prelude.py b/packages/coding-agent/src/eval/py/prelude.py index 0eb2ec942..e2ac422d0 100644 --- a/packages/coding-agent/src/eval/py/prelude.py +++ b/packages/coding-agent/src/eval/py/prelude.py @@ -519,21 +519,20 @@ if "__omp_prelude_loaded__" not in globals(): text = res.get("text") if isinstance(res, dict) else res return json.loads(text) if schema is not None else text - def agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None): + def agent(prompt, *, agent_type="task", model=None, label=None, schema=None): """Run a subagent and return its final output. `agent_type` selects the subagent definition (default "task"). Pass - `model` to override that agent's model, `context` for shared background, - `label` for the output artifact id, and `schema` to request structured - JSON output; when `schema` is supplied the parsed object is returned. + `model` to override that agent's model, `label` for the output artifact + id, and `schema` to request structured JSON output; when `schema` is + supplied the parsed object is returned. Share background by writing a + local:// file and referencing it in the prompt. """ args = {"prompt": prompt} if agent_type is not None: args["agentType"] = agent_type if model is not None: args["model"] = model - if context is not None: - args["context"] = context if label is not None: args["label"] = label if schema is not None: diff --git a/packages/coding-agent/src/prompts/system/subagent-system-prompt.md b/packages/coding-agent/src/prompts/system/subagent-system-prompt.md index a7a25dad0..d6f4345cd 100644 --- a/packages/coding-agent/src/prompts/system/subagent-system-prompt.md +++ b/packages/coding-agent/src/prompts/system/subagent-system-prompt.md @@ -3,13 +3,6 @@ ROLE {{agent}} -{{#if context}} -CONTEXT -=================================== - -{{context}} -{{/if}} - {{#if planReference}} PLAN =================================== @@ -32,11 +25,6 @@ You are working in an isolated working tree at `{{worktree}}` for this sub-task. You NEVER modify files outside this tree or in the original repository. {{/if}} -{{#if contextFile}} -# Conversation Context -If you need additional information, your conversation with the user is in {{contextFile}} — `read` its tail or `search` it for relevant terms. -{{/if}} - {{#if ircPeers}} # IRC Peers You can reach other live agents via the `irc` tool. Your id is `{{ircSelfId}}`. Currently visible peers: diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index 73085ec6e..a8e1c6f55 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -13,8 +13,8 @@ Worth it when the task benefits from decomposition + parallel coverage, or from State persists across cells, so scout in one cell and fan out in the next. Every cell has: -- `agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent_type` picks a discovered agent ("explore", "reviewer", "oracle", …); `context` is shared background; `label` names the artifact. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. -- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool runs as wide as a `task` tool batch — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one. +- `agent(prompt, *, agent_type="task", model=None, label=None, schema=None)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent_type` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. +- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one. - `pipeline(items, *stages)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. Same pool width as `parallel()`. - `completion(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out. - `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it. diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index 8e99ffc0f..52f25f101 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -46,9 +46,9 @@ tool.(args) → unknown Invoke any session tool by name. `args` is the tool's parameter object. completion(prompt, model?="default", system?=None, schema?=None) → str | dict Oneshot, stateless completion (no history, no tools). `model` picks a tier: "smol" (fast), "default" (this session's model), "slow" (most capable). Pass `system` for a system prompt. Pass a JSON-Schema `schema` to force structured output and get the parsed object back; otherwise returns the completion text. -{{#if spawns}}agent(prompt, agent_type?="task", model?=None, context?=None, label?=None, schema?=None) → str | dict - Run a subagent and return its final output. Defaults to the bundled "task" agent; pass `agent_type`/`agentType` for another discovered agent. Pass a JSON-Schema `schema` to force structured output and get the parsed object back. -{{#if js}} In JS, pass options as one trailing object — never positional: agent(prompt, { agentType, context, schema }). +{{#if spawns}}agent(prompt, agent_type?="task", model?=None, label?=None, schema?=None) → str | dict + Run a subagent and return its final output. Defaults to the bundled "task" agent; pass `agent_type`/`agentType` for another discovered agent. Pass a JSON-Schema `schema` to force structured output and get the parsed object back. Share background by writing a `local://` file and referencing it in the prompt. +{{#if js}} In JS, pass options as one trailing object — never positional: agent(prompt, { agentType, schema }). {{/if}} {{/if}} parallel(thunks) → list diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index def9a1629..022e34867 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -231,10 +231,8 @@ import type { AuthStorage } from "./auth-storage"; import type { ClientBridge, ClientBridgePermissionOption, ClientBridgePermissionOutcome } from "./client-bridge"; import { type BashExecutionMessage, - type CompactionSummaryMessage, type CustomMessage, convertToLlm, - type FileMentionMessage, type PythonExecutionMessage, readPendingDisplayTag, SILENT_ABORT_MARKER, @@ -9960,69 +9958,6 @@ export class AgentSession { }); } - /** - * Format the conversation as compact context for subagents. - * Includes only user messages and assistant text responses. - * Excludes: system prompt, tool definitions, tool calls/results, thinking blocks. - */ - formatCompactContext(): string { - const lines: string[] = []; - lines.push("# Conversation Context"); - lines.push(""); - lines.push( - "This is a summary of the parent conversation. Read this if you need additional context about what was discussed or decided.", - ); - lines.push(""); - - for (const msg of this.messages) { - if (msg.role === "user" || msg.role === "developer") { - lines.push(msg.role === "developer" ? "## Developer" : "## User"); - lines.push(""); - if (typeof msg.content === "string") { - lines.push(msg.content); - } else { - for (const c of msg.content) { - if (c.type === "text") { - lines.push(c.text); - } else if (c.type === "image") { - lines.push("[Image attached]"); - } - } - } - lines.push(""); - } else if (msg.role === "assistant") { - const assistantMsg = msg as AssistantMessage; - // Only include text content, skip tool calls and thinking - const textParts: string[] = []; - for (const c of assistantMsg.content) { - if (c.type === "text" && c.text.trim()) { - textParts.push(c.text); - } - } - if (textParts.length > 0) { - lines.push("## Assistant"); - lines.push(""); - lines.push(textParts.join("\n\n")); - lines.push(""); - } - } else if (msg.role === "fileMention") { - const fileMsg = msg as FileMentionMessage; - const paths = fileMsg.files.map(f => f.path).join(", "); - lines.push(`[Files referenced: ${paths}]`); - lines.push(""); - } else if (msg.role === "compactionSummary") { - const compactMsg = msg as CompactionSummaryMessage; - lines.push("## Earlier Context (Summarized)"); - lines.push(""); - lines.push(compactMsg.summary); - lines.push(""); - } - // Skip: toolResult, bashExecution, pythonExecution, branchSummary, custom, hookMessage - } - - return lines.join("\n").trim(); - } - // ========================================================================= // Extension System // ========================================================================= diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index c524ca646..c70a022bf 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -260,8 +260,6 @@ export interface ToolSession { recordEvalSubagentUsage?: (output: number) => void; /** Bridge to the connected client (e.g. ACP editor host). Tools should route fs/terminal/permission requests through this when available. */ getClientBridge?: () => ClientBridge | undefined; - /** Get compact conversation context for subagents (excludes tool results, system prompts) */ - getCompactContext?: () => string; /** Get cached todo phases for this session. */ getTodoPhases?: () => TodoPhase[]; /** Replace cached todo phases for this session. */