feat(coding-agent): removed context argument from eval agent() spawn
Shared background now flows through a '/Users/can/.omp/agent/sessions/-Projects-.tree-pi-commit/2026-06-10T15-36-32-782Z_019eb22d-970e-7000-8964-72c98becf3e8/local' file referenced in each prompt instead of a context string forwarded into the subagent's system prompt. The JS and Python preludes drop the context kwarg from agent(), the subagent system prompt drops the {{#if context}} block and the conversation-context file pointer, and runEvalAgent no longer writes a per-call conversation context file. AgentSession sheds the now-unused formatCompactContext() helper that supplied the file's body, and ToolSession.getCompactContext is removed alongside it.
This commit is contained in:
+4
-4
@@ -138,7 +138,7 @@ Implemented in `packages/coding-agent/src/eval/js/worker-core.ts`, `packages/cod
|
||||
- `await read(path, { offset?, limit? })`
|
||||
- `await tree(path = ".", { maxDepth?, hidden? })`
|
||||
- `sort(text, { reverse?, unique? })`, `uniq(text, { count? })`, `counter(items, { limit?, reverse? })`
|
||||
- `await agent(prompt, { agentType?, model?, context?, label?, schema? })`
|
||||
- `await agent(prompt, { agentType?, model?, label?, schema? })`
|
||||
- `await parallel([() => agent("a"), () => agent("b")])`
|
||||
- `await pipeline(items, stage1, stage2)`
|
||||
- `display(value)` behavior:
|
||||
@@ -192,11 +192,11 @@ Both runtimes expose `completion()` — a single stateless completion against a
|
||||
Both runtimes expose `agent()` — a single subagent invocation routed through `packages/coding-agent/src/eval/agent-bridge.ts` into the same `runSubprocess(...)` path used by the `task` tool. It uses the current eval session's spawn policy and inherits the parent eval executor id, so parent and subagent code share JS/Python runtime state.
|
||||
|
||||
- Signatures:
|
||||
- JS: `await agent(prompt, { agentType?, model?, context?, label?, schema? })`
|
||||
- Python: `agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None)`
|
||||
- JS: `await agent(prompt, { agentType?, model?, label?, schema? })`
|
||||
- Python: `agent(prompt, *, agent_type="task", model=None, label=None, schema=None)`
|
||||
- `agentType` / `agent_type` defaults to the bundled `task` agent and resolves through normal agent discovery, so project and user agents work.
|
||||
- `model` overrides the selected agent's model. Without it, normal per-agent settings and the agent frontmatter model apply.
|
||||
- `context` supplies shared background; `label` controls the `agent://<id>` output label prefix.
|
||||
- Shared background is passed via files: write a `local://` file and reference it in the prompt. `label` controls the `agent://<id>` output label prefix.
|
||||
- `schema` passes a JSON Schema to the subagent structured-output path. When present, the helper parses the final JSON text and returns an object.
|
||||
- Spawn restrictions use `session.getSessionSpawns()` exactly like the `task` tool. Eval-driven subagent recursion is capped at depth 3.
|
||||
- JS and Python both expose `parallel(thunks)` and `pipeline(items, ...stages)`; both use a bounded async/threaded pool whose width tracks the `task.maxConcurrency` setting (the same ceiling the `task` tool uses; `0` = run every item at once), preserve item order, and propagate rejections. The width is fetched live from the host via the `__concurrency__` bridge, so the helpers no longer take a `concurrency` argument.
|
||||
|
||||
@@ -178,7 +178,7 @@ describe("runEvalAgent", () => {
|
||||
expect(runSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("passes the parent execution context and only sets outputSchema when schema is supplied", async () => {
|
||||
it("passes parent execution options and only sets outputSchema when schema is supplied", async () => {
|
||||
mockAgents();
|
||||
const runSpy = vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => singleResult(options));
|
||||
const abortController = new AbortController();
|
||||
@@ -186,7 +186,7 @@ describe("runEvalAgent", () => {
|
||||
const session = makeSession({ depth: 2, activeModel: "p/current", modelString: "p/fallback" });
|
||||
|
||||
await runEvalAgent(
|
||||
{ prompt: " hello ", context: " context ", label: "My Agent", model: "p/override", schema },
|
||||
{ prompt: " hello ", label: "My Agent", model: "p/override", schema },
|
||||
{ session, signal: abortController.signal },
|
||||
);
|
||||
await runEvalAgent({ prompt: "plain" }, { session });
|
||||
@@ -199,7 +199,6 @@ describe("runEvalAgent", () => {
|
||||
expect(firstOptions.parentActiveModelPattern).toBe("p/current");
|
||||
expect(firstOptions.outputSchema).toBe(schema);
|
||||
expect(firstOptions.assignment).toBe("hello");
|
||||
expect(firstOptions.context).toBe("context");
|
||||
expect(firstOptions.description).toBe("My Agent");
|
||||
expect(firstOptions.modelOverride).toEqual(["p/override"]);
|
||||
expect(secondOptions.outputSchema).toBeUndefined();
|
||||
|
||||
@@ -34,7 +34,6 @@ const agentArgsSchema = z.object({
|
||||
prompt: z.string().min(1, "prompt must be a non-empty string"),
|
||||
agentType: z.string().min(1).optional(),
|
||||
model: z.union([z.string().min(1), z.array(z.string().min(1)).min(1)]).optional(),
|
||||
context: z.string().optional(),
|
||||
label: z.string().optional(),
|
||||
schema: z.unknown().optional(),
|
||||
});
|
||||
@@ -43,7 +42,6 @@ interface EvalAgentArgs {
|
||||
prompt: string;
|
||||
agentType?: string;
|
||||
model?: string | string[];
|
||||
context?: string;
|
||||
label?: string;
|
||||
schema?: unknown;
|
||||
}
|
||||
@@ -135,20 +133,12 @@ function getOutputManager(session: ToolSession): AgentOutputManager {
|
||||
async function getArtifacts(session: ToolSession): Promise<{
|
||||
sessionFile: string | null;
|
||||
artifactsDir: string;
|
||||
contextFile?: string;
|
||||
}> {
|
||||
const sessionFile = session.getSessionFile();
|
||||
const sessionArtifactsDir = sessionFile ? sessionFile.slice(0, -6) : null;
|
||||
const artifactsDir = sessionArtifactsDir ?? path.join(os.tmpdir(), `omp-eval-agent-${Snowflake.next()}`);
|
||||
await fs.mkdir(artifactsDir, { recursive: true });
|
||||
|
||||
const shouldWriteConversationContext = session.settings.get("irc.enabled") !== true;
|
||||
const compactContext = shouldWriteConversationContext ? session.getCompactContext?.() : undefined;
|
||||
if (!compactContext) return { sessionFile, artifactsDir };
|
||||
|
||||
const contextFile = path.join(artifactsDir, "context.md");
|
||||
await Bun.write(contextFile, compactContext);
|
||||
return { sessionFile, artifactsDir, contextFile };
|
||||
return { sessionFile, artifactsDir };
|
||||
}
|
||||
|
||||
function emitProgressStatus(emitStatus: ((event: JsStatusEvent) => void) | undefined, progress: AgentProgress): void {
|
||||
@@ -246,11 +236,10 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
|
||||
};
|
||||
const parentArtifactManager = options.session.getArtifactManager?.() ?? undefined;
|
||||
const mcpManager = options.session.mcpManager ?? MCPManager.instance();
|
||||
const { sessionFile, artifactsDir, contextFile } = await getArtifacts(options.session);
|
||||
const { sessionFile, artifactsDir } = await getArtifacts(options.session);
|
||||
const outputManager = getOutputManager(options.session);
|
||||
const id = await outputManager.allocate(outputIdBase(parsed.label, agentName));
|
||||
const assignment = parsed.prompt.trim();
|
||||
const context = trimToUndefined(parsed.context);
|
||||
// Suspend eval timeout accounting while the subagent owns control. The
|
||||
// timeout clock restarts once the bridge returns to the cell runtime.
|
||||
const result = await withBridgeTimeoutPause(options.emitStatus, () =>
|
||||
@@ -259,7 +248,6 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
|
||||
agent: effectiveAgent,
|
||||
task: renderSubagentPrompt(assignment),
|
||||
assignment,
|
||||
context,
|
||||
description: trimToUndefined(parsed.label),
|
||||
index: 0,
|
||||
id,
|
||||
@@ -271,7 +259,6 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
|
||||
sessionFile,
|
||||
persistArtifacts: Boolean(sessionFile),
|
||||
artifactsDir,
|
||||
contextFile,
|
||||
// Eval `agent()` subagents are short-lived programmatic helpers (data
|
||||
// collection, structured output, parallel() fan-out). LSP server
|
||||
// cold-start costs tens of seconds and is pure overhead here, so it is
|
||||
|
||||
@@ -65,7 +65,7 @@ if (!globalThis.__omp_js_prelude_loaded__) {
|
||||
};
|
||||
|
||||
const agent = async (prompt, opts, ...rest) => {
|
||||
const o = optionsArg("agent", opts, rest, "{ agentType, model, context, label, schema }");
|
||||
const o = optionsArg("agent", opts, rest, "{ agentType, model, label, schema }");
|
||||
const res = await globalThis.__omp_call_tool__("__agent__", { prompt, ...o });
|
||||
const text = res && typeof res === "object" ? res.text : res;
|
||||
return hasOwn(o, "schema") ? JSON.parse(text) : text;
|
||||
|
||||
@@ -519,21 +519,20 @@ if "__omp_prelude_loaded__" not in globals():
|
||||
text = res.get("text") if isinstance(res, dict) else res
|
||||
return json.loads(text) if schema is not None else text
|
||||
|
||||
def agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None):
|
||||
def agent(prompt, *, agent_type="task", model=None, label=None, schema=None):
|
||||
"""Run a subagent and return its final output.
|
||||
|
||||
`agent_type` selects the subagent definition (default "task"). Pass
|
||||
`model` to override that agent's model, `context` for shared background,
|
||||
`label` for the output artifact id, and `schema` to request structured
|
||||
JSON output; when `schema` is supplied the parsed object is returned.
|
||||
`model` to override that agent's model, `label` for the output artifact
|
||||
id, and `schema` to request structured JSON output; when `schema` is
|
||||
supplied the parsed object is returned. Share background by writing a
|
||||
local:// file and referencing it in the prompt.
|
||||
"""
|
||||
args = {"prompt": prompt}
|
||||
if agent_type is not None:
|
||||
args["agentType"] = agent_type
|
||||
if model is not None:
|
||||
args["model"] = model
|
||||
if context is not None:
|
||||
args["context"] = context
|
||||
if label is not None:
|
||||
args["label"] = label
|
||||
if schema is not None:
|
||||
|
||||
@@ -3,13 +3,6 @@ ROLE
|
||||
|
||||
{{agent}}
|
||||
|
||||
{{#if context}}
|
||||
CONTEXT
|
||||
===================================
|
||||
|
||||
{{context}}
|
||||
{{/if}}
|
||||
|
||||
{{#if planReference}}
|
||||
PLAN
|
||||
===================================
|
||||
@@ -32,11 +25,6 @@ You are working in an isolated working tree at `{{worktree}}` for this sub-task.
|
||||
You NEVER modify files outside this tree or in the original repository.
|
||||
{{/if}}
|
||||
|
||||
{{#if contextFile}}
|
||||
# Conversation Context
|
||||
If you need additional information, your conversation with the user is in {{contextFile}} — `read` its tail or `search` it for relevant terms.
|
||||
{{/if}}
|
||||
|
||||
{{#if ircPeers}}
|
||||
# IRC Peers
|
||||
You can reach other live agents via the `irc` tool. Your id is `{{ircSelfId}}`. Currently visible peers:
|
||||
|
||||
@@ -13,8 +13,8 @@ Worth it when the task benefits from decomposition + parallel coverage, or from
|
||||
<helpers>
|
||||
State persists across cells, so scout in one cell and fan out in the next. Every cell has:
|
||||
|
||||
- `agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent_type` picks a discovered agent ("explore", "reviewer", "oracle", …); `context` is shared background; `label` names the artifact. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep.
|
||||
- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool runs as wide as a `task` tool batch — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one.
|
||||
- `agent(prompt, *, agent_type="task", model=None, label=None, schema=None)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent_type` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep.
|
||||
- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one.
|
||||
- `pipeline(items, *stages)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. Same pool width as `parallel()`.
|
||||
- `completion(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out.
|
||||
- `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it.
|
||||
|
||||
@@ -46,9 +46,9 @@ tool.<name>(args) → unknown
|
||||
Invoke any session tool by name. `args` is the tool's parameter object.
|
||||
completion(prompt, model?="default", system?=None, schema?=None) → str | dict
|
||||
Oneshot, stateless completion (no history, no tools). `model` picks a tier: "smol" (fast), "default" (this session's model), "slow" (most capable). Pass `system` for a system prompt. Pass a JSON-Schema `schema` to force structured output and get the parsed object back; otherwise returns the completion text.
|
||||
{{#if spawns}}agent(prompt, agent_type?="task", model?=None, context?=None, label?=None, schema?=None) → str | dict
|
||||
Run a subagent and return its final output. Defaults to the bundled "task" agent; pass `agent_type`/`agentType` for another discovered agent. Pass a JSON-Schema `schema` to force structured output and get the parsed object back.
|
||||
{{#if js}} In JS, pass options as one trailing object — never positional: agent(prompt, { agentType, context, schema }).
|
||||
{{#if spawns}}agent(prompt, agent_type?="task", model?=None, label?=None, schema?=None) → str | dict
|
||||
Run a subagent and return its final output. Defaults to the bundled "task" agent; pass `agent_type`/`agentType` for another discovered agent. Pass a JSON-Schema `schema` to force structured output and get the parsed object back. Share background by writing a `local://` file and referencing it in the prompt.
|
||||
{{#if js}} In JS, pass options as one trailing object — never positional: agent(prompt, { agentType, schema }).
|
||||
{{/if}}
|
||||
{{/if}}
|
||||
parallel(thunks) → list
|
||||
|
||||
@@ -231,10 +231,8 @@ import type { AuthStorage } from "./auth-storage";
|
||||
import type { ClientBridge, ClientBridgePermissionOption, ClientBridgePermissionOutcome } from "./client-bridge";
|
||||
import {
|
||||
type BashExecutionMessage,
|
||||
type CompactionSummaryMessage,
|
||||
type CustomMessage,
|
||||
convertToLlm,
|
||||
type FileMentionMessage,
|
||||
type PythonExecutionMessage,
|
||||
readPendingDisplayTag,
|
||||
SILENT_ABORT_MARKER,
|
||||
@@ -9960,69 +9958,6 @@ export class AgentSession {
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Format the conversation as compact context for subagents.
|
||||
* Includes only user messages and assistant text responses.
|
||||
* Excludes: system prompt, tool definitions, tool calls/results, thinking blocks.
|
||||
*/
|
||||
formatCompactContext(): string {
|
||||
const lines: string[] = [];
|
||||
lines.push("# Conversation Context");
|
||||
lines.push("");
|
||||
lines.push(
|
||||
"This is a summary of the parent conversation. Read this if you need additional context about what was discussed or decided.",
|
||||
);
|
||||
lines.push("");
|
||||
|
||||
for (const msg of this.messages) {
|
||||
if (msg.role === "user" || msg.role === "developer") {
|
||||
lines.push(msg.role === "developer" ? "## Developer" : "## User");
|
||||
lines.push("");
|
||||
if (typeof msg.content === "string") {
|
||||
lines.push(msg.content);
|
||||
} else {
|
||||
for (const c of msg.content) {
|
||||
if (c.type === "text") {
|
||||
lines.push(c.text);
|
||||
} else if (c.type === "image") {
|
||||
lines.push("[Image attached]");
|
||||
}
|
||||
}
|
||||
}
|
||||
lines.push("");
|
||||
} else if (msg.role === "assistant") {
|
||||
const assistantMsg = msg as AssistantMessage;
|
||||
// Only include text content, skip tool calls and thinking
|
||||
const textParts: string[] = [];
|
||||
for (const c of assistantMsg.content) {
|
||||
if (c.type === "text" && c.text.trim()) {
|
||||
textParts.push(c.text);
|
||||
}
|
||||
}
|
||||
if (textParts.length > 0) {
|
||||
lines.push("## Assistant");
|
||||
lines.push("");
|
||||
lines.push(textParts.join("\n\n"));
|
||||
lines.push("");
|
||||
}
|
||||
} else if (msg.role === "fileMention") {
|
||||
const fileMsg = msg as FileMentionMessage;
|
||||
const paths = fileMsg.files.map(f => f.path).join(", ");
|
||||
lines.push(`[Files referenced: ${paths}]`);
|
||||
lines.push("");
|
||||
} else if (msg.role === "compactionSummary") {
|
||||
const compactMsg = msg as CompactionSummaryMessage;
|
||||
lines.push("## Earlier Context (Summarized)");
|
||||
lines.push("");
|
||||
lines.push(compactMsg.summary);
|
||||
lines.push("");
|
||||
}
|
||||
// Skip: toolResult, bashExecution, pythonExecution, branchSummary, custom, hookMessage
|
||||
}
|
||||
|
||||
return lines.join("\n").trim();
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Extension System
|
||||
// =========================================================================
|
||||
|
||||
@@ -260,8 +260,6 @@ export interface ToolSession {
|
||||
recordEvalSubagentUsage?: (output: number) => void;
|
||||
/** Bridge to the connected client (e.g. ACP editor host). Tools should route fs/terminal/permission requests through this when available. */
|
||||
getClientBridge?: () => ClientBridge | undefined;
|
||||
/** Get compact conversation context for subagents (excludes tool results, system prompts) */
|
||||
getCompactContext?: () => string;
|
||||
/** Get cached todo phases for this session. */
|
||||
getTodoPhases?: () => TodoPhase[];
|
||||
/** Replace cached todo phases for this session. */
|
||||
|
||||
Reference in New Issue
Block a user