feat(coding-agent): removed context argument from eval agent() spawn

Shared background now flows through a '/Users/can/.omp/agent/sessions/-Projects-.tree-pi-commit/2026-06-10T15-36-32-782Z_019eb22d-970e-7000-8964-72c98becf3e8/local' file referenced in each prompt instead of a context string forwarded into the subagent's system prompt. The JS and Python preludes drop the context kwarg from agent(), the subagent system prompt drops the {{#if context}} block and the conversation-context file pointer, and runEvalAgent no longer writes a per-call conversation context file. AgentSession sheds the now-unused formatCompactContext() helper that supplied the file's body, and ToolSession.getCompactContext is removed alongside it.
This commit is contained in:
can1357
2026-06-10 17:49:01 +02:00
parent fcb8663de8
commit a92d2ce989
10 changed files with 19 additions and 113 deletions
+4 -4
View File
@@ -138,7 +138,7 @@ Implemented in `packages/coding-agent/src/eval/js/worker-core.ts`, `packages/cod
- `await read(path, { offset?, limit? })`
- `await tree(path = ".", { maxDepth?, hidden? })`
- `sort(text, { reverse?, unique? })`, `uniq(text, { count? })`, `counter(items, { limit?, reverse? })`
- `await agent(prompt, { agentType?, model?, context?, label?, schema? })`
- `await agent(prompt, { agentType?, model?, label?, schema? })`
- `await parallel([() => agent("a"), () => agent("b")])`
- `await pipeline(items, stage1, stage2)`
- `display(value)` behavior:
@@ -192,11 +192,11 @@ Both runtimes expose `completion()` — a single stateless completion against a
Both runtimes expose `agent()` — a single subagent invocation routed through `packages/coding-agent/src/eval/agent-bridge.ts` into the same `runSubprocess(...)` path used by the `task` tool. It uses the current eval session's spawn policy and inherits the parent eval executor id, so parent and subagent code share JS/Python runtime state.
- Signatures:
- JS: `await agent(prompt, { agentType?, model?, context?, label?, schema? })`
- Python: `agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None)`
- JS: `await agent(prompt, { agentType?, model?, label?, schema? })`
- Python: `agent(prompt, *, agent_type="task", model=None, label=None, schema=None)`
- `agentType` / `agent_type` defaults to the bundled `task` agent and resolves through normal agent discovery, so project and user agents work.
- `model` overrides the selected agent's model. Without it, normal per-agent settings and the agent frontmatter model apply.
- `context` supplies shared background; `label` controls the `agent://<id>` output label prefix.
- Shared background is passed via files: write a `local://` file and reference it in the prompt. `label` controls the `agent://<id>` output label prefix.
- `schema` passes a JSON Schema to the subagent structured-output path. When present, the helper parses the final JSON text and returns an object.
- Spawn restrictions use `session.getSessionSpawns()` exactly like the `task` tool. Eval-driven subagent recursion is capped at depth 3.
- JS and Python both expose `parallel(thunks)` and `pipeline(items, ...stages)`; both use a bounded async/threaded pool whose width tracks the `task.maxConcurrency` setting (the same ceiling the `task` tool uses; `0` = run every item at once), preserve item order, and propagate rejections. The width is fetched live from the host via the `__concurrency__` bridge, so the helpers no longer take a `concurrency` argument.
@@ -178,7 +178,7 @@ describe("runEvalAgent", () => {
expect(runSpy).not.toHaveBeenCalled();
});
it("passes the parent execution context and only sets outputSchema when schema is supplied", async () => {
it("passes parent execution options and only sets outputSchema when schema is supplied", async () => {
mockAgents();
const runSpy = vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => singleResult(options));
const abortController = new AbortController();
@@ -186,7 +186,7 @@ describe("runEvalAgent", () => {
const session = makeSession({ depth: 2, activeModel: "p/current", modelString: "p/fallback" });
await runEvalAgent(
{ prompt: " hello ", context: " context ", label: "My Agent", model: "p/override", schema },
{ prompt: " hello ", label: "My Agent", model: "p/override", schema },
{ session, signal: abortController.signal },
);
await runEvalAgent({ prompt: "plain" }, { session });
@@ -199,7 +199,6 @@ describe("runEvalAgent", () => {
expect(firstOptions.parentActiveModelPattern).toBe("p/current");
expect(firstOptions.outputSchema).toBe(schema);
expect(firstOptions.assignment).toBe("hello");
expect(firstOptions.context).toBe("context");
expect(firstOptions.description).toBe("My Agent");
expect(firstOptions.modelOverride).toEqual(["p/override"]);
expect(secondOptions.outputSchema).toBeUndefined();
+2 -15
View File
@@ -34,7 +34,6 @@ const agentArgsSchema = z.object({
prompt: z.string().min(1, "prompt must be a non-empty string"),
agentType: z.string().min(1).optional(),
model: z.union([z.string().min(1), z.array(z.string().min(1)).min(1)]).optional(),
context: z.string().optional(),
label: z.string().optional(),
schema: z.unknown().optional(),
});
@@ -43,7 +42,6 @@ interface EvalAgentArgs {
prompt: string;
agentType?: string;
model?: string | string[];
context?: string;
label?: string;
schema?: unknown;
}
@@ -135,20 +133,12 @@ function getOutputManager(session: ToolSession): AgentOutputManager {
async function getArtifacts(session: ToolSession): Promise<{
sessionFile: string | null;
artifactsDir: string;
contextFile?: string;
}> {
const sessionFile = session.getSessionFile();
const sessionArtifactsDir = sessionFile ? sessionFile.slice(0, -6) : null;
const artifactsDir = sessionArtifactsDir ?? path.join(os.tmpdir(), `omp-eval-agent-${Snowflake.next()}`);
await fs.mkdir(artifactsDir, { recursive: true });
const shouldWriteConversationContext = session.settings.get("irc.enabled") !== true;
const compactContext = shouldWriteConversationContext ? session.getCompactContext?.() : undefined;
if (!compactContext) return { sessionFile, artifactsDir };
const contextFile = path.join(artifactsDir, "context.md");
await Bun.write(contextFile, compactContext);
return { sessionFile, artifactsDir, contextFile };
return { sessionFile, artifactsDir };
}
function emitProgressStatus(emitStatus: ((event: JsStatusEvent) => void) | undefined, progress: AgentProgress): void {
@@ -246,11 +236,10 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
};
const parentArtifactManager = options.session.getArtifactManager?.() ?? undefined;
const mcpManager = options.session.mcpManager ?? MCPManager.instance();
const { sessionFile, artifactsDir, contextFile } = await getArtifacts(options.session);
const { sessionFile, artifactsDir } = await getArtifacts(options.session);
const outputManager = getOutputManager(options.session);
const id = await outputManager.allocate(outputIdBase(parsed.label, agentName));
const assignment = parsed.prompt.trim();
const context = trimToUndefined(parsed.context);
// Suspend eval timeout accounting while the subagent owns control. The
// timeout clock restarts once the bridge returns to the cell runtime.
const result = await withBridgeTimeoutPause(options.emitStatus, () =>
@@ -259,7 +248,6 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
agent: effectiveAgent,
task: renderSubagentPrompt(assignment),
assignment,
context,
description: trimToUndefined(parsed.label),
index: 0,
id,
@@ -271,7 +259,6 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
sessionFile,
persistArtifacts: Boolean(sessionFile),
artifactsDir,
contextFile,
// Eval `agent()` subagents are short-lived programmatic helpers (data
// collection, structured output, parallel() fan-out). LSP server
// cold-start costs tens of seconds and is pure overhead here, so it is
@@ -65,7 +65,7 @@ if (!globalThis.__omp_js_prelude_loaded__) {
};
const agent = async (prompt, opts, ...rest) => {
const o = optionsArg("agent", opts, rest, "{ agentType, model, context, label, schema }");
const o = optionsArg("agent", opts, rest, "{ agentType, model, label, schema }");
const res = await globalThis.__omp_call_tool__("__agent__", { prompt, ...o });
const text = res && typeof res === "object" ? res.text : res;
return hasOwn(o, "schema") ? JSON.parse(text) : text;
+5 -6
View File
@@ -519,21 +519,20 @@ if "__omp_prelude_loaded__" not in globals():
text = res.get("text") if isinstance(res, dict) else res
return json.loads(text) if schema is not None else text
def agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None):
def agent(prompt, *, agent_type="task", model=None, label=None, schema=None):
"""Run a subagent and return its final output.
`agent_type` selects the subagent definition (default "task"). Pass
`model` to override that agent's model, `context` for shared background,
`label` for the output artifact id, and `schema` to request structured
JSON output; when `schema` is supplied the parsed object is returned.
`model` to override that agent's model, `label` for the output artifact
id, and `schema` to request structured JSON output; when `schema` is
supplied the parsed object is returned. Share background by writing a
local:// file and referencing it in the prompt.
"""
args = {"prompt": prompt}
if agent_type is not None:
args["agentType"] = agent_type
if model is not None:
args["model"] = model
if context is not None:
args["context"] = context
if label is not None:
args["label"] = label
if schema is not None:
@@ -3,13 +3,6 @@ ROLE
{{agent}}
{{#if context}}
CONTEXT
===================================
{{context}}
{{/if}}
{{#if planReference}}
PLAN
===================================
@@ -32,11 +25,6 @@ You are working in an isolated working tree at `{{worktree}}` for this sub-task.
You NEVER modify files outside this tree or in the original repository.
{{/if}}
{{#if contextFile}}
# Conversation Context
If you need additional information, your conversation with the user is in {{contextFile}} — `read` its tail or `search` it for relevant terms.
{{/if}}
{{#if ircPeers}}
# IRC Peers
You can reach other live agents via the `irc` tool. Your id is `{{ircSelfId}}`. Currently visible peers:
@@ -13,8 +13,8 @@ Worth it when the task benefits from decomposition + parallel coverage, or from
<helpers>
State persists across cells, so scout in one cell and fan out in the next. Every cell has:
- `agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent_type` picks a discovered agent ("explore", "reviewer", "oracle", …); `context` is shared background; `label` names the artifact. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep.
- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool runs as wide as a `task` tool batch — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one.
- `agent(prompt, *, agent_type="task", model=None, label=None, schema=None)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent_type` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep.
- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one.
- `pipeline(items, *stages)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. Same pool width as `parallel()`.
- `completion(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out.
- `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it.
@@ -46,9 +46,9 @@ tool.<name>(args) → unknown
Invoke any session tool by name. `args` is the tool's parameter object.
completion(prompt, model?="default", system?=None, schema?=None) → str | dict
Oneshot, stateless completion (no history, no tools). `model` picks a tier: "smol" (fast), "default" (this session's model), "slow" (most capable). Pass `system` for a system prompt. Pass a JSON-Schema `schema` to force structured output and get the parsed object back; otherwise returns the completion text.
{{#if spawns}}agent(prompt, agent_type?="task", model?=None, context?=None, label?=None, schema?=None) → str | dict
Run a subagent and return its final output. Defaults to the bundled "task" agent; pass `agent_type`/`agentType` for another discovered agent. Pass a JSON-Schema `schema` to force structured output and get the parsed object back.
{{#if js}} In JS, pass options as one trailing object — never positional: agent(prompt, { agentType, context, schema }).
{{#if spawns}}agent(prompt, agent_type?="task", model?=None, label?=None, schema?=None) → str | dict
Run a subagent and return its final output. Defaults to the bundled "task" agent; pass `agent_type`/`agentType` for another discovered agent. Pass a JSON-Schema `schema` to force structured output and get the parsed object back. Share background by writing a `local://` file and referencing it in the prompt.
{{#if js}} In JS, pass options as one trailing object — never positional: agent(prompt, { agentType, schema }).
{{/if}}
{{/if}}
parallel(thunks) → list
@@ -231,10 +231,8 @@ import type { AuthStorage } from "./auth-storage";
import type { ClientBridge, ClientBridgePermissionOption, ClientBridgePermissionOutcome } from "./client-bridge";
import {
type BashExecutionMessage,
type CompactionSummaryMessage,
type CustomMessage,
convertToLlm,
type FileMentionMessage,
type PythonExecutionMessage,
readPendingDisplayTag,
SILENT_ABORT_MARKER,
@@ -9960,69 +9958,6 @@ export class AgentSession {
});
}
/**
* Format the conversation as compact context for subagents.
* Includes only user messages and assistant text responses.
* Excludes: system prompt, tool definitions, tool calls/results, thinking blocks.
*/
formatCompactContext(): string {
const lines: string[] = [];
lines.push("# Conversation Context");
lines.push("");
lines.push(
"This is a summary of the parent conversation. Read this if you need additional context about what was discussed or decided.",
);
lines.push("");
for (const msg of this.messages) {
if (msg.role === "user" || msg.role === "developer") {
lines.push(msg.role === "developer" ? "## Developer" : "## User");
lines.push("");
if (typeof msg.content === "string") {
lines.push(msg.content);
} else {
for (const c of msg.content) {
if (c.type === "text") {
lines.push(c.text);
} else if (c.type === "image") {
lines.push("[Image attached]");
}
}
}
lines.push("");
} else if (msg.role === "assistant") {
const assistantMsg = msg as AssistantMessage;
// Only include text content, skip tool calls and thinking
const textParts: string[] = [];
for (const c of assistantMsg.content) {
if (c.type === "text" && c.text.trim()) {
textParts.push(c.text);
}
}
if (textParts.length > 0) {
lines.push("## Assistant");
lines.push("");
lines.push(textParts.join("\n\n"));
lines.push("");
}
} else if (msg.role === "fileMention") {
const fileMsg = msg as FileMentionMessage;
const paths = fileMsg.files.map(f => f.path).join(", ");
lines.push(`[Files referenced: ${paths}]`);
lines.push("");
} else if (msg.role === "compactionSummary") {
const compactMsg = msg as CompactionSummaryMessage;
lines.push("## Earlier Context (Summarized)");
lines.push("");
lines.push(compactMsg.summary);
lines.push("");
}
// Skip: toolResult, bashExecution, pythonExecution, branchSummary, custom, hookMessage
}
return lines.join("\n").trim();
}
// =========================================================================
// Extension System
// =========================================================================
-2
View File
@@ -260,8 +260,6 @@ export interface ToolSession {
recordEvalSubagentUsage?: (output: number) => void;
/** Bridge to the connected client (e.g. ACP editor host). Tools should route fs/terminal/permission requests through this when available. */
getClientBridge?: () => ClientBridge | undefined;
/** Get compact conversation context for subagents (excludes tool results, system prompts) */
getCompactContext?: () => string;
/** Get cached todo phases for this session. */
getTodoPhases?: () => TodoPhase[];
/** Replace cached todo phases for this session. */