Files
oh-my-pi/packages/coding-agent/test/eval/agent-bridge.test.ts
T
can1357 9d99ae1af0 feat(coding-agent): rewrote the task tool to spawn one persistent subagent per call
The task tool now takes a single { agent, assignment, description, ... } and always runs the subagent in the background — the batch tasks[] array and shared context parameter are gone. Fan-out is parallel task calls; shared background flows through a '/Users/can/.omp/agent/sessions/-Projects-.tree-pi-commit/2026-06-10T15-36-32-782Z_019eb22d-970e-7000-8964-72c98becf3e8/local' file referenced in each assignment.\n\nIntroduces a persistent subagent lifecycle: finished subagents stay live as idle, the lifecycle manager parks them to disk after task.agentIdleTtlMs (default 7 minutes; 0 keeps them live until exit), and they revive automatically when prompted from the Agent Hub, messaged on IRC, or resumed via task. New task(resume: "<id>") revives an idle or parked subagent and runs a follow-up assignment in its existing session.\n\nAdds soft request budgets (explore/quick_task 40, others 90, configurable via task.softRequestBudget, 0 disables): crossing the budget injects a one-time wrap-up steer into the child; crossing 1.5× aborts the run gracefully. Cancelled/aborted subagent salvage replaces the old (no output) with the child's last activity snippet plus request/token stats; SingleResult tracks a per-child requests counter (assistant message_end events) used to sort agent lists in runtime-ascending order in both the live progress view (finished agents above pending/running) and the finalized result view, so rows no longer reshuffle on finalize. Adds a task gallery fixture variant for the resume path (renderer key separated from fixture key).\n\nAll task tests are reshaped around the single-call contract; tests for the discarded shared-context flow are removed, and new task-guards/task-resume/task-schema tests pin the new contract surface.
2026-06-10 17:54:47 +02:00

65 lines
2.1 KiB
TypeScript

import { afterEach, describe, expect, it, vi } from "bun:test";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { runEvalAgent } from "@oh-my-pi/pi-coding-agent/eval/agent-bridge";
import type { LocalProtocolOptions } from "@oh-my-pi/pi-coding-agent/internal-urls";
import type { MCPManager } from "@oh-my-pi/pi-coding-agent/mcp";
import * as taskDiscovery from "@oh-my-pi/pi-coding-agent/task/discovery";
import * as taskExecutor from "@oh-my-pi/pi-coding-agent/task/executor";
import type { AgentDefinition, SingleResult } from "@oh-my-pi/pi-coding-agent/task/types";
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
function createResult(): SingleResult {
return {
index: 0,
id: "0-Task",
agent: "task",
agentSource: "bundled",
task: "do work",
exitCode: 0,
output: "done",
stderr: "",
truncated: false,
durationMs: 1,
tokens: 0,
requests: 0,
};
}
describe("runEvalAgent", () => {
afterEach(() => {
vi.restoreAllMocks();
});
it("forwards session-scoped MCP and local protocol options", async () => {
const agent: AgentDefinition = {
name: "task",
description: "Task agent",
systemPrompt: "Handle task",
source: "bundled",
};
vi.spyOn(taskDiscovery, "discoverAgents").mockResolvedValue({ agents: [agent], projectAgentsDir: null });
const runSubprocessSpy = vi.spyOn(taskExecutor, "runSubprocess").mockResolvedValue(createResult());
const mcpManager = { sentinel: "mcp" } as unknown as MCPManager;
const localProtocolOptions: LocalProtocolOptions = {
getArtifactsDir: () => "/tmp/parent-artifacts",
getSessionId: () => "parent-session",
};
const session = {
cwd: "/tmp",
settings: Settings.isolated(),
getSessionSpawns: () => "*",
getSessionFile: () => null,
mcpManager,
localProtocolOptions,
} as unknown as ToolSession;
await runEvalAgent({ prompt: "do work", agentType: "task" }, { session });
expect(runSubprocessSpy).toHaveBeenCalledTimes(1);
const options = runSubprocessSpy.mock.calls[0]?.[0];
expect(options?.mcpManager).toBe(mcpManager);
expect(options?.localProtocolOptions).toBe(localProtocolOptions);
});
});