5ff277349c
- Added the `xd://` virtual device protocol (`internal-urls/xd-protocol.ts`, `tools/xdev.ts`): tools declaring `loadMode: "discoverable"` are unmounted from the request tools array and driven via `read xd://` (list/docs+schema) and `write xd://<tool>` (execute), gated by the `tools.xdev` setting (default on) and inlined into the system prompt. - Merged the `irc`, `job`, and `launch` tools into a single `hub` tool (`tools/hub/`, `async/job-manager.ts`): messaging keeps `send`/`inbox`/`list`, job control maps to `wait`/`cancel`/`jobs`, process supervision keeps `start`/`logs`/`stop`/`restart`/`describe` with `ps`, and the unified `wait` races background jobs against peer messages; SDK `IrcTool`/`JobTool`/`LaunchTool` are replaced by `HubTool`. - Removed the hidden `resolve` tool in favor of the `xd://resolve`/`xd://reject`/`xd://propose` resolution devices, auto-including `write` whenever a deferrable tool or plan mode is present. - Removed the BM25 tool-discovery system: the `search_tool_bm25` tool, the `tool-discovery` module, the `tools.discoveryMode`/`mcp.discoveryMode`/`mcp.discoveryDefaultServers`/`tools.essentialOverride` settings, per-tool MCP selection, and the `mcp_tool_selection` message type. - Unified tool presentation on `ToolLoadMode` (`essential`|`discoverable`), replacing the custom-tool `xdev?: boolean` opt-out; custom, extension, MCP, RPC host, image-generation, and TTS tools now default to `discoverable`, and added a `satisfies` predicate to `SoftToolRequirement`. - Removed the standalone `ssh` command tool and `ssh/ssh-executor` (the `ssh://` read/write/search protocol stays), and made `--tools` address hidden built-ins. - Updated collab-web to render `xd://` dispatches and `hub` op families, dropped the `search_tool_bm25`/`ssh`/`report-finding` renderers, refreshed tool docs and prompts, and migrated the affected tests and changelogs.
122 lines
4.3 KiB
TypeScript
122 lines
4.3 KiB
TypeScript
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
|
|
import * as path from "node:path";
|
|
import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core";
|
|
import type { AssistantMessage, Context } from "@oh-my-pi/pi-ai";
|
|
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
|
|
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
|
|
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
|
|
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
|
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
|
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
|
import { type CustomMessage, convertToLlm } from "@oh-my-pi/pi-coding-agent/session/messages";
|
|
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
|
import { TempDir } from "@oh-my-pi/pi-utils";
|
|
import { type } from "arktype";
|
|
|
|
const zeroUsage = {
|
|
input: 0,
|
|
output: 0,
|
|
cacheRead: 0,
|
|
cacheWrite: 0,
|
|
totalTokens: 0,
|
|
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
|
} satisfies AssistantMessage["usage"];
|
|
|
|
describe("AgentSession tool-call loop guard", () => {
|
|
let tempDir: TempDir;
|
|
let authStorage: AuthStorage;
|
|
let session: AgentSession | undefined;
|
|
|
|
beforeEach(async () => {
|
|
tempDir = TempDir.createSync("@pi-tool-call-loop-guard-");
|
|
authStorage = await AuthStorage.create(path.join(tempDir.path(), "auth.db"));
|
|
authStorage.setRuntimeApiKey("openai", "openai-test-key");
|
|
});
|
|
|
|
afterEach(async () => {
|
|
await session?.dispose();
|
|
authStorage.close();
|
|
tempDir.removeSync();
|
|
});
|
|
|
|
it("injects a hidden redirect before the next model call", async () => {
|
|
const model = createMockModel({ provider: "openai", id: "gpt-test" }).model;
|
|
const modelRegistry = new ModelRegistry(authStorage);
|
|
const contexts: Context[] = [];
|
|
const bashTool: AgentTool = {
|
|
name: "bash",
|
|
label: "Bash",
|
|
description: "Mock bash tool",
|
|
parameters: type({ "command?": "string" }),
|
|
execute: async () => ({ content: [{ type: "text" as const, text: "1263 passed, 4 skipped" }] }),
|
|
};
|
|
let callCount = 0;
|
|
const agent = new Agent({
|
|
getApiKey: () => "test-key",
|
|
initialState: { model, systemPrompt: ["Test"], tools: [bashTool], messages: [] },
|
|
convertToLlm,
|
|
streamFn: (_model, context) => {
|
|
contexts.push(context);
|
|
const toolCallTurn = callCount < 5;
|
|
const toolCallId = `tc-${callCount}`;
|
|
callCount++;
|
|
const message: AssistantMessage = toolCallTurn
|
|
? {
|
|
role: "assistant",
|
|
content: [{ type: "toolCall", id: toolCallId, name: "bash", arguments: { command: "pytest -q" } }],
|
|
api: model.api,
|
|
provider: model.provider,
|
|
model: model.id,
|
|
usage: zeroUsage,
|
|
stopReason: "toolUse",
|
|
timestamp: Date.now(),
|
|
}
|
|
: {
|
|
role: "assistant",
|
|
content: [{ type: "text", text: "Stopped repeating." }],
|
|
api: model.api,
|
|
provider: model.provider,
|
|
model: model.id,
|
|
usage: zeroUsage,
|
|
stopReason: "stop",
|
|
timestamp: Date.now(),
|
|
};
|
|
const stream = new AssistantMessageEventStream();
|
|
queueMicrotask(() => {
|
|
stream.push({ type: "start", partial: message });
|
|
stream.push({ type: "done", reason: toolCallTurn ? "toolUse" : "stop", message });
|
|
});
|
|
return stream;
|
|
},
|
|
});
|
|
const settings = Settings.isolated({
|
|
"compaction.enabled": false,
|
|
"todo.enabled": false,
|
|
"model.toolCallLoopGuard.enabled": true,
|
|
"model.toolCallLoopGuard.threshold": 5,
|
|
"model.toolCallLoopGuard.exemptTools": ["hub"],
|
|
});
|
|
settings.setModelRole("default", `${model.provider}/${model.id}`);
|
|
session = new AgentSession({
|
|
agent,
|
|
sessionManager: SessionManager.inMemory(tempDir.path()),
|
|
settings,
|
|
modelRegistry,
|
|
toolRegistry: new Map([[bashTool.name, bashTool]]),
|
|
});
|
|
|
|
await session.prompt("run checks");
|
|
await session.waitForIdle();
|
|
|
|
expect(contexts).toHaveLength(6);
|
|
expect(JSON.stringify(contexts[5]!.messages)).toContain("tool_call_loop_detected");
|
|
expect(JSON.stringify(contexts[5]!.messages)).toContain("1263 passed, 4 skipped");
|
|
const redirects = session.agent.state.messages.filter(
|
|
(message): message is CustomMessage =>
|
|
message.role === "custom" && message.customType === "tool-call-loop-redirect",
|
|
);
|
|
expect(redirects).toHaveLength(1);
|
|
expect(redirects[0]!.display).toBe(false);
|
|
});
|
|
});
|