chore: cleanup dumb tests
This commit is contained in:
@@ -1,11 +1,6 @@
|
||||
import { afterEach, describe, expect, test, vi } from "bun:test";
|
||||
import type { AgentMessage, AgentTool } from "@oh-my-pi/pi-agent-core";
|
||||
import {
|
||||
AUTO_HANDOFF_THRESHOLD_FOCUS,
|
||||
generateHandoff,
|
||||
generateHandoffFromContext,
|
||||
renderHandoffPrompt,
|
||||
} from "@oh-my-pi/pi-agent-core/compaction";
|
||||
import { generateHandoff, generateHandoffFromContext, renderHandoffPrompt } from "@oh-my-pi/pi-agent-core/compaction";
|
||||
import { ThinkingLevel } from "@oh-my-pi/pi-agent-core/thinking";
|
||||
import type { AssistantMessage, Model, ToolCall } from "@oh-my-pi/pi-ai";
|
||||
import * as ai from "@oh-my-pi/pi-ai";
|
||||
@@ -72,12 +67,6 @@ describe("handoff helpers", () => {
|
||||
expect(rendered).toContain("preserve failing test name");
|
||||
});
|
||||
|
||||
test("exports the threshold focus text used by auto-handoff", () => {
|
||||
expect(AUTO_HANDOFF_THRESHOLD_FOCUS).toBe(
|
||||
"Threshold-triggered maintenance: preserve critical implementation state and immediate next actions.",
|
||||
);
|
||||
});
|
||||
|
||||
test("generates handoff with the live cache prefix and tool use disabled", async () => {
|
||||
const strayToolCall: ToolCall = { type: "toolCall", id: "call_1", name: "read", arguments: {} };
|
||||
const completeSimpleSpy = vi
|
||||
|
||||
@@ -403,8 +403,6 @@ describe("GitLab Duo Workflow provider protocol", () => {
|
||||
// This goal IS a multi-turn ChatML transcript, so the system slot appends the
|
||||
// history-note telling the model the `<|im_start|>`/`<ran …>` markers are a past
|
||||
// record, not a tool-call syntax to emit.
|
||||
expect(flowPrompt?.prompt_template.system).toContain("written as a plain-text log");
|
||||
expect(flowPrompt?.prompt_template.system).toContain("never write `<ran …>`");
|
||||
});
|
||||
|
||||
it("strips the OMP-internal intent (i) field from replayed tool-call args", () => {
|
||||
|
||||
@@ -424,6 +424,7 @@ async function discoverOllamaModelMetadata(
|
||||
try {
|
||||
const payload = await withTimeoutSignal(discoveryProbeTimeoutMs(endpoint, 150, customTimeoutMs), async signal => {
|
||||
const response = await ctx.fetch(showUrl, {
|
||||
method: "POST",
|
||||
headers: { ...(headers ?? {}), "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ model: modelId }),
|
||||
signal,
|
||||
|
||||
@@ -1778,9 +1778,7 @@ describe("ACP agent", () => {
|
||||
if (!customMessage) throw new Error("expected ACP skill prompt custom message");
|
||||
expect(customMessage.customType).toBe("skill-prompt");
|
||||
expect(customMessage.content).toContain("# Sample\nDo work.");
|
||||
expect(customMessage.content).toContain('The user has invoked the "sample" skill');
|
||||
expect(customMessage.content).toContain(`[Skill directory: ${skillDir}]`);
|
||||
expect(customMessage.content).toMatch(/[Rr]esolve any relative paths/);
|
||||
expect(customMessage.content).toContain("User: extra context");
|
||||
expect(session.customMessageOptions[0]).toEqual({ streamingBehavior: "steer" });
|
||||
|
||||
|
||||
@@ -142,14 +142,9 @@ describe("advisor watchdog prompt discovery", () => {
|
||||
fs.writeFileSync(path.join(cwd, "WATCHDOG.md"), watchdogContent, "utf8");
|
||||
|
||||
await withAdvisorHistory(tempDir, cwd, dump => {
|
||||
expect(dump).toContain("Especially pay attention to:");
|
||||
expect(dump).toContain("exactly one direct child git repository");
|
||||
expect(dump).toContain("`active-project`");
|
||||
expect(dump).toContain("Do not claim work is missing, destroyed, or absent at the parent cwd");
|
||||
expect(dump).toContain(watchdogContent);
|
||||
expect(dump.indexOf(watchdogContent)).toBeLessThan(
|
||||
dump.indexOf("Do not claim work is missing, destroyed, or absent at the parent cwd"),
|
||||
);
|
||||
expect(dump.indexOf(watchdogContent)).toBeLessThan(dump.indexOf("`active-project`"));
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -30,7 +30,6 @@ import type { Settings } from "../../src/config/settings";
|
||||
import { type AdvisorConfigDeps, AdvisorConfigOverlayComponent } from "../../src/modes/components/advisor-config";
|
||||
import { createAdvisorMessageCard } from "../../src/modes/components/advisor-message";
|
||||
import { getThemeByName, setThemeInstance } from "../../src/modes/theme/theme";
|
||||
import advisorSystemPrompt from "../../src/prompts/advisor/system.md" with { type: "text" };
|
||||
import { SecretObfuscator } from "../../src/secrets/obfuscator";
|
||||
import { formatSessionHistoryMarkdown } from "../../src/session/session-history-format";
|
||||
import { YieldQueue } from "../../src/session/yield-queue";
|
||||
@@ -74,9 +73,6 @@ describe("advisor", () => {
|
||||
|
||||
expect(rendered).toContain("→ grep(needle @ packages/coding-agent/src) ⇒ error");
|
||||
expect(rendered).not.toContain("paths[0]");
|
||||
expect(advisorSystemPrompt).toContain("Arguments absent from the rendered transcript are UNKNOWN");
|
||||
expect(advisorSystemPrompt).toContain("NEVER assert concrete values, array indexes");
|
||||
expect(advisorSystemPrompt).toContain("NEVER claim `paths[0]`, array flattening, or malformed `paths`");
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -231,9 +231,17 @@ describe("AgentSession checkpoint rewind branch context", () => {
|
||||
expect(mock.calls.length).toBe(3);
|
||||
const finalCall = mock.calls[2];
|
||||
if (!finalCall) throw new Error("Expected final post-rewind provider call");
|
||||
const summaryIndex = finalCall.context.messages.findIndex(
|
||||
message => message.role === "user" && messageText(message).includes("summary of a branch"),
|
||||
);
|
||||
const summaryIndex = finalCall.context.messages.findIndex(message => {
|
||||
if (message.role !== "user") return false;
|
||||
const text = messageText(message);
|
||||
const openIndex = text.indexOf("<summary>");
|
||||
const closeIndex = text.indexOf("</summary>", openIndex + "<summary>".length);
|
||||
return (
|
||||
openIndex >= 0 &&
|
||||
closeIndex > openIndex + "<summary>".length &&
|
||||
text.slice(openIndex + "<summary>".length, closeIndex).trim().length > 0
|
||||
);
|
||||
});
|
||||
const reportIndex = finalCall.context.messages.findIndex(
|
||||
message => message.role === "developer" && messageText(message).includes(report),
|
||||
);
|
||||
@@ -242,8 +250,7 @@ describe("AgentSession checkpoint rewind branch context", () => {
|
||||
const reportMessage = finalCall.context.messages[reportIndex];
|
||||
if (!reportMessage) throw new Error("Expected rewind report context");
|
||||
const reportText = messageText(reportMessage);
|
||||
expect(reportText).toContain("Checkpoint completed.");
|
||||
expect(reportText).toContain("Do not call `rewind` again");
|
||||
expect(reportText).toContain("`rewind`");
|
||||
expect(reportText).toContain(report);
|
||||
|
||||
expect(
|
||||
|
||||
@@ -193,8 +193,6 @@ describe("AgentSession eager task prelude", () => {
|
||||
expect(observedCalls).toHaveLength(1);
|
||||
expect(observedCalls[0]?.toolChoice).toBeUndefined();
|
||||
expect(observedCalls[0]?.messageRoles).toEqual(["developer", "user"]);
|
||||
expect(observedCalls[0]?.messageTexts[0]).toContain("delegation is enabled");
|
||||
expect(observedCalls[0]?.messageTexts[0]).toContain("Batch independent slices");
|
||||
expect(observedCalls[0]?.messageTexts[0]).toContain("`task`");
|
||||
expect(
|
||||
observedCalls[0]?.messageTexts.filter(text => text.includes("refactor the parser across modules")),
|
||||
@@ -278,7 +276,6 @@ describe("AgentSession eager task prelude", () => {
|
||||
|
||||
expect(observedCalls).toHaveLength(1);
|
||||
expect(observedCalls[0]?.messageRoles).toEqual(["developer", "user"]);
|
||||
expect(observedCalls[0]?.messageTexts[0]).toContain("delegation is enabled");
|
||||
});
|
||||
|
||||
it("prepends both todo and task preludes when both are eager, keeping the forced todo choice", async () => {
|
||||
@@ -298,18 +295,7 @@ describe("AgentSession eager task prelude", () => {
|
||||
const texts = observedCalls[0]?.messageTexts ?? [];
|
||||
expect(texts.at(-1)).toBe("refactor the parser across modules");
|
||||
// the task reminder is the second prelude (after the todo reminder)
|
||||
expect(texts.findIndex(text => text.includes("delegation is enabled"))).toBe(1);
|
||||
});
|
||||
|
||||
it("omits batch-call guidance from the eager task reminder when task.batch is disabled", async () => {
|
||||
const { session, observedCalls } = await createHarness({ "task.batch": false });
|
||||
|
||||
await session.prompt("refactor the parser across modules");
|
||||
|
||||
expect(observedCalls).toHaveLength(1);
|
||||
const reminder = observedCalls[0]?.messageTexts[0] ?? "";
|
||||
expect(reminder).toContain("delegation is enabled");
|
||||
expect(reminder).not.toContain("Batch independent slices");
|
||||
expect(texts.findIndex(text => text.includes("`task`"))).toBe(1);
|
||||
});
|
||||
|
||||
it("renders the task tool's wire name in the eager reminder", async () => {
|
||||
|
||||
@@ -15,7 +15,6 @@ import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
|
||||
import { TodoTool } from "@oh-my-pi/pi-coding-agent/tools";
|
||||
import { setInteractiveHost, TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { type } from "arktype";
|
||||
import eagerTodoPrompt from "../src/prompts/system/eager-todo.md" with { type: "text" };
|
||||
import { createAssistantMessage } from "./helpers/agent-session-setup";
|
||||
|
||||
type ObservedPromptCall = {
|
||||
@@ -234,14 +233,6 @@ describe("AgentSession eager todo enforcement", () => {
|
||||
tempDir.removeSync();
|
||||
});
|
||||
|
||||
it("keeps eager init instructions aligned with the todo schema", () => {
|
||||
expect(eagerTodoPrompt).toContain("single `init` op");
|
||||
expect(eagerTodoPrompt).toContain("phase names and task-label strings");
|
||||
expect(eagerTodoPrompt).not.toContain("`details`");
|
||||
expect(eagerTodoPrompt).not.toContain("in_progress");
|
||||
expect(eagerTodoPrompt).not.toContain("pending");
|
||||
});
|
||||
|
||||
it("prepends a hidden eager todo reminder without repeating the prompt text", async () => {
|
||||
await session.prompt("list all work trees");
|
||||
|
||||
@@ -257,7 +248,6 @@ describe("AgentSession eager todo enforcement", () => {
|
||||
expect(observedCalls[0]?.messageTexts.filter(text => text.includes("list all work trees"))).toHaveLength(1);
|
||||
expect(observedCalls[0]?.messageTexts[0]).not.toContain("list all work trees");
|
||||
// `always` renders the hard, forced reminder.
|
||||
expect(observedCalls[0]?.messageTexts[0]).toContain("You MUST call");
|
||||
expect(session.formatSessionAsText()).not.toContain("<user-request>");
|
||||
});
|
||||
|
||||
@@ -529,8 +519,5 @@ describe("AgentSession eager todo enforcement", () => {
|
||||
expect(observedCalls[0]?.messageTexts.at(-1)).toBe("list all work trees");
|
||||
expect(observedCalls[0]?.messageTexts[0]).not.toContain("list all work trees");
|
||||
// `preferred` renders the soft nudge, never the hard MUST directive.
|
||||
expect(observedCalls[0]?.messageTexts[0]).toContain("Consider calling");
|
||||
expect(observedCalls[0]?.messageTexts[0]).not.toContain("You MUST call");
|
||||
expect(observedCalls[0]?.messageTexts[0]).not.toContain("Before substantive work, create a phased todo.");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -132,13 +132,16 @@ describe("AgentSession magic keyword settings", () => {
|
||||
|
||||
await session.prompt("please workflowz this");
|
||||
|
||||
const promptMessages = promptSpy.mock.calls[0]![0] as unknown as Array<{ content?: string; customType?: string }>;
|
||||
const notice = promptMessages.find(message => message.customType === "workflow-notice")?.content ?? "";
|
||||
expect(notice).toContain("Author the orchestration in the `eval` tool");
|
||||
expect(notice).toContain("Every eval call has:");
|
||||
expect(notice).toContain("`parallel(thunks)`");
|
||||
expect(notice).toContain("**Python (`eval`, Python backend):**");
|
||||
expect(notice).toContain("**JavaScript (`eval`, JavaScript backend):**");
|
||||
const promptMessages = promptSpy.mock.calls[0]![0] as unknown as Array<{
|
||||
content?: string;
|
||||
customType?: string;
|
||||
}>;
|
||||
const notice = promptMessages.find(message => message.customType === "workflow-notice");
|
||||
expect(notice?.customType).toBe("workflow-notice");
|
||||
expect(notice?.content).toContain("`eval`");
|
||||
expect(notice?.content).toContain("`parallel(thunks)`");
|
||||
expect(notice?.content).toContain("**Python (`eval`, Python backend):**");
|
||||
expect(notice?.content).toContain("**JavaScript (`eval`, JavaScript backend):**");
|
||||
});
|
||||
|
||||
it("skips workflowz notice when the task tool is inactive", async () => {
|
||||
|
||||
@@ -1404,9 +1404,10 @@ describe("AgentSession message pipeline", () => {
|
||||
expect(getConvertedUserText(lastMessage)).toBe("Side Question?");
|
||||
|
||||
expect(secondToLast?.role).toBe("developer");
|
||||
expect(secondToLast?.content).toBeDefined();
|
||||
const textContent = secondToLast?.content as { text?: string }[];
|
||||
expect(textContent[0].text).toContain("tool catalog stays attached");
|
||||
const textContent = secondToLast?.content as TextContent[];
|
||||
expect(textContent).toHaveLength(1);
|
||||
expect(textContent[0]?.type).toBe("text");
|
||||
expect(textContent[0]?.text).toMatch(/^<system-reminder>\n[\s\S]+\n<\/system-reminder>\n?$/);
|
||||
|
||||
// Tool choice must be undefined (not "none") for cache hits
|
||||
expect(capturedOptions?.toolChoice).toBeUndefined();
|
||||
|
||||
@@ -281,11 +281,7 @@ describe("auto thinking classifier helpers", () => {
|
||||
const optedIn = createOnlineFixture(buildLadderModel("mock-max", MAX_LADDER), "high", "max");
|
||||
await classifyDifficulty("refactor the scheduler", optedIn.deps);
|
||||
const optedInRequest = optedIn.completeSimpleMock.mock.calls[0]?.[1] as { systemPrompt: string[] };
|
||||
// The label alone is inert: the criteria and the tie-break exception are
|
||||
// what make the tier reachable, so all three must ship together.
|
||||
expect(optedInRequest.systemPrompt[0]).toContain("`max`");
|
||||
expect(optedInRequest.systemPrompt[0]).toContain("no reproduction to work from");
|
||||
expect(optedInRequest.systemPrompt[0]).toContain("except between `xhigh` and `max`");
|
||||
|
||||
vi.restoreAllMocks();
|
||||
|
||||
@@ -294,10 +290,6 @@ describe("auto thinking classifier helpers", () => {
|
||||
const defaultedRequest = defaulted.completeSimpleMock.mock.calls[0]?.[1] as { systemPrompt: string[] };
|
||||
expect(defaultedRequest.systemPrompt[0]).not.toMatch(/\bmax\b/);
|
||||
expect(defaultedRequest.systemPrompt[0]).toContain("`xhigh`");
|
||||
// The tie-break exception is what makes the top tier reachable, so it must
|
||||
// not leak into the prompt of a user who did not opt in.
|
||||
expect(defaultedRequest.systemPrompt[0]).toContain("choose the lower one.");
|
||||
expect(defaultedRequest.systemPrompt[0]).not.toContain("no reproduction to work from");
|
||||
|
||||
vi.restoreAllMocks();
|
||||
|
||||
@@ -305,7 +297,6 @@ describe("auto thinking classifier helpers", () => {
|
||||
await classifyDifficulty("refactor the scheduler", unsupported.deps);
|
||||
const unsupportedRequest = unsupported.completeSimpleMock.mock.calls[0]?.[1] as { systemPrompt: string[] };
|
||||
expect(unsupportedRequest.systemPrompt[0]).not.toMatch(/\bmax\b/);
|
||||
expect(unsupportedRequest.systemPrompt[0]).toContain("choose the lower one.");
|
||||
});
|
||||
|
||||
it("resolves max only when opted in, and snaps it to the ceiling otherwise", async () => {
|
||||
|
||||
@@ -144,27 +144,6 @@ describe("AutoLearnController", () => {
|
||||
expect(session.captures).toHaveLength(0);
|
||||
});
|
||||
|
||||
it("the auto-continue nudge is terminal — capture then stop, never assume approval (#3504)", () => {
|
||||
// Regression: with autoContinue on, the synthetic capture turn carries
|
||||
// the nudge as its only user-role payload. Without an explicit "stop /
|
||||
// not a user reply / do not assume approval" contract, the agent reads
|
||||
// its own unanswered prior question (e.g. "Want me to commit and
|
||||
// push?") as accepted and continues — exactly the scenario in #3504.
|
||||
const session = new FakeSession();
|
||||
install(session, { "autolearn.autoContinue": true });
|
||||
session.toolCalls(5);
|
||||
session.agentEnd();
|
||||
const body = session.captures[0] ?? "";
|
||||
// Frames the prompt as automated, not as the user's response.
|
||||
expect(body).toMatch(/not a user reply|not from the user/i);
|
||||
// Forbids inferring approval / acting on pending questions.
|
||||
expect(body).toMatch(/not.*(approval|accept|pending|prior)/i);
|
||||
// Demands a hard stop after capture, with no continuation.
|
||||
expect(body).toMatch(/then stop\./i);
|
||||
expect(body).toMatch(/do not.*(continue|resume|other tools)/i);
|
||||
expect(body).toMatch(/wait for the user'?s next prompt/i);
|
||||
});
|
||||
|
||||
it("does not nudge below the threshold", () => {
|
||||
const session = new FakeSession();
|
||||
install(session, { "autolearn.autoContinue": true });
|
||||
|
||||
@@ -947,7 +947,6 @@ describe("remote compaction setting", () => {
|
||||
})
|
||||
.join("\n");
|
||||
|
||||
expect(promptText).toContain("Previous snapcompact archive source text:");
|
||||
expect(promptText).toContain("Archived snapcompact source");
|
||||
expect(result.preserveData).toEqual({ otherState: "keep-me" });
|
||||
});
|
||||
|
||||
@@ -271,7 +271,6 @@ describe("InteractiveMode goal mode integration", () => {
|
||||
expect(content).toContain("- [completed] Identify gaps");
|
||||
expect(content).toContain("- [in_progress] Choose <next> & slice </todo_context>");
|
||||
expect(content).toContain("- [pending] Run focused checks");
|
||||
expect(content).toContain("call the `todo` tool first");
|
||||
expect(content.match(/<\/todo_context>/g)).toHaveLength(1);
|
||||
});
|
||||
|
||||
|
||||
@@ -134,7 +134,6 @@ describe("guided goal setup", () => {
|
||||
expect(promptSpy).toHaveBeenCalledTimes(1);
|
||||
const [text] = promptSpy.mock.calls[0]!;
|
||||
expect(text).not.toContain("<rough-goal>");
|
||||
expect(text).toContain("not stated an objective");
|
||||
} finally {
|
||||
await harness.cleanup();
|
||||
}
|
||||
|
||||
@@ -331,10 +331,8 @@ describe("compaction skill re-invocation", () => {
|
||||
// Bug fix contract: a re-invoked user skill identifies itself and exposes its
|
||||
// skill directory so relative skill paths resolve after compaction.
|
||||
expect(renderedText.text).toContain("Do the thing.");
|
||||
expect(renderedText.text).toContain('The user has invoked the "test-skill" skill');
|
||||
expect(renderedText.text).toContain(`[Skill directory: ${tempDir.path()}]`);
|
||||
expect(renderedText.text).toMatch(/[Rr]esolve any relative paths/);
|
||||
expect(renderedText.text).toContain("User: arg1 arg2");
|
||||
expect(renderedText.text).toContain("arg1 arg2");
|
||||
expect(message.content[1]).toEqual(image);
|
||||
expect(message.details).toMatchObject({ name: "test-skill", args: "arg1 arg2", lineCount: 1 });
|
||||
expect(options).toEqual({
|
||||
|
||||
@@ -156,8 +156,6 @@ describe("InteractiveMode vibe mode toggle", () => {
|
||||
const content = typeof message.content === "string" ? message.content : "";
|
||||
expect(message.customType).toBe("vibe-mode-context");
|
||||
expect(content).toContain("`todo`");
|
||||
expect(content).toContain("parent session's list");
|
||||
expect(content).toContain("Workers do not own this bookkeeping.");
|
||||
|
||||
// Toggle off: the empty previous toolset must come back — only the
|
||||
// ephemeral vibe tools must leave the registry.
|
||||
|
||||
@@ -1170,6 +1170,7 @@ providers:
|
||||
custom-remote:
|
||||
baseUrl: "http://127.0.0.1:8080"
|
||||
api: "openai-completions"
|
||||
auth: "none"
|
||||
discovery:
|
||||
type: "llama.cpp"
|
||||
timeoutMs: 45000
|
||||
@@ -1194,7 +1195,7 @@ providers:
|
||||
throw new Error(`Unexpected URL: ${url}`);
|
||||
};
|
||||
|
||||
const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock, modelsConfigPath: customConfigPath });
|
||||
const registry = new ModelRegistry(authStorage, customConfigPath, { fetch: fetchMock });
|
||||
await registry.refresh();
|
||||
const state = registry.getProviderDiscoveryState("custom-remote");
|
||||
expect(state?.status).toBe("ok");
|
||||
|
||||
@@ -67,21 +67,13 @@ describe("workflow keyword highlighting", () => {
|
||||
describe("workflow notice", () => {
|
||||
it("renders the Workflowz trigger with eval orchestration helper guidance", () => {
|
||||
expect(WORKFLOW_NOTICE).toContain("**workflowz** keyword");
|
||||
expect(WORKFLOW_NOTICE).toContain("Author the orchestration in the `eval` tool");
|
||||
expect(WORKFLOW_NOTICE).toContain("JavaScript (`eval`, JavaScript backend):");
|
||||
expect(WORKFLOW_NOTICE).toContain("Use ordinary code between calls to flatten/map/filter");
|
||||
expect(WORKFLOW_NOTICE).toContain("State persists across eval calls");
|
||||
expect(WORKFLOW_NOTICE).toContain("`parallel(thunks)`");
|
||||
expect(WORKFLOW_NOTICE).toContain("a negative value disables the cap");
|
||||
expect(WORKFLOW_NOTICE).toContain("await budget.remaining()");
|
||||
});
|
||||
|
||||
it("renders the same eval notice when task.batch is disabled", () => {
|
||||
const notice = renderWorkflowNotice({ taskBatch: false });
|
||||
expect(notice).toContain("**workflowz** keyword");
|
||||
expect(notice).toContain("Author the orchestration in the `eval` tool");
|
||||
expect(notice).toContain("JavaScript (`eval`, JavaScript backend):");
|
||||
expect(notice).toContain("State persists across eval calls");
|
||||
expect(notice).toContain("`parallel(thunks)`");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -20,22 +20,4 @@ describe("plan-mode re-entry prompt", () => {
|
||||
expect(render({ reentry: false, planExists: true })).not.toContain("## Re-entry");
|
||||
expect(render({ reentry: true, planExists: true })).toContain("## Re-entry");
|
||||
});
|
||||
|
||||
it("anchors the turn on the new request, not the old plan", () => {
|
||||
const rendered = render({ reentry: true, planExists: true });
|
||||
const reentry = rendered.slice(rendered.indexOf("## Re-entry"));
|
||||
// The new request is the primary input; the old plan is reference only.
|
||||
expect(reentry).toMatch(/NEW request[\s\S]*primary input/);
|
||||
// Corrections to an incomplete old plan must be folded into the new plan,
|
||||
// never substituted for it (the reported failure: dropping the new request).
|
||||
expect(reentry).toMatch(/combine, never substitute/);
|
||||
});
|
||||
|
||||
it("does not contradict the planExists guidance on a different task", () => {
|
||||
const rendered = render({ reentry: true, planExists: true });
|
||||
// planExists branch: different task -> leave old plan, write a fresh file.
|
||||
expect(rendered).toContain("leave that plan in place and start a fresh");
|
||||
// Re-entry must agree; the old "Different task -> overwrite it" directive is gone.
|
||||
expect(rendered).not.toMatch(/[Dd]ifferent task → overwrite/);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -35,10 +35,8 @@ describe("tryRunRpcSkillCommand", () => {
|
||||
expect(handled).toEqual({ agentInvoked: true });
|
||||
expect(message?.customType).toBe(SKILL_PROMPT_MESSAGE_TYPE);
|
||||
expect(message?.content).toContain("Review the supplied code carefully.");
|
||||
expect(message?.content).toContain('The user has invoked the "reviewer" skill');
|
||||
expect(message?.content).toContain(`[Skill directory: ${dir}]`);
|
||||
expect(message?.content).toMatch(/[Rr]esolve any relative paths/);
|
||||
expect(message?.content).toContain("User: focus on risks");
|
||||
expect(message?.content).toContain("focus on risks");
|
||||
expect(message?.display).toBe(true);
|
||||
expect(message?.attribution).toBe("user");
|
||||
expect(options).toEqual({ streamingBehavior: "steer" });
|
||||
|
||||
@@ -26,9 +26,6 @@ import {
|
||||
// instructions and the installed Context Mode server's absent instructions.
|
||||
const FIXTURE_PATH = path.join(import.meta.dir, "fixtures", "instructions-mcp.ts");
|
||||
const MCP_TOOL_NAME = "mcp__instr_do_thing";
|
||||
const MCP_MAPPING_FALLBACK =
|
||||
"Additional mounted MCP tool mappings were omitted to keep this prompt bounded. Inspect `xd://` for the exact current paths.";
|
||||
const MCP_EXECUTION_GUIDANCE = "Execute each mounted tool by writing JSON arguments to its mounted path:";
|
||||
const MCP_ROUTE_SECTION = "## MCP Tool Routes";
|
||||
const CONTEXT_MODE_ROUTE = '- "ctx_execute" → `xd://mcp__context_mode_ctx_execute`';
|
||||
const CONTEXT_MODE_MCP_TOOL_NAME = "mcp__context_mode_ctx_execute";
|
||||
@@ -132,7 +129,6 @@ describe("createAgentSession MCP server instructions (deferred UI)", () => {
|
||||
// normalized name actually mounted in the live xd:// registry.
|
||||
expect(prompt).toContain("MCP Server Instructions");
|
||||
expect(prompt).toContain('- "do\\u0060thing" → `xd://mcp__instr_do_thing`');
|
||||
expect(prompt).toContain(MCP_EXECUTION_GUIDANCE);
|
||||
} finally {
|
||||
await session.dispose();
|
||||
}
|
||||
@@ -185,7 +181,6 @@ describe("createAgentSession MCP server instructions (deferred UI)", () => {
|
||||
expect(prompt).toContain(CONTEXT_MODE_ROUTE);
|
||||
expect(session.getXdevToolEntries().map(entry => entry.name)).toContain(CONTEXT_MODE_MCP_TOOL_NAME);
|
||||
expect(session.getActiveToolNames()).not.toContain(CONTEXT_MODE_MCP_TOOL_NAME);
|
||||
expect(prompt.split(MCP_EXECUTION_GUIDANCE)).toHaveLength(2);
|
||||
expect(prompt.split(MCP_ROUTE_SECTION)).toHaveLength(2);
|
||||
expect(prompt).not.toContain(SERVER_INSTRUCTIONS);
|
||||
expect(prompt).not.toContain("## MCP Server Instructions");
|
||||
@@ -242,7 +237,8 @@ describe("createAgentSession MCP server instructions (deferred UI)", () => {
|
||||
expect(renderedMappings[0]).toBe('- "row_aa" → `xd://mcp__instr_row_aa`');
|
||||
expect(renderedMappings[63]).toBe('- "row_cl" → `xd://mcp__instr_row_cl`');
|
||||
expect(prompt).not.toContain('- "row_cm" → `xd://mcp__instr_row_cm`');
|
||||
expect(prompt).toContain(MCP_MAPPING_FALLBACK);
|
||||
// Truncation notice present (row_cm absent above proves the cap applied).
|
||||
expect(prompt).toContain("omitted");
|
||||
} finally {
|
||||
await session.dispose();
|
||||
}
|
||||
|
||||
@@ -28,10 +28,8 @@ describe("buildSkillPromptMessage", () => {
|
||||
const built = await buildSkillPromptMessage(skill, "focus on risks");
|
||||
|
||||
expect(built.message).toContain("Review the supplied code carefully.");
|
||||
expect(built.message).toContain('The user has invoked the "reviewer" skill');
|
||||
expect(built.message).toContain(`[Skill directory: ${dir}]`);
|
||||
expect(built.message).toMatch(/[Rr]esolve any relative paths/);
|
||||
expect(built.message).toContain("User: focus on risks");
|
||||
expect(built.message).toContain("focus on risks");
|
||||
expect(built.details).toMatchObject({
|
||||
name: "reviewer",
|
||||
path: skill.filePath,
|
||||
|
||||
@@ -283,7 +283,6 @@ describe("SnapcompactInlineTransformer", () => {
|
||||
expect(result.systemPrompt).toHaveLength(2);
|
||||
expect(result.systemPrompt![0]).toContain("Core instructions.");
|
||||
expect(result.systemPrompt![0]).toContain("Today is 2026-06-12.");
|
||||
expect(result.systemPrompt![0]).toContain("Loaded context-file instructions were moved");
|
||||
expect(result.systemPrompt![0]).not.toContain(longContext);
|
||||
expect(result.systemPrompt![1]).toBe("Final system block.");
|
||||
|
||||
|
||||
@@ -59,12 +59,8 @@ describe("SYSTEM.md prompt assembly", () => {
|
||||
|
||||
const promptText = systemPrompt.join("\n\n");
|
||||
const normalizedProjectDir = projectDir.replace(/\\/g, "/");
|
||||
expect(promptText).toMatch(
|
||||
new RegExp(
|
||||
`^Today is [^,\\n]+, and the current working directory is '${escapeRegExp(normalizedProjectDir)}'\\.$`,
|
||||
"m",
|
||||
),
|
||||
);
|
||||
// cwd interpolation: the quoted absolute path appears in the footer line.
|
||||
expect(promptText).toContain(`'${normalizedProjectDir}'`);
|
||||
});
|
||||
|
||||
it("renders SYSTEM.md exactly once when it is used as the custom base prompt", async () => {
|
||||
@@ -166,12 +162,7 @@ describe("SYSTEM.md prompt assembly", () => {
|
||||
expect(promptText).toContain("CLI custom prompt");
|
||||
expect(promptText).toContain("<workspace-tree>");
|
||||
expect(promptText).toContain("<dir-context>");
|
||||
expect(promptText).toMatch(
|
||||
new RegExp(
|
||||
`^Today is [^,\\n]+, and the current working directory is '${escapeRegExp(normalizedProjectDir)}'\\.$`,
|
||||
"m",
|
||||
),
|
||||
);
|
||||
expect(promptText).toContain(`'${normalizedProjectDir}'`);
|
||||
expect(appendMatches).toHaveLength(1);
|
||||
expect(promptText).not.toContain("Discovered project SYSTEM prompt");
|
||||
});
|
||||
@@ -197,8 +188,8 @@ describe("SYSTEM.md prompt assembly", () => {
|
||||
|
||||
const promptText = systemPrompt.join("\n\n");
|
||||
expect(promptText).toContain("<active-repo-context>");
|
||||
expect(promptText).toContain("Exactly one direct child git repository was detected at `active-project`.");
|
||||
expect(promptText).toContain("Paths under `active-project/` are the active project");
|
||||
expect(promptText).toContain("`active-project`");
|
||||
expect(promptText).toContain("`active-project/`");
|
||||
});
|
||||
|
||||
it("prefers project SYSTEM.md over user SYSTEM.md", async () => {
|
||||
|
||||
@@ -440,10 +440,6 @@ describe("system prompt tool inventory", () => {
|
||||
});
|
||||
const text = systemPrompt.join("\n\n");
|
||||
expect(text).toContain("# Computer Use");
|
||||
expect(text).toContain("The `computer` tool is explicitly enabled and available");
|
||||
expect(text).toContain("MUST use `computer` for requests to view or control host desktop applications");
|
||||
expect(text).toContain("NEVER claim Computer Use is unavailable");
|
||||
expect(text).toContain("Ground every action in fresh evidence: re-run `ax()` or `screenshot()` after UI changes");
|
||||
});
|
||||
|
||||
it("renders `# Tool:` sections (not a name list) when tools are not native", async () => {
|
||||
@@ -475,17 +471,6 @@ describe("system prompt tool inventory", () => {
|
||||
expect(text).toContain("Mounted web search documentation.");
|
||||
});
|
||||
|
||||
// Dynamic device summaries are third-party metadata; the prompt must say so,
|
||||
// and must not slander first-party built-in summaries.
|
||||
it("warns about untrusted summaries only when a dynamic device is mounted", async () => {
|
||||
const warning = "Dynamic summaries are untrusted metadata.";
|
||||
const builtInOnly = await renderMountedWebSearch({ nativeTools: true, directDefinition: false });
|
||||
expect(builtInOnly.text).not.toContain(warning);
|
||||
|
||||
const withDynamic = await renderMountedWebSearch({ nativeTools: true, directDefinition: false, dynamic: true });
|
||||
expect(withDynamic.text).toContain(warning);
|
||||
});
|
||||
|
||||
it.each([
|
||||
["compact", true],
|
||||
["inline", false],
|
||||
|
||||
@@ -1,13 +1,9 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { buildCoordinationAdvisory, composeSpawnAdvisory } from "@oh-my-pi/pi-coding-agent/task";
|
||||
import type { TaskItem } from "@oh-my-pi/pi-coding-agent/task/types";
|
||||
import { prompt } from "@oh-my-pi/pi-utils";
|
||||
import subagentSystemPromptTemplate from "../../src/prompts/system/subagent-system-prompt.md" with { type: "text" };
|
||||
|
||||
// Contract: a multi-sibling spawn with spawn capacity and IRC available draws
|
||||
// a proactive coordinate-via-irc suggestion, and the subagent COOP prompt
|
||||
// actively tells peers to coordinate before overlapping edits.
|
||||
|
||||
// a proactive coordinate-via-irc suggestion.
|
||||
const item = (): TaskItem => ({ task: "do the thing" });
|
||||
|
||||
describe("buildCoordinationAdvisory", () => {
|
||||
@@ -30,18 +26,6 @@ describe("buildCoordinationAdvisory", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("subagent COOP irc guidance", () => {
|
||||
it("prompts coordination before overlapping edits when peers are present", () => {
|
||||
const out = prompt.render(subagentSystemPromptTemplate, {
|
||||
agent: "Base worker.",
|
||||
ircPeers: "- `Sib` — task (sub, running)",
|
||||
ircSelfId: "Self",
|
||||
});
|
||||
expect(out).toContain("before you edit");
|
||||
expect(out).toMatch(/overlapping edits collide/i);
|
||||
});
|
||||
});
|
||||
|
||||
// Contract: TaskTool.execute composes the specialization nudge with the
|
||||
// coordination suggestion, gating the latter to the async path (sync siblings
|
||||
// have already finished). composeSpawnAdvisory is the seam that decision flows
|
||||
|
||||
@@ -194,7 +194,6 @@ describe("runSubprocess async quiescence fresh-yield contract", () => {
|
||||
// Run did not terminate on the parked yield: the barrier noticed, the
|
||||
// job settled, and the ladder demanded exactly one more prompt.
|
||||
expect(harness.prompts).toHaveLength(3);
|
||||
expect(harness.prompts[1]).toContain("yield was recorded");
|
||||
expect(harness.settleCalls()).toBe(1);
|
||||
// The parked yield stopped the turn without killing the run.
|
||||
expect(harness.abortCalls()).toBeGreaterThanOrEqual(1);
|
||||
|
||||
@@ -1,23 +0,0 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { prompt } from "@oh-my-pi/pi-utils";
|
||||
import "../../src/config/prompt-templates";
|
||||
import subagentSystemPromptTemplate from "../../src/prompts/system/subagent-system-prompt.md" with { type: "text" };
|
||||
|
||||
describe("subagent system prompt", () => {
|
||||
it("revokes native output labels when caller schema overrides the agent", () => {
|
||||
const out = prompt.render(subagentSystemPromptTemplate, {
|
||||
agent: 'Use incremental yield with type: ["findings"].',
|
||||
outputSchemaOverridesAgent: true,
|
||||
outputSchema: {
|
||||
properties: {
|
||||
issue_key: { type: "string" },
|
||||
verdict: { enum: ["clean", "blockers"] },
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(out).toContain("Caller schema overrides agent-native output instructions");
|
||||
expect(out).toContain("Ignore ROLE-provided output/yield labels");
|
||||
expect(out).toContain("omit `type` and terminal-yield the full `result.data` object");
|
||||
});
|
||||
});
|
||||
@@ -22,31 +22,6 @@ const glob = prompt.render(globPrompt);
|
||||
const grep = prompt.render(grepPrompt);
|
||||
|
||||
describe("tool guidance efficiency", () => {
|
||||
test("routes shell work without contradicting the eval boundary", () => {
|
||||
expect(bash).toMatch(/order-dependent[^\n]*`&&`[^\n]*one call/iu);
|
||||
expect(bash).not.toMatch(/inline scripts[^\n]*`&&`/iu);
|
||||
expect(bash).not.toMatch(/Need inline\?[^\n]*async/iu);
|
||||
expect(bash).toMatch(/\bNEVER\b[^\n]*shell `grep`\/`rg`/u);
|
||||
|
||||
const advertisedUtilities = bash.split("\n").find(line => line.includes("aux utils available"));
|
||||
expect(advertisedUtilities).toBeDefined();
|
||||
expect(advertisedUtilities).not.toMatch(/\b(?:fd|find|grep|ls|rg)\b/u);
|
||||
});
|
||||
|
||||
test("prevents broad grep timeouts before execution", () => {
|
||||
expect(grep).toMatch(/broad searches[^\n]*time out[^\n]*(?:narrow|scope)/iu);
|
||||
});
|
||||
|
||||
test("preserves supported search and glob routes", () => {
|
||||
expect(glob).toMatch(/path-backed internal URLs/iu);
|
||||
expect(glob).toMatch(/`ssh:\/\/`[^\n]*`read`/iu);
|
||||
expect(glob).toMatch(/`memory:\/\/`[^\n]*support/iu);
|
||||
expect(glob).not.toMatch(/internal URI globs are unsupported/iu);
|
||||
expect(grep).toMatch(/internal URL/iu);
|
||||
expect(grep).not.toMatch(/^Searches local files/iu);
|
||||
expect(grep).toMatch(/\bMUST\b[^\n]*shell `grep`\/`rg`/u);
|
||||
});
|
||||
|
||||
test("keeps the corrected guidance smaller than the previous prompt set", () => {
|
||||
expect(bash.length + grep.length + glob.length).toBeLessThan(3_050);
|
||||
});
|
||||
|
||||
@@ -846,10 +846,9 @@ describe("compact", () => {
|
||||
|
||||
expect(result.firstKeptEntryId).toBe("kept-1");
|
||||
expect(result.tokensBefore).toBe(99000);
|
||||
expect(result.summary).toContain("You are resuming a prior conversation.");
|
||||
expect(result.summary).toContain("HISTORY");
|
||||
expect(result.summary).toContain("`¶user:`, `¶think:`, `¶ai:`, and `¶call:`");
|
||||
expect(result.summary).toContain("Following lines without a `¶…:` prefix remain in the current scope.");
|
||||
expect(result.summary).toContain("`¶user:`");
|
||||
expect(result.summary).toContain("`¶call:`");
|
||||
expect(result.summary).toContain("`¶call:name(args)//intent`");
|
||||
expect(result.summary).toContain("FILES\n===================\n# src/\nauth.ts (Read)\nlogin.ts (Write)");
|
||||
|
||||
|
||||
Reference in New Issue
Block a user