test(coding-agent): updated test suites and assertions for coding agent

- Updated eager compaction and plan reference tests to track call indices and task delegation markers instead of text strings.
- Removed obsolete context message marker checks, vibe mode assertions, and prompt gating test cases.
- Simplified prewalk, workflow, and Gemini instruction test expectations across agent modules.
- Removed the system prompt personality test suite entirely.
This commit is contained in:
can1357
2026-08-12 03:12:15 +02:00
parent 63b39b7040
commit b6dc98eb63
10 changed files with 49 additions and 241 deletions
@@ -21,9 +21,10 @@ import { TempDir } from "@oh-my-pi/pi-utils";
// the delegate-via-tasks / phased-todo guidance. The post-compaction auto-continuation
// turn must carry the gated reminders again (reminder-only — never a forced tool_choice).
const CONTINUE_MARKER = "Resume work on the user's most recent intent";
const TASK_DELEGATION_MARKER = "Task delegation enabled";
type ObservedPromptCall = {
callIndex: number;
toolChoice: string | undefined;
messageTexts: string[];
};
@@ -202,6 +203,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
getToolChoice: () => session?.nextToolChoiceDirective(),
streamFn: (_model, context, options) => {
const call: ObservedPromptCall = {
callIndex: observedCalls.length,
toolChoice: getToolChoiceName(options?.toolChoice),
messageTexts: context.messages.map(message => getMessageText(message)),
};
@@ -276,7 +278,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
activateOngoingGoal(session);
await session.prompt("refactor the parser across modules");
emitHighUsageTurn(session);
return waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER)));
return waitForCall(call => call.callIndex > 0);
}
it("re-injects the eager task reminder on the auto-continuation turn (task.eager always)", async () => {
@@ -285,7 +287,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
const continuation = await runToContinuation(session, waitForCall);
const reminder = continuation.messageTexts.find(text => text.includes("delegation is enabled"));
const reminder = continuation.messageTexts.find(text => text.includes(TASK_DELEGATION_MARKER));
expect(reminder).toBeDefined();
expect(reminder).toContain("`task`");
// Reminder-only: the post-compaction nudge never forces a tool on the resumed turn.
@@ -298,7 +300,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
const continuation = await runToContinuation(session, waitForCall);
expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false);
expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false);
});
it("does not re-inject the eager task reminder when task.eager is preferred", async () => {
@@ -307,7 +309,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
const continuation = await runToContinuation(session, waitForCall);
expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false);
expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false);
});
it("does not re-inject the eager task reminder for subagent sessions", async () => {
@@ -316,7 +318,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
const continuation = await runToContinuation(session, waitForCall);
expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false);
expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false);
});
it("does not re-inject the eager task reminder in plan mode", async () => {
@@ -326,7 +328,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
const continuation = await runToContinuation(session, waitForCall);
expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false);
expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false);
});
it("re-injects the eager todo reminder on the auto-continuation turn (todo.eager preferred)", async () => {
@@ -373,7 +375,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
});
stubCompaction(todoEntryId);
const continuationPromise = waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER)));
const continuationPromise = waitForCall(call => call.callIndex > 0);
emitHighUsageTurn(session);
const continuation = await continuationPromise;
@@ -30,9 +30,7 @@ import { AuthStorage } from "../src/session/auth-storage";
import { convertToLlm } from "../src/session/messages";
import { SessionManager } from "../src/session/session-manager";
const CONTINUE_MARKER = "Resume work on the user's most recent intent";
type ObservedPromptCall = { messageTexts: string[] };
type ObservedPromptCall = { callIndex: number; messageTexts: string[] };
type Harness = {
session: AgentSession;
@@ -163,8 +161,11 @@ describe("AgentSession approved-plan reference re-injection after compaction (is
convertToLlm,
getToolChoice: () => session?.nextToolChoiceDirective(),
streamFn: (_model, context) => {
observedCalls.push({ messageTexts: context.messages.map(message => getMessageText(message)) });
const call = observedCalls[observedCalls.length - 1];
const call = {
callIndex: observedCalls.length,
messageTexts: context.messages.map(message => getMessageText(message)),
};
observedCalls.push(call);
for (let i = waiters.length - 1; i >= 0; i--) {
const waiter = waiters[i];
if (waiter?.predicate(call)) {
@@ -256,7 +257,7 @@ describe("AgentSession approved-plan reference re-injection after compaction (is
// Auto-compaction fires, replacing history (dropping the delivered reference),
// then schedules the auto-continuation turn.
emitHighUsageTurn(session);
const continuation = await waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER)));
const continuation = await waitForCall(call => call.callIndex > 0);
// The post-compaction continuation MUST carry the durable plan reference again.
expect(continuation.messageTexts.some(text => text.includes(planMarker))).toBe(false);
@@ -280,7 +281,7 @@ describe("AgentSession approved-plan reference re-injection after compaction (is
expect(firstCall.messageTexts.some(text => text.includes(planMarker))).toBe(false);
emitHighUsageTurn(session);
const continuation = await waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER)));
const continuation = await waitForCall(call => call.callIndex > 0);
expect(continuation.messageTexts.some(text => text.includes(planMarker))).toBe(false);
expect(continuation.messageTexts.some(text => text.includes(planUrl))).toBe(true);
@@ -297,7 +298,7 @@ describe("AgentSession approved-plan reference re-injection after compaction (is
await session.prompt("do some ordinary work");
emitHighUsageTurn(session);
const continuation = await waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER)));
const continuation = await waitForCall(call => call.callIndex > 0);
expect(continuation.messageTexts.some(text => text.includes("## Existing Plan"))).toBe(false);
});
@@ -100,31 +100,13 @@ describe("AgentSession prewalk", () => {
return { content: [{ type: "toolCall", id, name, arguments: {} }], stopReason: "toolUse" };
}
function contextMessagesHaveMarker(contextMessages: ReadonlyArray<{ role: string }>, marker: string): boolean {
return contextMessages.some(message => {
if (message.role !== "user" && message.role !== "developer") return false;
if (!("content" in message)) return false;
const content: unknown = message.content;
if (typeof content === "string") return content.includes(marker);
if (!Array.isArray(content)) return false;
return content.some(block => {
if (typeof block !== "object" || block === null) return false;
if (!("type" in block) || block.type !== "text") return false;
return "text" in block && typeof block.text === "string" && block.text.includes(marker);
});
});
}
it("prewalks at the first edit/write after the todo gate opens; bash and todo don't trigger", async () => {
const primary = modelOrThrow("claude-sonnet-4-5");
const target = modelOrThrow("claude-sonnet-4-6");
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
const planMarker = "complete plan in your NEXT reply";
const checklistMarker = "grep for every other call site";
// Turn 1: read-only (nudge injected after). Turn 2: bash — excluded.
// Turn 3: todo — opens the gate, must NOT itself switch. Turn 4: write —
// first post-todo edit/write, switch.
// Turn 1: read-only. Turn 2: bash is excluded. Turn 3: todo opens the gate.
// Turn 4: write is the first post-todo edit/write, so it switches.
const mock = createMockModel({
responses: [
toolCall("t1", "record"),
@@ -134,7 +116,7 @@ describe("AgentSession prewalk", () => {
{ content: ["done"] },
],
});
const calls: Array<{ model: string; hasNudge: boolean; hasChecklist: boolean }> = [];
const calls: string[] = [];
const agent = new Agent({
getApiKey: () => "test-key",
initialState: {
@@ -145,13 +127,9 @@ describe("AgentSession prewalk", () => {
thinkingLevel: Effort.Medium,
},
convertToLlm,
streamFn: (model, context, options) => {
calls.push({
model: `${model.provider}/${model.id}`,
hasNudge: contextMessagesHaveMarker(context.messages, planMarker),
hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker),
});
return mock.stream(model, context, options);
streamFn: (model, _context, options) => {
calls.push(`${model.provider}/${model.id}`);
return mock.stream(model, _context, options);
},
});
session = new AgentSession({
@@ -165,17 +143,13 @@ describe("AgentSession prewalk", () => {
await session.prompt("do the task");
expect(calls.map(call => call.model)).toEqual([
expect(calls).toEqual([
`${primary.provider}/${primary.id}`,
`${primary.provider}/${primary.id}`,
`${primary.provider}/${primary.id}`,
`${primary.provider}/${primary.id}`,
`${target.provider}/${target.id}`,
]);
// Nudge absent on turn 1 (not yet injected), present turns 2-4, scrubbed after the switch.
expect(calls.map(call => call.hasNudge)).toEqual([false, true, true, true, false]);
// Checklist present only once the target model is running.
expect(calls.map(call => call.hasChecklist)).toEqual([false, false, false, false, true]);
expect(session.model?.id).toBe(target.id);
});
@@ -183,11 +157,9 @@ describe("AgentSession prewalk", () => {
const primary = modelOrThrow("claude-sonnet-4-5");
const target = modelOrThrow("claude-sonnet-4-6");
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
const planMarker = "complete plan in your NEXT reply";
// Turn 1: exploration (nudge after). Turn 2: write with the gate still
// closed — no switch; the fast model must not inherit a todo-less run.
// Turn 3: todo — gate opens. Turn 4: write — switch.
// Turn 1: exploration. Turn 2: write while the gate is closed.
// Turn 3: todo opens the gate. Turn 4: write switches.
const mock = createMockModel({
responses: [
toolCall("t1", "record"),
@@ -197,7 +169,7 @@ describe("AgentSession prewalk", () => {
{ content: ["done"] },
],
});
const calls: Array<{ model: string; hasNudge: boolean }> = [];
const calls: string[] = [];
const agent = new Agent({
getApiKey: () => "test-key",
initialState: {
@@ -208,12 +180,9 @@ describe("AgentSession prewalk", () => {
thinkingLevel: Effort.Medium,
},
convertToLlm,
streamFn: (model, context, options) => {
calls.push({
model: `${model.provider}/${model.id}`,
hasNudge: contextMessagesHaveMarker(context.messages, planMarker),
});
return mock.stream(model, context, options);
streamFn: (model, _context, options) => {
calls.push(`${model.provider}/${model.id}`);
return mock.stream(model, _context, options);
},
});
session = new AgentSession({
@@ -227,15 +196,13 @@ describe("AgentSession prewalk", () => {
await session.prompt("do the task");
expect(calls.map(call => call.model)).toEqual([
expect(calls).toEqual([
`${primary.provider}/${primary.id}`,
`${primary.provider}/${primary.id}`,
`${primary.provider}/${primary.id}`,
`${primary.provider}/${primary.id}`,
`${target.provider}/${target.id}`,
]);
// The turn-2 write landed while the gate was closed — still primary on turn 3.
expect(calls.map(call => call.hasNudge)).toEqual([false, true, true, true, false]);
expect(session.model?.id).toBe(target.id);
});
@@ -707,12 +674,10 @@ describe("AgentSession prewalk", () => {
).toHaveLength(1);
});
it("armPrewalk rejects a same-model same-effort no-op before injecting the plan nudge", async () => {
it("armPrewalk rejects a same-model same-effort no-op", async () => {
const model = modelOrThrow("claude-sonnet-4-5");
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
const planMarker = "complete plan in your NEXT reply";
const mock = createMockModel({ responses: [{ content: ["status only"] }] });
const calls: Array<{ hasNudge: boolean }> = [];
const agent = new Agent({
getApiKey: () => "test-key",
initialState: {
@@ -723,10 +688,7 @@ describe("AgentSession prewalk", () => {
thinkingLevel: Effort.Medium,
},
convertToLlm,
streamFn: (streamModel, context, options) => {
calls.push({ hasNudge: contextMessagesHaveMarker(context.messages, planMarker) });
return mock.stream(streamModel, context, options);
},
streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options),
});
session = new AgentSession({
agent,
@@ -744,7 +706,6 @@ describe("AgentSession prewalk", () => {
expect(session.armPrewalk(model, Effort.Medium)).toBe(false);
await session.prompt("report current status");
expect(calls).toEqual([{ hasNudge: false }]);
expect(notices.some(message => message.includes("nothing to switch"))).toBe(true);
});
@@ -866,13 +827,11 @@ describe("AgentSession prewalk", () => {
// this must still switch.
const model = modelOrThrow("claude-sonnet-4-5");
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
const checklistMarker = "grep for every other call site";
// todo excluded from the active slate → the gate opens; record then write.
const mock = createMockModel({
responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }],
});
const calls: Array<{ model: string; hasChecklist: boolean }> = [];
const agent = new Agent({
getApiKey: () => "test-key",
initialState: {
@@ -883,13 +842,7 @@ describe("AgentSession prewalk", () => {
thinkingLevel: Effort.Medium,
},
convertToLlm,
streamFn: (streamModel, context, options) => {
calls.push({
model: `${streamModel.provider}/${streamModel.id}`,
hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker),
});
return mock.stream(streamModel, context, options);
},
streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options),
});
session = new AgentSession({
agent,
@@ -908,23 +861,16 @@ describe("AgentSession prewalk", () => {
// The model id never changes, but the effort drops after the first write.
expect(session.model?.id).toBe(model.id);
expect(session.thinkingLevel).toBe(Effort.Low);
// The switch ran: the post-switch checklist is present on the final turn.
expect(calls.at(-1)?.hasChecklist).toBe(true);
});
it("emits a notice and skips the checklist when the prewalk target is a genuine no-op", async () => {
// Same model AND same effective thinking level: nothing to switch. The
// early return must be visible (a notice), not silent, and must not fire
// the post-switch checklist.
it("emits a notice when the prewalk target is a genuine no-op", async () => {
// Same model and same effective thinking level: no state change.
const model = modelOrThrow("claude-sonnet-4-5");
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
const planMarker = "complete plan in your NEXT reply";
const checklistMarker = "grep for every other call site";
const mock = createMockModel({
responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }],
});
const calls: Array<{ hasNudge: boolean; hasChecklist: boolean }> = [];
const agent = new Agent({
getApiKey: () => "test-key",
initialState: {
@@ -935,13 +881,7 @@ describe("AgentSession prewalk", () => {
thinkingLevel: Effort.Medium,
},
convertToLlm,
streamFn: (streamModel, context, options) => {
calls.push({
hasNudge: contextMessagesHaveMarker(context.messages, planMarker),
hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker),
});
return mock.stream(streamModel, context, options);
},
streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options),
});
session = new AgentSession({
agent,
@@ -963,25 +903,17 @@ describe("AgentSession prewalk", () => {
expect(session.thinkingLevel).toBe(Effort.Medium);
// The no-op is announced, not silent.
expect(notices.some(message => message.includes("nothing to switch"))).toBe(true);
// The checklist steer only fires on a real switch.
// A genuine no-op must be rejected before the disruptive plan nudge is injected.
expect(calls.every(call => !call.hasNudge)).toBe(true);
expect(calls.every(call => !call.hasChecklist)).toBe(true);
});
it("treats a target effort the model clamps back to the active effort as a no-op", async () => {
// Review edge case: a model capped at `high` running at `high` with a
// prewalk target of `xhigh`. The raw selectors differ, but the target
// clamps to `high`, so switching would reset the model and inject the
// nudges for no effective change — it must be recognized as a no-op.
// A model capped at high resolves an xhigh target back to high.
// The equal effective settings must be recognized as a no-op.
const model = modelOrThrow("claude-sonnet-4-6"); // supported efforts cap at high
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
const checklistMarker = "grep for every other call site";
const mock = createMockModel({
responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }],
});
const calls: Array<{ hasChecklist: boolean }> = [];
const agent = new Agent({
getApiKey: () => "test-key",
initialState: {
@@ -992,10 +924,7 @@ describe("AgentSession prewalk", () => {
thinkingLevel: Effort.High,
},
convertToLlm,
streamFn: (streamModel, context, options) => {
calls.push({ hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker) });
return mock.stream(streamModel, context, options);
},
streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options),
});
session = new AgentSession({
agent,
@@ -1015,7 +944,6 @@ describe("AgentSession prewalk", () => {
expect(session.thinkingLevel).toBe(Effort.High);
expect(notices.some(message => message.includes("nothing to switch"))).toBe(true);
expect(calls.every(call => !call.hasChecklist)).toBe(true);
});
it("switches when a same-model target clears auto mode even though efforts both resolve to undefined", async () => {
@@ -1025,12 +953,10 @@ describe("AgentSession prewalk", () => {
// collapse to a no-op.
const model = modelOrThrow("claude-sonnet-4-5");
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
const checklistMarker = "grep for every other call site";
const mock = createMockModel({
responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }],
});
const calls: Array<{ hasChecklist: boolean }> = [];
const agent = new Agent({
getApiKey: () => "test-key",
initialState: {
@@ -1041,10 +967,7 @@ describe("AgentSession prewalk", () => {
thinkingLevel: Effort.Medium,
},
convertToLlm,
streamFn: (streamModel, context, options) => {
calls.push({ hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker) });
return mock.stream(streamModel, context, options);
},
streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options),
});
session = new AgentSession({
agent,
@@ -1064,9 +987,8 @@ describe("AgentSession prewalk", () => {
await session.prompt("do the task");
// The hand-off ran: auto is cleared and the post-switch checklist fired.
// The hand-off clears automatic thinking.
expect(session.isAutoThinking).toBe(false);
expect(notices.some(message => message.includes("nothing to switch"))).toBe(false);
expect(calls.at(-1)?.hasChecklist).toBe(true);
});
});
@@ -17,7 +17,6 @@ import { InteractiveMode } from "@oh-my-pi/pi-coding-agent/modes/interactive-mod
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { normalizeCustomMessagePayload } from "@oh-my-pi/pi-coding-agent/session/messages";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { FileSessionStorage, type WriteTextAtomicOptions } from "@oh-my-pi/pi-coding-agent/session/session-storage";
import { VIBE_TOOL_NAMES } from "@oh-my-pi/pi-coding-agent/tools/vibe";
@@ -150,13 +149,6 @@ describe("InteractiveMode vibe mode toggle", () => {
expect(inMode.toSorted()).toEqual(["read", "todo", ...VIBE_TOOL_NAMES].toSorted());
expect(session.getAllToolNames().toSorted()).toEqual(["read", "todo", ...VIBE_TOOL_NAMES].toSorted());
const sendCustomMessage = vi.spyOn(session, "sendCustomMessage");
await session.sendVibeModeContext({ deliverAs: "steer" });
const message = normalizeCustomMessagePayload(sendCustomMessage.mock.calls[0]?.[0]);
const content = typeof message.content === "string" ? message.content : "";
expect(message.customType).toBe("vibe-mode-context");
expect(content).toContain("`todo`");
// Toggle off: the empty previous toolset must come back — only the
// ephemeral vibe tools must leave the registry.
await mode.handleVibeModeCommand();
@@ -198,13 +190,6 @@ describe("InteractiveMode vibe mode toggle", () => {
await foreignTodoMode.handleVibeModeCommand();
expect(foreignTodoSession.getActiveToolNames().toSorted()).toEqual(["read", ...VIBE_TOOL_NAMES].toSorted());
const sendCustomMessage = vi.spyOn(foreignTodoSession, "sendCustomMessage");
await foreignTodoSession.sendVibeModeContext({ deliverAs: "steer" });
const message = normalizeCustomMessagePayload(sendCustomMessage.mock.calls[0]?.[0]);
const content = typeof message.content === "string" ? message.content : "";
expect(content).not.toContain("`todo`");
expect(content).not.toContain("parent session's list");
await foreignTodoMode.handleVibeModeCommand();
expect(foreignTodoSession.getActiveToolNames()).toEqual([]);
expect(foreignTodoSession.getAllToolNames().toSorted()).toEqual(["read", "todo"]);
@@ -234,12 +219,6 @@ describe("InteractiveMode vibe mode toggle", () => {
expect(mode.vibeModeEnabled).toBe(true);
expect(session.getActiveToolNames()).toEqual(expect.arrayContaining(["read", "todo", ...VIBE_TOOL_NAMES]));
const sendCustomMessage = vi.spyOn(session, "sendCustomMessage");
await session.sendVibeModeContext({ deliverAs: "steer" });
const message = normalizeCustomMessagePayload(sendCustomMessage.mock.calls[0]?.[0]);
const content = typeof message.content === "string" ? message.content : "";
expect(content).toContain("`todo`");
expect(content).toContain("parent session's list");
expect(suspend).toHaveBeenCalledTimes(1);
expect(terminate).not.toHaveBeenCalled();
expect(vibeModeEntryCount(session.sessionManager)).toBe(1);
@@ -60,16 +60,11 @@ test("keeps Gemini 3.6 advisor context and accepts a silent review", async () =>
expect(bodies).toHaveLength(1);
expect(bodies[0]).toMatchObject({
systemInstruction: {
parts: [{ text: expect.stringContaining("You bring a different angle") }],
parts: [{ text: expect.any(String) }],
},
tools: [
{
functionDeclarations: [
{
name: "advise",
description: expect.stringContaining("Send one concrete"),
},
],
functionDeclarations: [{ name: "advise" }],
},
],
});
@@ -1,11 +1,6 @@
import { beforeAll, describe, expect, it } from "bun:test";
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import {
containsWorkflow,
highlightWorkflow,
renderWorkflowNotice,
WORKFLOW_NOTICE,
} from "@oh-my-pi/pi-coding-agent/modes/workflow";
import { containsWorkflow, highlightWorkflow } from "@oh-my-pi/pi-coding-agent/modes/workflow";
beforeAll(() => {
// highlightWorkflow reads the global theme's color mode.
@@ -63,17 +58,3 @@ describe("workflow keyword highlighting", () => {
expect(highlightWorkflow(filePath)).toBe(filePath);
});
});
describe("workflow notice", () => {
it("renders the Workflowz trigger with eval orchestration helper guidance", () => {
expect(WORKFLOW_NOTICE).toContain("**workflowz** keyword");
expect(WORKFLOW_NOTICE).toContain("`parallel(thunks)`");
expect(WORKFLOW_NOTICE).toContain("await budget.remaining()");
});
it("renders the same eval notice when task.batch is disabled", () => {
const notice = renderWorkflowNotice({ taskBatch: false });
expect(notice).toContain("**workflowz** keyword");
expect(notice).toContain("`parallel(thunks)`");
});
});
@@ -1,63 +0,0 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import type { Personality } from "@oh-my-pi/pi-coding-agent/config/settings-schema";
import { buildSystemPrompt } from "@oh-my-pi/pi-coding-agent/system-prompt";
import { cleanupTempHome } from "./helpers/temp-home-cleanup";
const EMPTY_TREE = {
rootPath: "",
rendered: "",
truncated: false,
totalLines: 0,
agentsMdFiles: [],
};
describe("system prompt personality block", () => {
let tempDir = "";
let tempHomeDir = "";
let originalHome: string | undefined;
beforeEach(() => {
tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-prompt-personality-"));
tempHomeDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-prompt-personality-home-"));
originalHome = process.env.HOME;
process.env.HOME = tempHomeDir;
});
afterEach(cleanupTempHome(() => ({ tempDir, tempHomeDir, originalHome })));
async function render(personality?: Personality): Promise<string> {
const { systemPrompt } = await buildSystemPrompt({
cwd: tempDir,
contextFiles: [],
skills: [],
rules: [],
toolNames: [],
workspaceTree: { ...EMPTY_TREE, rootPath: tempDir },
personality,
});
return systemPrompt.join("\n\n");
}
it("injects the default personality when the option is unset", async () => {
const rendered = await render();
expect(rendered).toContain("<personality>");
expect(rendered).toContain("</personality>");
expect(rendered).toContain("terse, evidence-first engineer");
});
it("replaces the default spec when a non-default personality is selected", async () => {
const rendered = await render("friendly");
expect(rendered).toContain("<personality>");
expect(rendered).toContain("warm, supportive collaborator");
expect(rendered).not.toContain("terse, evidence-first engineer");
});
it('omits the personality block entirely for "none"', async () => {
const rendered = await render("none");
expect(rendered).not.toContain("<personality>");
expect(rendered).not.toContain("</personality>");
});
});
@@ -259,9 +259,8 @@ describe("runSubprocess soft request budget", () => {
// wrap-up reminder; the second abort (after the terminal yield) is the
// normal post-yield terminate.
expect(abortCallsAtReminder).toBe(1);
// The wrap-up reminder is the budget-stop variant with a forced tool choice.
// The budget stop forces a synthetic terminal yield.
expect(handle.prompts).toHaveLength(2);
expect(handle.prompts[1]?.text).toMatch(/request budget/);
expect(handle.prompts[1]?.options?.synthetic).toBe(true);
expect(handle.prompts[1]?.options?.toolChoice).toEqual({ type: "tool", name: "yield" });
// The forced yield finalizes as a normal completion, not an abort.
@@ -234,7 +234,7 @@ describe("runSubprocess yield reminders", () => {
expect(systemPrompt).toHaveLength(4);
expect(systemPrompt?.[0]).toBe("system");
expect(systemPrompt?.[1]).toBe("project");
expect(systemPrompt?.[2]).toMatch(/ROLE\n=+\n\ntest/);
expect(systemPrompt?.[2]).toContain(baseAgent.systemPrompt);
// The parent-conversation CONTEXT section is gone: subagents get their
// background inside the assignment (or a local:// file), never a dump.
expect(systemPrompt?.[2]).not.toMatch(/CONTEXT\n=+/);
@@ -1,6 +1,5 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { Settings } from "../../src/config/settings";
import initAgentPrompt from "../../src/prompts/agents/init.md" with { type: "text" };
import * as taskDiscovery from "../../src/task/discovery";
import { TaskTool } from "../../src/task/index";
import { isScoutSpawnable } from "../../src/task/spawn-policy";
@@ -131,10 +130,3 @@ describe("task tool description scout gating", () => {
expect(description).toContain("### reviewer");
});
});
describe("bundled agent prompt scout gating", () => {
it("does not hard-code scout in the init agent prompt", () => {
expect(initAgentPrompt.toLowerCase()).not.toContain("scout");
expect(initAgentPrompt).toContain("multiple research agents");
});
});