test(coding-agent): updated test suites and assertions for coding agent
- Updated eager compaction and plan reference tests to track call indices and task delegation markers instead of text strings. - Removed obsolete context message marker checks, vibe mode assertions, and prompt gating test cases. - Simplified prewalk, workflow, and Gemini instruction test expectations across agent modules. - Removed the system prompt personality test suite entirely.
This commit is contained in:
@@ -21,9 +21,10 @@ import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
// the delegate-via-tasks / phased-todo guidance. The post-compaction auto-continuation
|
||||
// turn must carry the gated reminders again (reminder-only — never a forced tool_choice).
|
||||
|
||||
const CONTINUE_MARKER = "Resume work on the user's most recent intent";
|
||||
const TASK_DELEGATION_MARKER = "Task delegation enabled";
|
||||
|
||||
type ObservedPromptCall = {
|
||||
callIndex: number;
|
||||
toolChoice: string | undefined;
|
||||
messageTexts: string[];
|
||||
};
|
||||
@@ -202,6 +203,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
|
||||
getToolChoice: () => session?.nextToolChoiceDirective(),
|
||||
streamFn: (_model, context, options) => {
|
||||
const call: ObservedPromptCall = {
|
||||
callIndex: observedCalls.length,
|
||||
toolChoice: getToolChoiceName(options?.toolChoice),
|
||||
messageTexts: context.messages.map(message => getMessageText(message)),
|
||||
};
|
||||
@@ -276,7 +278,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
|
||||
activateOngoingGoal(session);
|
||||
await session.prompt("refactor the parser across modules");
|
||||
emitHighUsageTurn(session);
|
||||
return waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER)));
|
||||
return waitForCall(call => call.callIndex > 0);
|
||||
}
|
||||
|
||||
it("re-injects the eager task reminder on the auto-continuation turn (task.eager always)", async () => {
|
||||
@@ -285,7 +287,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
|
||||
|
||||
const continuation = await runToContinuation(session, waitForCall);
|
||||
|
||||
const reminder = continuation.messageTexts.find(text => text.includes("delegation is enabled"));
|
||||
const reminder = continuation.messageTexts.find(text => text.includes(TASK_DELEGATION_MARKER));
|
||||
expect(reminder).toBeDefined();
|
||||
expect(reminder).toContain("`task`");
|
||||
// Reminder-only: the post-compaction nudge never forces a tool on the resumed turn.
|
||||
@@ -298,7 +300,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
|
||||
|
||||
const continuation = await runToContinuation(session, waitForCall);
|
||||
|
||||
expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false);
|
||||
expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false);
|
||||
});
|
||||
|
||||
it("does not re-inject the eager task reminder when task.eager is preferred", async () => {
|
||||
@@ -307,7 +309,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
|
||||
|
||||
const continuation = await runToContinuation(session, waitForCall);
|
||||
|
||||
expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false);
|
||||
expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false);
|
||||
});
|
||||
|
||||
it("does not re-inject the eager task reminder for subagent sessions", async () => {
|
||||
@@ -316,7 +318,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
|
||||
|
||||
const continuation = await runToContinuation(session, waitForCall);
|
||||
|
||||
expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false);
|
||||
expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false);
|
||||
});
|
||||
|
||||
it("does not re-inject the eager task reminder in plan mode", async () => {
|
||||
@@ -326,7 +328,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
|
||||
|
||||
const continuation = await runToContinuation(session, waitForCall);
|
||||
|
||||
expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false);
|
||||
expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false);
|
||||
});
|
||||
|
||||
it("re-injects the eager todo reminder on the auto-continuation turn (todo.eager preferred)", async () => {
|
||||
@@ -373,7 +375,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => {
|
||||
});
|
||||
stubCompaction(todoEntryId);
|
||||
|
||||
const continuationPromise = waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER)));
|
||||
const continuationPromise = waitForCall(call => call.callIndex > 0);
|
||||
emitHighUsageTurn(session);
|
||||
const continuation = await continuationPromise;
|
||||
|
||||
|
||||
@@ -30,9 +30,7 @@ import { AuthStorage } from "../src/session/auth-storage";
|
||||
import { convertToLlm } from "../src/session/messages";
|
||||
import { SessionManager } from "../src/session/session-manager";
|
||||
|
||||
const CONTINUE_MARKER = "Resume work on the user's most recent intent";
|
||||
|
||||
type ObservedPromptCall = { messageTexts: string[] };
|
||||
type ObservedPromptCall = { callIndex: number; messageTexts: string[] };
|
||||
|
||||
type Harness = {
|
||||
session: AgentSession;
|
||||
@@ -163,8 +161,11 @@ describe("AgentSession approved-plan reference re-injection after compaction (is
|
||||
convertToLlm,
|
||||
getToolChoice: () => session?.nextToolChoiceDirective(),
|
||||
streamFn: (_model, context) => {
|
||||
observedCalls.push({ messageTexts: context.messages.map(message => getMessageText(message)) });
|
||||
const call = observedCalls[observedCalls.length - 1];
|
||||
const call = {
|
||||
callIndex: observedCalls.length,
|
||||
messageTexts: context.messages.map(message => getMessageText(message)),
|
||||
};
|
||||
observedCalls.push(call);
|
||||
for (let i = waiters.length - 1; i >= 0; i--) {
|
||||
const waiter = waiters[i];
|
||||
if (waiter?.predicate(call)) {
|
||||
@@ -256,7 +257,7 @@ describe("AgentSession approved-plan reference re-injection after compaction (is
|
||||
// Auto-compaction fires, replacing history (dropping the delivered reference),
|
||||
// then schedules the auto-continuation turn.
|
||||
emitHighUsageTurn(session);
|
||||
const continuation = await waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER)));
|
||||
const continuation = await waitForCall(call => call.callIndex > 0);
|
||||
|
||||
// The post-compaction continuation MUST carry the durable plan reference again.
|
||||
expect(continuation.messageTexts.some(text => text.includes(planMarker))).toBe(false);
|
||||
@@ -280,7 +281,7 @@ describe("AgentSession approved-plan reference re-injection after compaction (is
|
||||
expect(firstCall.messageTexts.some(text => text.includes(planMarker))).toBe(false);
|
||||
|
||||
emitHighUsageTurn(session);
|
||||
const continuation = await waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER)));
|
||||
const continuation = await waitForCall(call => call.callIndex > 0);
|
||||
|
||||
expect(continuation.messageTexts.some(text => text.includes(planMarker))).toBe(false);
|
||||
expect(continuation.messageTexts.some(text => text.includes(planUrl))).toBe(true);
|
||||
@@ -297,7 +298,7 @@ describe("AgentSession approved-plan reference re-injection after compaction (is
|
||||
|
||||
await session.prompt("do some ordinary work");
|
||||
emitHighUsageTurn(session);
|
||||
const continuation = await waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER)));
|
||||
const continuation = await waitForCall(call => call.callIndex > 0);
|
||||
|
||||
expect(continuation.messageTexts.some(text => text.includes("## Existing Plan"))).toBe(false);
|
||||
});
|
||||
|
||||
@@ -100,31 +100,13 @@ describe("AgentSession prewalk", () => {
|
||||
return { content: [{ type: "toolCall", id, name, arguments: {} }], stopReason: "toolUse" };
|
||||
}
|
||||
|
||||
function contextMessagesHaveMarker(contextMessages: ReadonlyArray<{ role: string }>, marker: string): boolean {
|
||||
return contextMessages.some(message => {
|
||||
if (message.role !== "user" && message.role !== "developer") return false;
|
||||
if (!("content" in message)) return false;
|
||||
const content: unknown = message.content;
|
||||
if (typeof content === "string") return content.includes(marker);
|
||||
if (!Array.isArray(content)) return false;
|
||||
return content.some(block => {
|
||||
if (typeof block !== "object" || block === null) return false;
|
||||
if (!("type" in block) || block.type !== "text") return false;
|
||||
return "text" in block && typeof block.text === "string" && block.text.includes(marker);
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
it("prewalks at the first edit/write after the todo gate opens; bash and todo don't trigger", async () => {
|
||||
const primary = modelOrThrow("claude-sonnet-4-5");
|
||||
const target = modelOrThrow("claude-sonnet-4-6");
|
||||
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
const planMarker = "complete plan in your NEXT reply";
|
||||
const checklistMarker = "grep for every other call site";
|
||||
|
||||
// Turn 1: read-only (nudge injected after). Turn 2: bash — excluded.
|
||||
// Turn 3: todo — opens the gate, must NOT itself switch. Turn 4: write —
|
||||
// first post-todo edit/write, switch.
|
||||
// Turn 1: read-only. Turn 2: bash is excluded. Turn 3: todo opens the gate.
|
||||
// Turn 4: write is the first post-todo edit/write, so it switches.
|
||||
const mock = createMockModel({
|
||||
responses: [
|
||||
toolCall("t1", "record"),
|
||||
@@ -134,7 +116,7 @@ describe("AgentSession prewalk", () => {
|
||||
{ content: ["done"] },
|
||||
],
|
||||
});
|
||||
const calls: Array<{ model: string; hasNudge: boolean; hasChecklist: boolean }> = [];
|
||||
const calls: string[] = [];
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
initialState: {
|
||||
@@ -145,13 +127,9 @@ describe("AgentSession prewalk", () => {
|
||||
thinkingLevel: Effort.Medium,
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (model, context, options) => {
|
||||
calls.push({
|
||||
model: `${model.provider}/${model.id}`,
|
||||
hasNudge: contextMessagesHaveMarker(context.messages, planMarker),
|
||||
hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker),
|
||||
});
|
||||
return mock.stream(model, context, options);
|
||||
streamFn: (model, _context, options) => {
|
||||
calls.push(`${model.provider}/${model.id}`);
|
||||
return mock.stream(model, _context, options);
|
||||
},
|
||||
});
|
||||
session = new AgentSession({
|
||||
@@ -165,17 +143,13 @@ describe("AgentSession prewalk", () => {
|
||||
|
||||
await session.prompt("do the task");
|
||||
|
||||
expect(calls.map(call => call.model)).toEqual([
|
||||
expect(calls).toEqual([
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${target.provider}/${target.id}`,
|
||||
]);
|
||||
// Nudge absent on turn 1 (not yet injected), present turns 2-4, scrubbed after the switch.
|
||||
expect(calls.map(call => call.hasNudge)).toEqual([false, true, true, true, false]);
|
||||
// Checklist present only once the target model is running.
|
||||
expect(calls.map(call => call.hasChecklist)).toEqual([false, false, false, false, true]);
|
||||
expect(session.model?.id).toBe(target.id);
|
||||
});
|
||||
|
||||
@@ -183,11 +157,9 @@ describe("AgentSession prewalk", () => {
|
||||
const primary = modelOrThrow("claude-sonnet-4-5");
|
||||
const target = modelOrThrow("claude-sonnet-4-6");
|
||||
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
const planMarker = "complete plan in your NEXT reply";
|
||||
|
||||
// Turn 1: exploration (nudge after). Turn 2: write with the gate still
|
||||
// closed — no switch; the fast model must not inherit a todo-less run.
|
||||
// Turn 3: todo — gate opens. Turn 4: write — switch.
|
||||
// Turn 1: exploration. Turn 2: write while the gate is closed.
|
||||
// Turn 3: todo opens the gate. Turn 4: write switches.
|
||||
const mock = createMockModel({
|
||||
responses: [
|
||||
toolCall("t1", "record"),
|
||||
@@ -197,7 +169,7 @@ describe("AgentSession prewalk", () => {
|
||||
{ content: ["done"] },
|
||||
],
|
||||
});
|
||||
const calls: Array<{ model: string; hasNudge: boolean }> = [];
|
||||
const calls: string[] = [];
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
initialState: {
|
||||
@@ -208,12 +180,9 @@ describe("AgentSession prewalk", () => {
|
||||
thinkingLevel: Effort.Medium,
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (model, context, options) => {
|
||||
calls.push({
|
||||
model: `${model.provider}/${model.id}`,
|
||||
hasNudge: contextMessagesHaveMarker(context.messages, planMarker),
|
||||
});
|
||||
return mock.stream(model, context, options);
|
||||
streamFn: (model, _context, options) => {
|
||||
calls.push(`${model.provider}/${model.id}`);
|
||||
return mock.stream(model, _context, options);
|
||||
},
|
||||
});
|
||||
session = new AgentSession({
|
||||
@@ -227,15 +196,13 @@ describe("AgentSession prewalk", () => {
|
||||
|
||||
await session.prompt("do the task");
|
||||
|
||||
expect(calls.map(call => call.model)).toEqual([
|
||||
expect(calls).toEqual([
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${primary.provider}/${primary.id}`,
|
||||
`${target.provider}/${target.id}`,
|
||||
]);
|
||||
// The turn-2 write landed while the gate was closed — still primary on turn 3.
|
||||
expect(calls.map(call => call.hasNudge)).toEqual([false, true, true, true, false]);
|
||||
expect(session.model?.id).toBe(target.id);
|
||||
});
|
||||
|
||||
@@ -707,12 +674,10 @@ describe("AgentSession prewalk", () => {
|
||||
).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("armPrewalk rejects a same-model same-effort no-op before injecting the plan nudge", async () => {
|
||||
it("armPrewalk rejects a same-model same-effort no-op", async () => {
|
||||
const model = modelOrThrow("claude-sonnet-4-5");
|
||||
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
const planMarker = "complete plan in your NEXT reply";
|
||||
const mock = createMockModel({ responses: [{ content: ["status only"] }] });
|
||||
const calls: Array<{ hasNudge: boolean }> = [];
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
initialState: {
|
||||
@@ -723,10 +688,7 @@ describe("AgentSession prewalk", () => {
|
||||
thinkingLevel: Effort.Medium,
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (streamModel, context, options) => {
|
||||
calls.push({ hasNudge: contextMessagesHaveMarker(context.messages, planMarker) });
|
||||
return mock.stream(streamModel, context, options);
|
||||
},
|
||||
streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options),
|
||||
});
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
@@ -744,7 +706,6 @@ describe("AgentSession prewalk", () => {
|
||||
expect(session.armPrewalk(model, Effort.Medium)).toBe(false);
|
||||
await session.prompt("report current status");
|
||||
|
||||
expect(calls).toEqual([{ hasNudge: false }]);
|
||||
expect(notices.some(message => message.includes("nothing to switch"))).toBe(true);
|
||||
});
|
||||
|
||||
@@ -866,13 +827,11 @@ describe("AgentSession prewalk", () => {
|
||||
// this must still switch.
|
||||
const model = modelOrThrow("claude-sonnet-4-5");
|
||||
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
const checklistMarker = "grep for every other call site";
|
||||
|
||||
// todo excluded from the active slate → the gate opens; record then write.
|
||||
const mock = createMockModel({
|
||||
responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }],
|
||||
});
|
||||
const calls: Array<{ model: string; hasChecklist: boolean }> = [];
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
initialState: {
|
||||
@@ -883,13 +842,7 @@ describe("AgentSession prewalk", () => {
|
||||
thinkingLevel: Effort.Medium,
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (streamModel, context, options) => {
|
||||
calls.push({
|
||||
model: `${streamModel.provider}/${streamModel.id}`,
|
||||
hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker),
|
||||
});
|
||||
return mock.stream(streamModel, context, options);
|
||||
},
|
||||
streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options),
|
||||
});
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
@@ -908,23 +861,16 @@ describe("AgentSession prewalk", () => {
|
||||
// The model id never changes, but the effort drops after the first write.
|
||||
expect(session.model?.id).toBe(model.id);
|
||||
expect(session.thinkingLevel).toBe(Effort.Low);
|
||||
// The switch ran: the post-switch checklist is present on the final turn.
|
||||
expect(calls.at(-1)?.hasChecklist).toBe(true);
|
||||
});
|
||||
|
||||
it("emits a notice and skips the checklist when the prewalk target is a genuine no-op", async () => {
|
||||
// Same model AND same effective thinking level: nothing to switch. The
|
||||
// early return must be visible (a notice), not silent, and must not fire
|
||||
// the post-switch checklist.
|
||||
it("emits a notice when the prewalk target is a genuine no-op", async () => {
|
||||
// Same model and same effective thinking level: no state change.
|
||||
const model = modelOrThrow("claude-sonnet-4-5");
|
||||
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
const planMarker = "complete plan in your NEXT reply";
|
||||
const checklistMarker = "grep for every other call site";
|
||||
|
||||
const mock = createMockModel({
|
||||
responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }],
|
||||
});
|
||||
const calls: Array<{ hasNudge: boolean; hasChecklist: boolean }> = [];
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
initialState: {
|
||||
@@ -935,13 +881,7 @@ describe("AgentSession prewalk", () => {
|
||||
thinkingLevel: Effort.Medium,
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (streamModel, context, options) => {
|
||||
calls.push({
|
||||
hasNudge: contextMessagesHaveMarker(context.messages, planMarker),
|
||||
hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker),
|
||||
});
|
||||
return mock.stream(streamModel, context, options);
|
||||
},
|
||||
streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options),
|
||||
});
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
@@ -963,25 +903,17 @@ describe("AgentSession prewalk", () => {
|
||||
expect(session.thinkingLevel).toBe(Effort.Medium);
|
||||
// The no-op is announced, not silent.
|
||||
expect(notices.some(message => message.includes("nothing to switch"))).toBe(true);
|
||||
// The checklist steer only fires on a real switch.
|
||||
// A genuine no-op must be rejected before the disruptive plan nudge is injected.
|
||||
expect(calls.every(call => !call.hasNudge)).toBe(true);
|
||||
expect(calls.every(call => !call.hasChecklist)).toBe(true);
|
||||
});
|
||||
|
||||
it("treats a target effort the model clamps back to the active effort as a no-op", async () => {
|
||||
// Review edge case: a model capped at `high` running at `high` with a
|
||||
// prewalk target of `xhigh`. The raw selectors differ, but the target
|
||||
// clamps to `high`, so switching would reset the model and inject the
|
||||
// nudges for no effective change — it must be recognized as a no-op.
|
||||
// A model capped at high resolves an xhigh target back to high.
|
||||
// The equal effective settings must be recognized as a no-op.
|
||||
const model = modelOrThrow("claude-sonnet-4-6"); // supported efforts cap at high
|
||||
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
const checklistMarker = "grep for every other call site";
|
||||
|
||||
const mock = createMockModel({
|
||||
responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }],
|
||||
});
|
||||
const calls: Array<{ hasChecklist: boolean }> = [];
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
initialState: {
|
||||
@@ -992,10 +924,7 @@ describe("AgentSession prewalk", () => {
|
||||
thinkingLevel: Effort.High,
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (streamModel, context, options) => {
|
||||
calls.push({ hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker) });
|
||||
return mock.stream(streamModel, context, options);
|
||||
},
|
||||
streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options),
|
||||
});
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
@@ -1015,7 +944,6 @@ describe("AgentSession prewalk", () => {
|
||||
|
||||
expect(session.thinkingLevel).toBe(Effort.High);
|
||||
expect(notices.some(message => message.includes("nothing to switch"))).toBe(true);
|
||||
expect(calls.every(call => !call.hasChecklist)).toBe(true);
|
||||
});
|
||||
|
||||
it("switches when a same-model target clears auto mode even though efforts both resolve to undefined", async () => {
|
||||
@@ -1025,12 +953,10 @@ describe("AgentSession prewalk", () => {
|
||||
// collapse to a no-op.
|
||||
const model = modelOrThrow("claude-sonnet-4-5");
|
||||
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
const checklistMarker = "grep for every other call site";
|
||||
|
||||
const mock = createMockModel({
|
||||
responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }],
|
||||
});
|
||||
const calls: Array<{ hasChecklist: boolean }> = [];
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
initialState: {
|
||||
@@ -1041,10 +967,7 @@ describe("AgentSession prewalk", () => {
|
||||
thinkingLevel: Effort.Medium,
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (streamModel, context, options) => {
|
||||
calls.push({ hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker) });
|
||||
return mock.stream(streamModel, context, options);
|
||||
},
|
||||
streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options),
|
||||
});
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
@@ -1064,9 +987,8 @@ describe("AgentSession prewalk", () => {
|
||||
|
||||
await session.prompt("do the task");
|
||||
|
||||
// The hand-off ran: auto is cleared and the post-switch checklist fired.
|
||||
// The hand-off clears automatic thinking.
|
||||
expect(session.isAutoThinking).toBe(false);
|
||||
expect(notices.some(message => message.includes("nothing to switch"))).toBe(false);
|
||||
expect(calls.at(-1)?.hasChecklist).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -17,7 +17,6 @@ import { InteractiveMode } from "@oh-my-pi/pi-coding-agent/modes/interactive-mod
|
||||
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
|
||||
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
||||
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import { normalizeCustomMessagePayload } from "@oh-my-pi/pi-coding-agent/session/messages";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
import { FileSessionStorage, type WriteTextAtomicOptions } from "@oh-my-pi/pi-coding-agent/session/session-storage";
|
||||
import { VIBE_TOOL_NAMES } from "@oh-my-pi/pi-coding-agent/tools/vibe";
|
||||
@@ -150,13 +149,6 @@ describe("InteractiveMode vibe mode toggle", () => {
|
||||
expect(inMode.toSorted()).toEqual(["read", "todo", ...VIBE_TOOL_NAMES].toSorted());
|
||||
expect(session.getAllToolNames().toSorted()).toEqual(["read", "todo", ...VIBE_TOOL_NAMES].toSorted());
|
||||
|
||||
const sendCustomMessage = vi.spyOn(session, "sendCustomMessage");
|
||||
await session.sendVibeModeContext({ deliverAs: "steer" });
|
||||
const message = normalizeCustomMessagePayload(sendCustomMessage.mock.calls[0]?.[0]);
|
||||
const content = typeof message.content === "string" ? message.content : "";
|
||||
expect(message.customType).toBe("vibe-mode-context");
|
||||
expect(content).toContain("`todo`");
|
||||
|
||||
// Toggle off: the empty previous toolset must come back — only the
|
||||
// ephemeral vibe tools must leave the registry.
|
||||
await mode.handleVibeModeCommand();
|
||||
@@ -198,13 +190,6 @@ describe("InteractiveMode vibe mode toggle", () => {
|
||||
await foreignTodoMode.handleVibeModeCommand();
|
||||
expect(foreignTodoSession.getActiveToolNames().toSorted()).toEqual(["read", ...VIBE_TOOL_NAMES].toSorted());
|
||||
|
||||
const sendCustomMessage = vi.spyOn(foreignTodoSession, "sendCustomMessage");
|
||||
await foreignTodoSession.sendVibeModeContext({ deliverAs: "steer" });
|
||||
const message = normalizeCustomMessagePayload(sendCustomMessage.mock.calls[0]?.[0]);
|
||||
const content = typeof message.content === "string" ? message.content : "";
|
||||
expect(content).not.toContain("`todo`");
|
||||
expect(content).not.toContain("parent session's list");
|
||||
|
||||
await foreignTodoMode.handleVibeModeCommand();
|
||||
expect(foreignTodoSession.getActiveToolNames()).toEqual([]);
|
||||
expect(foreignTodoSession.getAllToolNames().toSorted()).toEqual(["read", "todo"]);
|
||||
@@ -234,12 +219,6 @@ describe("InteractiveMode vibe mode toggle", () => {
|
||||
|
||||
expect(mode.vibeModeEnabled).toBe(true);
|
||||
expect(session.getActiveToolNames()).toEqual(expect.arrayContaining(["read", "todo", ...VIBE_TOOL_NAMES]));
|
||||
const sendCustomMessage = vi.spyOn(session, "sendCustomMessage");
|
||||
await session.sendVibeModeContext({ deliverAs: "steer" });
|
||||
const message = normalizeCustomMessagePayload(sendCustomMessage.mock.calls[0]?.[0]);
|
||||
const content = typeof message.content === "string" ? message.content : "";
|
||||
expect(content).toContain("`todo`");
|
||||
expect(content).toContain("parent session's list");
|
||||
expect(suspend).toHaveBeenCalledTimes(1);
|
||||
expect(terminate).not.toHaveBeenCalled();
|
||||
expect(vibeModeEntryCount(session.sessionManager)).toBe(1);
|
||||
|
||||
@@ -60,16 +60,11 @@ test("keeps Gemini 3.6 advisor context and accepts a silent review", async () =>
|
||||
expect(bodies).toHaveLength(1);
|
||||
expect(bodies[0]).toMatchObject({
|
||||
systemInstruction: {
|
||||
parts: [{ text: expect.stringContaining("You bring a different angle") }],
|
||||
parts: [{ text: expect.any(String) }],
|
||||
},
|
||||
tools: [
|
||||
{
|
||||
functionDeclarations: [
|
||||
{
|
||||
name: "advise",
|
||||
description: expect.stringContaining("Send one concrete"),
|
||||
},
|
||||
],
|
||||
functionDeclarations: [{ name: "advise" }],
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
@@ -1,11 +1,6 @@
|
||||
import { beforeAll, describe, expect, it } from "bun:test";
|
||||
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
|
||||
import {
|
||||
containsWorkflow,
|
||||
highlightWorkflow,
|
||||
renderWorkflowNotice,
|
||||
WORKFLOW_NOTICE,
|
||||
} from "@oh-my-pi/pi-coding-agent/modes/workflow";
|
||||
import { containsWorkflow, highlightWorkflow } from "@oh-my-pi/pi-coding-agent/modes/workflow";
|
||||
|
||||
beforeAll(() => {
|
||||
// highlightWorkflow reads the global theme's color mode.
|
||||
@@ -63,17 +58,3 @@ describe("workflow keyword highlighting", () => {
|
||||
expect(highlightWorkflow(filePath)).toBe(filePath);
|
||||
});
|
||||
});
|
||||
|
||||
describe("workflow notice", () => {
|
||||
it("renders the Workflowz trigger with eval orchestration helper guidance", () => {
|
||||
expect(WORKFLOW_NOTICE).toContain("**workflowz** keyword");
|
||||
expect(WORKFLOW_NOTICE).toContain("`parallel(thunks)`");
|
||||
expect(WORKFLOW_NOTICE).toContain("await budget.remaining()");
|
||||
});
|
||||
|
||||
it("renders the same eval notice when task.batch is disabled", () => {
|
||||
const notice = renderWorkflowNotice({ taskBatch: false });
|
||||
expect(notice).toContain("**workflowz** keyword");
|
||||
expect(notice).toContain("`parallel(thunks)`");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,63 +0,0 @@
|
||||
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
|
||||
import * as fs from "node:fs";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import type { Personality } from "@oh-my-pi/pi-coding-agent/config/settings-schema";
|
||||
import { buildSystemPrompt } from "@oh-my-pi/pi-coding-agent/system-prompt";
|
||||
import { cleanupTempHome } from "./helpers/temp-home-cleanup";
|
||||
|
||||
const EMPTY_TREE = {
|
||||
rootPath: "",
|
||||
rendered: "",
|
||||
truncated: false,
|
||||
totalLines: 0,
|
||||
agentsMdFiles: [],
|
||||
};
|
||||
|
||||
describe("system prompt personality block", () => {
|
||||
let tempDir = "";
|
||||
let tempHomeDir = "";
|
||||
let originalHome: string | undefined;
|
||||
|
||||
beforeEach(() => {
|
||||
tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-prompt-personality-"));
|
||||
tempHomeDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-prompt-personality-home-"));
|
||||
originalHome = process.env.HOME;
|
||||
process.env.HOME = tempHomeDir;
|
||||
});
|
||||
|
||||
afterEach(cleanupTempHome(() => ({ tempDir, tempHomeDir, originalHome })));
|
||||
|
||||
async function render(personality?: Personality): Promise<string> {
|
||||
const { systemPrompt } = await buildSystemPrompt({
|
||||
cwd: tempDir,
|
||||
contextFiles: [],
|
||||
skills: [],
|
||||
rules: [],
|
||||
toolNames: [],
|
||||
workspaceTree: { ...EMPTY_TREE, rootPath: tempDir },
|
||||
personality,
|
||||
});
|
||||
return systemPrompt.join("\n\n");
|
||||
}
|
||||
|
||||
it("injects the default personality when the option is unset", async () => {
|
||||
const rendered = await render();
|
||||
expect(rendered).toContain("<personality>");
|
||||
expect(rendered).toContain("</personality>");
|
||||
expect(rendered).toContain("terse, evidence-first engineer");
|
||||
});
|
||||
|
||||
it("replaces the default spec when a non-default personality is selected", async () => {
|
||||
const rendered = await render("friendly");
|
||||
expect(rendered).toContain("<personality>");
|
||||
expect(rendered).toContain("warm, supportive collaborator");
|
||||
expect(rendered).not.toContain("terse, evidence-first engineer");
|
||||
});
|
||||
|
||||
it('omits the personality block entirely for "none"', async () => {
|
||||
const rendered = await render("none");
|
||||
expect(rendered).not.toContain("<personality>");
|
||||
expect(rendered).not.toContain("</personality>");
|
||||
});
|
||||
});
|
||||
@@ -259,9 +259,8 @@ describe("runSubprocess soft request budget", () => {
|
||||
// wrap-up reminder; the second abort (after the terminal yield) is the
|
||||
// normal post-yield terminate.
|
||||
expect(abortCallsAtReminder).toBe(1);
|
||||
// The wrap-up reminder is the budget-stop variant with a forced tool choice.
|
||||
// The budget stop forces a synthetic terminal yield.
|
||||
expect(handle.prompts).toHaveLength(2);
|
||||
expect(handle.prompts[1]?.text).toMatch(/request budget/);
|
||||
expect(handle.prompts[1]?.options?.synthetic).toBe(true);
|
||||
expect(handle.prompts[1]?.options?.toolChoice).toEqual({ type: "tool", name: "yield" });
|
||||
// The forced yield finalizes as a normal completion, not an abort.
|
||||
|
||||
@@ -234,7 +234,7 @@ describe("runSubprocess yield reminders", () => {
|
||||
expect(systemPrompt).toHaveLength(4);
|
||||
expect(systemPrompt?.[0]).toBe("system");
|
||||
expect(systemPrompt?.[1]).toBe("project");
|
||||
expect(systemPrompt?.[2]).toMatch(/ROLE\n=+\n\ntest/);
|
||||
expect(systemPrompt?.[2]).toContain(baseAgent.systemPrompt);
|
||||
// The parent-conversation CONTEXT section is gone: subagents get their
|
||||
// background inside the assignment (or a local:// file), never a dump.
|
||||
expect(systemPrompt?.[2]).not.toMatch(/CONTEXT\n=+/);
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { Settings } from "../../src/config/settings";
|
||||
import initAgentPrompt from "../../src/prompts/agents/init.md" with { type: "text" };
|
||||
import * as taskDiscovery from "../../src/task/discovery";
|
||||
import { TaskTool } from "../../src/task/index";
|
||||
import { isScoutSpawnable } from "../../src/task/spawn-policy";
|
||||
@@ -131,10 +130,3 @@ describe("task tool description scout gating", () => {
|
||||
expect(description).toContain("### reviewer");
|
||||
});
|
||||
});
|
||||
|
||||
describe("bundled agent prompt scout gating", () => {
|
||||
it("does not hard-code scout in the init agent prompt", () => {
|
||||
expect(initAgentPrompt.toLowerCase()).not.toContain("scout");
|
||||
expect(initAgentPrompt).toContain("multiple research agents");
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user