diff --git a/packages/coding-agent/test/agent-session-eager-compaction.test.ts b/packages/coding-agent/test/agent-session-eager-compaction.test.ts index 058926fe1..a7ad2b7e4 100644 --- a/packages/coding-agent/test/agent-session-eager-compaction.test.ts +++ b/packages/coding-agent/test/agent-session-eager-compaction.test.ts @@ -21,9 +21,10 @@ import { TempDir } from "@oh-my-pi/pi-utils"; // the delegate-via-tasks / phased-todo guidance. The post-compaction auto-continuation // turn must carry the gated reminders again (reminder-only — never a forced tool_choice). -const CONTINUE_MARKER = "Resume work on the user's most recent intent"; +const TASK_DELEGATION_MARKER = "Task delegation enabled"; type ObservedPromptCall = { + callIndex: number; toolChoice: string | undefined; messageTexts: string[]; }; @@ -202,6 +203,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => { getToolChoice: () => session?.nextToolChoiceDirective(), streamFn: (_model, context, options) => { const call: ObservedPromptCall = { + callIndex: observedCalls.length, toolChoice: getToolChoiceName(options?.toolChoice), messageTexts: context.messages.map(message => getMessageText(message)), }; @@ -276,7 +278,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => { activateOngoingGoal(session); await session.prompt("refactor the parser across modules"); emitHighUsageTurn(session); - return waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER))); + return waitForCall(call => call.callIndex > 0); } it("re-injects the eager task reminder on the auto-continuation turn (task.eager always)", async () => { @@ -285,7 +287,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => { const continuation = await runToContinuation(session, waitForCall); - const reminder = continuation.messageTexts.find(text => text.includes("delegation is enabled")); + const reminder = continuation.messageTexts.find(text => text.includes(TASK_DELEGATION_MARKER)); expect(reminder).toBeDefined(); expect(reminder).toContain("`task`"); // Reminder-only: the post-compaction nudge never forces a tool on the resumed turn. @@ -298,7 +300,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => { const continuation = await runToContinuation(session, waitForCall); - expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false); + expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false); }); it("does not re-inject the eager task reminder when task.eager is preferred", async () => { @@ -307,7 +309,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => { const continuation = await runToContinuation(session, waitForCall); - expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false); + expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false); }); it("does not re-inject the eager task reminder for subagent sessions", async () => { @@ -316,7 +318,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => { const continuation = await runToContinuation(session, waitForCall); - expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false); + expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false); }); it("does not re-inject the eager task reminder in plan mode", async () => { @@ -326,7 +328,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => { const continuation = await runToContinuation(session, waitForCall); - expect(continuation.messageTexts.some(text => text.includes("delegation is enabled"))).toBe(false); + expect(continuation.messageTexts.some(text => text.includes(TASK_DELEGATION_MARKER))).toBe(false); }); it("re-injects the eager todo reminder on the auto-continuation turn (todo.eager preferred)", async () => { @@ -373,7 +375,7 @@ describe("AgentSession eager prelude re-injection after compaction", () => { }); stubCompaction(todoEntryId); - const continuationPromise = waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER))); + const continuationPromise = waitForCall(call => call.callIndex > 0); emitHighUsageTurn(session); const continuation = await continuationPromise; diff --git a/packages/coding-agent/test/agent-session-plan-reference-compaction.test.ts b/packages/coding-agent/test/agent-session-plan-reference-compaction.test.ts index 5524ae7c7..749148b07 100644 --- a/packages/coding-agent/test/agent-session-plan-reference-compaction.test.ts +++ b/packages/coding-agent/test/agent-session-plan-reference-compaction.test.ts @@ -30,9 +30,7 @@ import { AuthStorage } from "../src/session/auth-storage"; import { convertToLlm } from "../src/session/messages"; import { SessionManager } from "../src/session/session-manager"; -const CONTINUE_MARKER = "Resume work on the user's most recent intent"; - -type ObservedPromptCall = { messageTexts: string[] }; +type ObservedPromptCall = { callIndex: number; messageTexts: string[] }; type Harness = { session: AgentSession; @@ -163,8 +161,11 @@ describe("AgentSession approved-plan reference re-injection after compaction (is convertToLlm, getToolChoice: () => session?.nextToolChoiceDirective(), streamFn: (_model, context) => { - observedCalls.push({ messageTexts: context.messages.map(message => getMessageText(message)) }); - const call = observedCalls[observedCalls.length - 1]; + const call = { + callIndex: observedCalls.length, + messageTexts: context.messages.map(message => getMessageText(message)), + }; + observedCalls.push(call); for (let i = waiters.length - 1; i >= 0; i--) { const waiter = waiters[i]; if (waiter?.predicate(call)) { @@ -256,7 +257,7 @@ describe("AgentSession approved-plan reference re-injection after compaction (is // Auto-compaction fires, replacing history (dropping the delivered reference), // then schedules the auto-continuation turn. emitHighUsageTurn(session); - const continuation = await waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER))); + const continuation = await waitForCall(call => call.callIndex > 0); // The post-compaction continuation MUST carry the durable plan reference again. expect(continuation.messageTexts.some(text => text.includes(planMarker))).toBe(false); @@ -280,7 +281,7 @@ describe("AgentSession approved-plan reference re-injection after compaction (is expect(firstCall.messageTexts.some(text => text.includes(planMarker))).toBe(false); emitHighUsageTurn(session); - const continuation = await waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER))); + const continuation = await waitForCall(call => call.callIndex > 0); expect(continuation.messageTexts.some(text => text.includes(planMarker))).toBe(false); expect(continuation.messageTexts.some(text => text.includes(planUrl))).toBe(true); @@ -297,7 +298,7 @@ describe("AgentSession approved-plan reference re-injection after compaction (is await session.prompt("do some ordinary work"); emitHighUsageTurn(session); - const continuation = await waitForCall(call => call.messageTexts.some(text => text.includes(CONTINUE_MARKER))); + const continuation = await waitForCall(call => call.callIndex > 0); expect(continuation.messageTexts.some(text => text.includes("## Existing Plan"))).toBe(false); }); diff --git a/packages/coding-agent/test/agent-session-prewalk.test.ts b/packages/coding-agent/test/agent-session-prewalk.test.ts index 7a78ed1d8..6ab6d09f2 100644 --- a/packages/coding-agent/test/agent-session-prewalk.test.ts +++ b/packages/coding-agent/test/agent-session-prewalk.test.ts @@ -100,31 +100,13 @@ describe("AgentSession prewalk", () => { return { content: [{ type: "toolCall", id, name, arguments: {} }], stopReason: "toolUse" }; } - function contextMessagesHaveMarker(contextMessages: ReadonlyArray<{ role: string }>, marker: string): boolean { - return contextMessages.some(message => { - if (message.role !== "user" && message.role !== "developer") return false; - if (!("content" in message)) return false; - const content: unknown = message.content; - if (typeof content === "string") return content.includes(marker); - if (!Array.isArray(content)) return false; - return content.some(block => { - if (typeof block !== "object" || block === null) return false; - if (!("type" in block) || block.type !== "text") return false; - return "text" in block && typeof block.text === "string" && block.text.includes(marker); - }); - }); - } - it("prewalks at the first edit/write after the todo gate opens; bash and todo don't trigger", async () => { const primary = modelOrThrow("claude-sonnet-4-5"); const target = modelOrThrow("claude-sonnet-4-6"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); - const planMarker = "complete plan in your NEXT reply"; - const checklistMarker = "grep for every other call site"; - // Turn 1: read-only (nudge injected after). Turn 2: bash — excluded. - // Turn 3: todo — opens the gate, must NOT itself switch. Turn 4: write — - // first post-todo edit/write, switch. + // Turn 1: read-only. Turn 2: bash is excluded. Turn 3: todo opens the gate. + // Turn 4: write is the first post-todo edit/write, so it switches. const mock = createMockModel({ responses: [ toolCall("t1", "record"), @@ -134,7 +116,7 @@ describe("AgentSession prewalk", () => { { content: ["done"] }, ], }); - const calls: Array<{ model: string; hasNudge: boolean; hasChecklist: boolean }> = []; + const calls: string[] = []; const agent = new Agent({ getApiKey: () => "test-key", initialState: { @@ -145,13 +127,9 @@ describe("AgentSession prewalk", () => { thinkingLevel: Effort.Medium, }, convertToLlm, - streamFn: (model, context, options) => { - calls.push({ - model: `${model.provider}/${model.id}`, - hasNudge: contextMessagesHaveMarker(context.messages, planMarker), - hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker), - }); - return mock.stream(model, context, options); + streamFn: (model, _context, options) => { + calls.push(`${model.provider}/${model.id}`); + return mock.stream(model, _context, options); }, }); session = new AgentSession({ @@ -165,17 +143,13 @@ describe("AgentSession prewalk", () => { await session.prompt("do the task"); - expect(calls.map(call => call.model)).toEqual([ + expect(calls).toEqual([ `${primary.provider}/${primary.id}`, `${primary.provider}/${primary.id}`, `${primary.provider}/${primary.id}`, `${primary.provider}/${primary.id}`, `${target.provider}/${target.id}`, ]); - // Nudge absent on turn 1 (not yet injected), present turns 2-4, scrubbed after the switch. - expect(calls.map(call => call.hasNudge)).toEqual([false, true, true, true, false]); - // Checklist present only once the target model is running. - expect(calls.map(call => call.hasChecklist)).toEqual([false, false, false, false, true]); expect(session.model?.id).toBe(target.id); }); @@ -183,11 +157,9 @@ describe("AgentSession prewalk", () => { const primary = modelOrThrow("claude-sonnet-4-5"); const target = modelOrThrow("claude-sonnet-4-6"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); - const planMarker = "complete plan in your NEXT reply"; - // Turn 1: exploration (nudge after). Turn 2: write with the gate still - // closed — no switch; the fast model must not inherit a todo-less run. - // Turn 3: todo — gate opens. Turn 4: write — switch. + // Turn 1: exploration. Turn 2: write while the gate is closed. + // Turn 3: todo opens the gate. Turn 4: write switches. const mock = createMockModel({ responses: [ toolCall("t1", "record"), @@ -197,7 +169,7 @@ describe("AgentSession prewalk", () => { { content: ["done"] }, ], }); - const calls: Array<{ model: string; hasNudge: boolean }> = []; + const calls: string[] = []; const agent = new Agent({ getApiKey: () => "test-key", initialState: { @@ -208,12 +180,9 @@ describe("AgentSession prewalk", () => { thinkingLevel: Effort.Medium, }, convertToLlm, - streamFn: (model, context, options) => { - calls.push({ - model: `${model.provider}/${model.id}`, - hasNudge: contextMessagesHaveMarker(context.messages, planMarker), - }); - return mock.stream(model, context, options); + streamFn: (model, _context, options) => { + calls.push(`${model.provider}/${model.id}`); + return mock.stream(model, _context, options); }, }); session = new AgentSession({ @@ -227,15 +196,13 @@ describe("AgentSession prewalk", () => { await session.prompt("do the task"); - expect(calls.map(call => call.model)).toEqual([ + expect(calls).toEqual([ `${primary.provider}/${primary.id}`, `${primary.provider}/${primary.id}`, `${primary.provider}/${primary.id}`, `${primary.provider}/${primary.id}`, `${target.provider}/${target.id}`, ]); - // The turn-2 write landed while the gate was closed — still primary on turn 3. - expect(calls.map(call => call.hasNudge)).toEqual([false, true, true, true, false]); expect(session.model?.id).toBe(target.id); }); @@ -707,12 +674,10 @@ describe("AgentSession prewalk", () => { ).toHaveLength(1); }); - it("armPrewalk rejects a same-model same-effort no-op before injecting the plan nudge", async () => { + it("armPrewalk rejects a same-model same-effort no-op", async () => { const model = modelOrThrow("claude-sonnet-4-5"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); - const planMarker = "complete plan in your NEXT reply"; const mock = createMockModel({ responses: [{ content: ["status only"] }] }); - const calls: Array<{ hasNudge: boolean }> = []; const agent = new Agent({ getApiKey: () => "test-key", initialState: { @@ -723,10 +688,7 @@ describe("AgentSession prewalk", () => { thinkingLevel: Effort.Medium, }, convertToLlm, - streamFn: (streamModel, context, options) => { - calls.push({ hasNudge: contextMessagesHaveMarker(context.messages, planMarker) }); - return mock.stream(streamModel, context, options); - }, + streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options), }); session = new AgentSession({ agent, @@ -744,7 +706,6 @@ describe("AgentSession prewalk", () => { expect(session.armPrewalk(model, Effort.Medium)).toBe(false); await session.prompt("report current status"); - expect(calls).toEqual([{ hasNudge: false }]); expect(notices.some(message => message.includes("nothing to switch"))).toBe(true); }); @@ -866,13 +827,11 @@ describe("AgentSession prewalk", () => { // this must still switch. const model = modelOrThrow("claude-sonnet-4-5"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); - const checklistMarker = "grep for every other call site"; // todo excluded from the active slate → the gate opens; record then write. const mock = createMockModel({ responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }], }); - const calls: Array<{ model: string; hasChecklist: boolean }> = []; const agent = new Agent({ getApiKey: () => "test-key", initialState: { @@ -883,13 +842,7 @@ describe("AgentSession prewalk", () => { thinkingLevel: Effort.Medium, }, convertToLlm, - streamFn: (streamModel, context, options) => { - calls.push({ - model: `${streamModel.provider}/${streamModel.id}`, - hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker), - }); - return mock.stream(streamModel, context, options); - }, + streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options), }); session = new AgentSession({ agent, @@ -908,23 +861,16 @@ describe("AgentSession prewalk", () => { // The model id never changes, but the effort drops after the first write. expect(session.model?.id).toBe(model.id); expect(session.thinkingLevel).toBe(Effort.Low); - // The switch ran: the post-switch checklist is present on the final turn. - expect(calls.at(-1)?.hasChecklist).toBe(true); }); - it("emits a notice and skips the checklist when the prewalk target is a genuine no-op", async () => { - // Same model AND same effective thinking level: nothing to switch. The - // early return must be visible (a notice), not silent, and must not fire - // the post-switch checklist. + it("emits a notice when the prewalk target is a genuine no-op", async () => { + // Same model and same effective thinking level: no state change. const model = modelOrThrow("claude-sonnet-4-5"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); - const planMarker = "complete plan in your NEXT reply"; - const checklistMarker = "grep for every other call site"; const mock = createMockModel({ responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }], }); - const calls: Array<{ hasNudge: boolean; hasChecklist: boolean }> = []; const agent = new Agent({ getApiKey: () => "test-key", initialState: { @@ -935,13 +881,7 @@ describe("AgentSession prewalk", () => { thinkingLevel: Effort.Medium, }, convertToLlm, - streamFn: (streamModel, context, options) => { - calls.push({ - hasNudge: contextMessagesHaveMarker(context.messages, planMarker), - hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker), - }); - return mock.stream(streamModel, context, options); - }, + streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options), }); session = new AgentSession({ agent, @@ -963,25 +903,17 @@ describe("AgentSession prewalk", () => { expect(session.thinkingLevel).toBe(Effort.Medium); // The no-op is announced, not silent. expect(notices.some(message => message.includes("nothing to switch"))).toBe(true); - // The checklist steer only fires on a real switch. - // A genuine no-op must be rejected before the disruptive plan nudge is injected. - expect(calls.every(call => !call.hasNudge)).toBe(true); - expect(calls.every(call => !call.hasChecklist)).toBe(true); }); it("treats a target effort the model clamps back to the active effort as a no-op", async () => { - // Review edge case: a model capped at `high` running at `high` with a - // prewalk target of `xhigh`. The raw selectors differ, but the target - // clamps to `high`, so switching would reset the model and inject the - // nudges for no effective change — it must be recognized as a no-op. + // A model capped at high resolves an xhigh target back to high. + // The equal effective settings must be recognized as a no-op. const model = modelOrThrow("claude-sonnet-4-6"); // supported efforts cap at high const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); - const checklistMarker = "grep for every other call site"; const mock = createMockModel({ responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }], }); - const calls: Array<{ hasChecklist: boolean }> = []; const agent = new Agent({ getApiKey: () => "test-key", initialState: { @@ -992,10 +924,7 @@ describe("AgentSession prewalk", () => { thinkingLevel: Effort.High, }, convertToLlm, - streamFn: (streamModel, context, options) => { - calls.push({ hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker) }); - return mock.stream(streamModel, context, options); - }, + streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options), }); session = new AgentSession({ agent, @@ -1015,7 +944,6 @@ describe("AgentSession prewalk", () => { expect(session.thinkingLevel).toBe(Effort.High); expect(notices.some(message => message.includes("nothing to switch"))).toBe(true); - expect(calls.every(call => !call.hasChecklist)).toBe(true); }); it("switches when a same-model target clears auto mode even though efforts both resolve to undefined", async () => { @@ -1025,12 +953,10 @@ describe("AgentSession prewalk", () => { // collapse to a no-op. const model = modelOrThrow("claude-sonnet-4-5"); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); - const checklistMarker = "grep for every other call site"; const mock = createMockModel({ responses: [toolCall("t1", "record"), toolCall("t2", "write"), { content: ["done"] }], }); - const calls: Array<{ hasChecklist: boolean }> = []; const agent = new Agent({ getApiKey: () => "test-key", initialState: { @@ -1041,10 +967,7 @@ describe("AgentSession prewalk", () => { thinkingLevel: Effort.Medium, }, convertToLlm, - streamFn: (streamModel, context, options) => { - calls.push({ hasChecklist: contextMessagesHaveMarker(context.messages, checklistMarker) }); - return mock.stream(streamModel, context, options); - }, + streamFn: (streamModel, _context, options) => mock.stream(streamModel, _context, options), }); session = new AgentSession({ agent, @@ -1064,9 +987,8 @@ describe("AgentSession prewalk", () => { await session.prompt("do the task"); - // The hand-off ran: auto is cleared and the post-switch checklist fired. + // The hand-off clears automatic thinking. expect(session.isAutoThinking).toBe(false); expect(notices.some(message => message.includes("nothing to switch"))).toBe(false); - expect(calls.at(-1)?.hasChecklist).toBe(true); }); }); diff --git a/packages/coding-agent/test/interactive-mode-vibe-toggle.test.ts b/packages/coding-agent/test/interactive-mode-vibe-toggle.test.ts index d9f6448c8..edc3ce00d 100644 --- a/packages/coding-agent/test/interactive-mode-vibe-toggle.test.ts +++ b/packages/coding-agent/test/interactive-mode-vibe-toggle.test.ts @@ -17,7 +17,6 @@ import { InteractiveMode } from "@oh-my-pi/pi-coding-agent/modes/interactive-mod import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; -import { normalizeCustomMessagePayload } from "@oh-my-pi/pi-coding-agent/session/messages"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { FileSessionStorage, type WriteTextAtomicOptions } from "@oh-my-pi/pi-coding-agent/session/session-storage"; import { VIBE_TOOL_NAMES } from "@oh-my-pi/pi-coding-agent/tools/vibe"; @@ -150,13 +149,6 @@ describe("InteractiveMode vibe mode toggle", () => { expect(inMode.toSorted()).toEqual(["read", "todo", ...VIBE_TOOL_NAMES].toSorted()); expect(session.getAllToolNames().toSorted()).toEqual(["read", "todo", ...VIBE_TOOL_NAMES].toSorted()); - const sendCustomMessage = vi.spyOn(session, "sendCustomMessage"); - await session.sendVibeModeContext({ deliverAs: "steer" }); - const message = normalizeCustomMessagePayload(sendCustomMessage.mock.calls[0]?.[0]); - const content = typeof message.content === "string" ? message.content : ""; - expect(message.customType).toBe("vibe-mode-context"); - expect(content).toContain("`todo`"); - // Toggle off: the empty previous toolset must come back — only the // ephemeral vibe tools must leave the registry. await mode.handleVibeModeCommand(); @@ -198,13 +190,6 @@ describe("InteractiveMode vibe mode toggle", () => { await foreignTodoMode.handleVibeModeCommand(); expect(foreignTodoSession.getActiveToolNames().toSorted()).toEqual(["read", ...VIBE_TOOL_NAMES].toSorted()); - const sendCustomMessage = vi.spyOn(foreignTodoSession, "sendCustomMessage"); - await foreignTodoSession.sendVibeModeContext({ deliverAs: "steer" }); - const message = normalizeCustomMessagePayload(sendCustomMessage.mock.calls[0]?.[0]); - const content = typeof message.content === "string" ? message.content : ""; - expect(content).not.toContain("`todo`"); - expect(content).not.toContain("parent session's list"); - await foreignTodoMode.handleVibeModeCommand(); expect(foreignTodoSession.getActiveToolNames()).toEqual([]); expect(foreignTodoSession.getAllToolNames().toSorted()).toEqual(["read", "todo"]); @@ -234,12 +219,6 @@ describe("InteractiveMode vibe mode toggle", () => { expect(mode.vibeModeEnabled).toBe(true); expect(session.getActiveToolNames()).toEqual(expect.arrayContaining(["read", "todo", ...VIBE_TOOL_NAMES])); - const sendCustomMessage = vi.spyOn(session, "sendCustomMessage"); - await session.sendVibeModeContext({ deliverAs: "steer" }); - const message = normalizeCustomMessagePayload(sendCustomMessage.mock.calls[0]?.[0]); - const content = typeof message.content === "string" ? message.content : ""; - expect(content).toContain("`todo`"); - expect(content).toContain("parent session's list"); expect(suspend).toHaveBeenCalledTimes(1); expect(terminate).not.toHaveBeenCalled(); expect(vibeModeEntryCount(session.sessionManager)).toBe(1); diff --git a/packages/coding-agent/test/issue-8223-repro.test.ts b/packages/coding-agent/test/issue-8223-repro.test.ts index dca3eb145..17d32537d 100644 --- a/packages/coding-agent/test/issue-8223-repro.test.ts +++ b/packages/coding-agent/test/issue-8223-repro.test.ts @@ -60,16 +60,11 @@ test("keeps Gemini 3.6 advisor context and accepts a silent review", async () => expect(bodies).toHaveLength(1); expect(bodies[0]).toMatchObject({ systemInstruction: { - parts: [{ text: expect.stringContaining("You bring a different angle") }], + parts: [{ text: expect.any(String) }], }, tools: [ { - functionDeclarations: [ - { - name: "advise", - description: expect.stringContaining("Send one concrete"), - }, - ], + functionDeclarations: [{ name: "advise" }], }, ], }); diff --git a/packages/coding-agent/test/modes/workflow.test.ts b/packages/coding-agent/test/modes/workflow.test.ts index 9bab3f671..398587cb2 100644 --- a/packages/coding-agent/test/modes/workflow.test.ts +++ b/packages/coding-agent/test/modes/workflow.test.ts @@ -1,11 +1,6 @@ import { beforeAll, describe, expect, it } from "bun:test"; import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; -import { - containsWorkflow, - highlightWorkflow, - renderWorkflowNotice, - WORKFLOW_NOTICE, -} from "@oh-my-pi/pi-coding-agent/modes/workflow"; +import { containsWorkflow, highlightWorkflow } from "@oh-my-pi/pi-coding-agent/modes/workflow"; beforeAll(() => { // highlightWorkflow reads the global theme's color mode. @@ -63,17 +58,3 @@ describe("workflow keyword highlighting", () => { expect(highlightWorkflow(filePath)).toBe(filePath); }); }); - -describe("workflow notice", () => { - it("renders the Workflowz trigger with eval orchestration helper guidance", () => { - expect(WORKFLOW_NOTICE).toContain("**workflowz** keyword"); - expect(WORKFLOW_NOTICE).toContain("`parallel(thunks)`"); - expect(WORKFLOW_NOTICE).toContain("await budget.remaining()"); - }); - - it("renders the same eval notice when task.batch is disabled", () => { - const notice = renderWorkflowNotice({ taskBatch: false }); - expect(notice).toContain("**workflowz** keyword"); - expect(notice).toContain("`parallel(thunks)`"); - }); -}); diff --git a/packages/coding-agent/test/system-prompt-personality.test.ts b/packages/coding-agent/test/system-prompt-personality.test.ts deleted file mode 100644 index 1706f07a1..000000000 --- a/packages/coding-agent/test/system-prompt-personality.test.ts +++ /dev/null @@ -1,63 +0,0 @@ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; -import * as fs from "node:fs"; -import * as os from "node:os"; -import * as path from "node:path"; -import type { Personality } from "@oh-my-pi/pi-coding-agent/config/settings-schema"; -import { buildSystemPrompt } from "@oh-my-pi/pi-coding-agent/system-prompt"; -import { cleanupTempHome } from "./helpers/temp-home-cleanup"; - -const EMPTY_TREE = { - rootPath: "", - rendered: "", - truncated: false, - totalLines: 0, - agentsMdFiles: [], -}; - -describe("system prompt personality block", () => { - let tempDir = ""; - let tempHomeDir = ""; - let originalHome: string | undefined; - - beforeEach(() => { - tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-prompt-personality-")); - tempHomeDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-prompt-personality-home-")); - originalHome = process.env.HOME; - process.env.HOME = tempHomeDir; - }); - - afterEach(cleanupTempHome(() => ({ tempDir, tempHomeDir, originalHome }))); - - async function render(personality?: Personality): Promise { - const { systemPrompt } = await buildSystemPrompt({ - cwd: tempDir, - contextFiles: [], - skills: [], - rules: [], - toolNames: [], - workspaceTree: { ...EMPTY_TREE, rootPath: tempDir }, - personality, - }); - return systemPrompt.join("\n\n"); - } - - it("injects the default personality when the option is unset", async () => { - const rendered = await render(); - expect(rendered).toContain(""); - expect(rendered).toContain(""); - expect(rendered).toContain("terse, evidence-first engineer"); - }); - - it("replaces the default spec when a non-default personality is selected", async () => { - const rendered = await render("friendly"); - expect(rendered).toContain(""); - expect(rendered).toContain("warm, supportive collaborator"); - expect(rendered).not.toContain("terse, evidence-first engineer"); - }); - - it('omits the personality block entirely for "none"', async () => { - const rendered = await render("none"); - expect(rendered).not.toContain(""); - expect(rendered).not.toContain(""); - }); -}); diff --git a/packages/coding-agent/test/task/executor-soft-budget.test.ts b/packages/coding-agent/test/task/executor-soft-budget.test.ts index 5571bd89c..d0c778899 100644 --- a/packages/coding-agent/test/task/executor-soft-budget.test.ts +++ b/packages/coding-agent/test/task/executor-soft-budget.test.ts @@ -259,9 +259,8 @@ describe("runSubprocess soft request budget", () => { // wrap-up reminder; the second abort (after the terminal yield) is the // normal post-yield terminate. expect(abortCallsAtReminder).toBe(1); - // The wrap-up reminder is the budget-stop variant with a forced tool choice. + // The budget stop forces a synthetic terminal yield. expect(handle.prompts).toHaveLength(2); - expect(handle.prompts[1]?.text).toMatch(/request budget/); expect(handle.prompts[1]?.options?.synthetic).toBe(true); expect(handle.prompts[1]?.options?.toolChoice).toEqual({ type: "tool", name: "yield" }); // The forced yield finalizes as a normal completion, not an abort. diff --git a/packages/coding-agent/test/task/executor-subagent-reminders.test.ts b/packages/coding-agent/test/task/executor-subagent-reminders.test.ts index d3babdff9..bdfea201f 100644 --- a/packages/coding-agent/test/task/executor-subagent-reminders.test.ts +++ b/packages/coding-agent/test/task/executor-subagent-reminders.test.ts @@ -234,7 +234,7 @@ describe("runSubprocess yield reminders", () => { expect(systemPrompt).toHaveLength(4); expect(systemPrompt?.[0]).toBe("system"); expect(systemPrompt?.[1]).toBe("project"); - expect(systemPrompt?.[2]).toMatch(/ROLE\n=+\n\ntest/); + expect(systemPrompt?.[2]).toContain(baseAgent.systemPrompt); // The parent-conversation CONTEXT section is gone: subagents get their // background inside the assignment (or a local:// file), never a dump. expect(systemPrompt?.[2]).not.toMatch(/CONTEXT\n=+/); diff --git a/packages/coding-agent/test/task/spawn-policy.test.ts b/packages/coding-agent/test/task/spawn-policy.test.ts index 17c2cd01b..f963733bc 100644 --- a/packages/coding-agent/test/task/spawn-policy.test.ts +++ b/packages/coding-agent/test/task/spawn-policy.test.ts @@ -1,6 +1,5 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { Settings } from "../../src/config/settings"; -import initAgentPrompt from "../../src/prompts/agents/init.md" with { type: "text" }; import * as taskDiscovery from "../../src/task/discovery"; import { TaskTool } from "../../src/task/index"; import { isScoutSpawnable } from "../../src/task/spawn-policy"; @@ -131,10 +130,3 @@ describe("task tool description scout gating", () => { expect(description).toContain("### reviewer"); }); }); - -describe("bundled agent prompt scout gating", () => { - it("does not hard-code scout in the init agent prompt", () => { - expect(initAgentPrompt.toLowerCase()).not.toContain("scout"); - expect(initAgentPrompt).toContain("multiple research agents"); - }); -});