Merge remote-tracking branch 'can1357/main' into fix/empty-stop-guard-tooluse

This commit is contained in:
DarkPhilosophy
2026-06-07 15:04:44 +03:00
545 changed files with 26429 additions and 8342 deletions
+9 -8
View File
@@ -615,7 +615,7 @@ describe("ACP agent", () => {
await Bun.sleep(0);
});
it("plan-approval standing handler renames the plan and exits plan mode on apply", async () => {
it("plan-approval standing handler approves the agent-named plan and exits plan mode on apply", async () => {
const harness = await createHarness();
Settings.instance.set("plan.enabled", true);
@@ -625,7 +625,9 @@ describe("ACP agent", () => {
const artifactsDir = session.sessionManager.getArtifactsDir();
expect(artifactsDir).not.toBeNull();
const planPath = path.join(artifactsDir!, "local", "PLAN.md");
// The agent writes to its chosen `local://<slug>-plan.md` and resolves with
// the matching slug — the file is never renamed.
const planPath = path.join(artifactsDir!, "local", "words-counter-plan.md");
await Bun.write(planPath, "# Words Counter\n\nFile contents.");
const updatesBefore = harness.updates.length;
@@ -636,21 +638,20 @@ describe("ACP agent", () => {
extra: { title: "words-counter" },
})) as {
content: Array<{ type: string; text: string }>;
details: { sourceToolName: string; sourceResultDetails: { finalPlanFilePath: string; title: string } };
details: { sourceToolName: string; sourceResultDetails: { planFilePath: string; title: string } };
};
// Plan-approval payload is shaped for `event-controller` / ACP renderers.
expect(result.details.sourceToolName).toBe("plan_approval");
expect(result.details.sourceResultDetails.title).toBe("words-counter");
expect(result.details.sourceResultDetails.finalPlanFilePath).toBe("local://words-counter.md");
expect(result.details.sourceResultDetails.planFilePath).toBe("local://words-counter-plan.md");
expect(result.content[0]?.text).toMatch(/Plan approved/);
// Plan file is renamed and the source path no longer exists.
expect(await Bun.file(path.join(artifactsDir!, "local", "words-counter.md")).exists()).toBe(true);
expect(await Bun.file(planPath).exists()).toBe(false);
// Plan file keeps its agent-chosen name — no rename.
expect(await Bun.file(planPath).exists()).toBe(true);
// Mode + handler are cleared; the agent regains write tools next turn.
expect(session.planModeState).toBeUndefined();
expect(session.standingResolveHandler).toBeUndefined();
expect(session.planReferencePath).toBe("local://words-counter.md");
expect(session.planReferencePath).toBe("local://words-counter-plan.md");
const approvalUpdates = harness.updates.slice(updatesBefore);
// Mode-change notifications reached the client so Zed's UI and config
// selector both reflect the approval-driven exit.
@@ -492,20 +492,6 @@ describe("wave 3 commands", () => {
expect(output[0]).toContain("Usage: /memory");
});
it("/memory view: outputs memory payload (or empty message)", async () => {
const { output, runtime } = createRuntime();
const result = await executeAcpBuiltinSlashCommand("/memory view", runtime);
expect(result).toEqual({ consumed: true });
expect(output.length).toBeGreaterThan(0);
});
it("/memory (no args): defaults to view", async () => {
const { output, runtime } = createRuntime();
const result = await executeAcpBuiltinSlashCommand("/memory", runtime);
expect(result).toEqual({ consumed: true });
expect(output.length).toBeGreaterThan(0);
});
// /todo start fuzzy match
it("/todo start: finds pending task by substring and starts it", async () => {
const { output, session, runtime } = createRuntime();
@@ -667,37 +653,6 @@ describe("wave 4 commands", () => {
});
// /plugins
it("/plugins list: outputs without throwing when registries are empty", async () => {
const { MarketplaceManager } = await import("../src/extensibility/plugins/marketplace");
const { PluginManager } = await import("../src/extensibility/plugins");
const listInstalledSpy = spyOn(MarketplaceManager.prototype, "listInstalledPlugins").mockResolvedValue([]);
const npmListSpy = spyOn(PluginManager.prototype, "list").mockResolvedValue([]);
try {
const { output, runtime } = createRuntime();
const result = await executeAcpBuiltinSlashCommand("/plugins list", runtime);
expect(result).toEqual({ consumed: true });
expect(output.length).toBeGreaterThan(0);
} finally {
listInstalledSpy.mockRestore();
npmListSpy.mockRestore();
}
});
it("/plugins (no args): defaults to list", async () => {
const { MarketplaceManager } = await import("../src/extensibility/plugins/marketplace");
const { PluginManager } = await import("../src/extensibility/plugins");
const listInstalledSpy = spyOn(MarketplaceManager.prototype, "listInstalledPlugins").mockResolvedValue([]);
const npmListSpy = spyOn(PluginManager.prototype, "list").mockResolvedValue([]);
try {
const { output, runtime } = createRuntime();
const result = await executeAcpBuiltinSlashCommand("/plugins", runtime);
expect(result).toEqual({ consumed: true });
expect(output.length).toBeGreaterThan(0);
} finally {
listInstalledSpy.mockRestore();
npmListSpy.mockRestore();
}
});
// /todo start with in_progress status in fuzzy list
it("/todo start: resolves ambiguous matches by preferring active tasks", async () => {
@@ -164,7 +164,6 @@ describe("ACP stdout hygiene", () => {
proc.stdin.flush();
const firstLine = await readFirstFrame(proc.stdout);
expect(firstLine.length).toBeGreaterThan(0);
expect(firstLine[0]).toBe("{");
const message = JSON.parse(firstLine) as {
@@ -115,31 +115,6 @@ describe.skipIf(!e2eApiKey("ANTHROPIC_API_KEY"))("AgentSession compaction e2e",
expect(firstMsg.role).toBe("compactionSummary");
}, 120000);
it("should maintain valid session state after compaction", async () => {
await createSession();
// Build up history
await session.prompt("What is the capital of France? One word answer.");
await session.agent.waitForIdle();
await session.prompt("What is the capital of Germany? One word answer.");
await session.agent.waitForIdle();
// Compact
await session.compact();
// Session should still be usable
await session.prompt("What is the capital of Italy? One word answer.");
await session.agent.waitForIdle();
// Should have messages after compaction
expect(session.messages.length).toBeGreaterThan(0);
// The agent should have responded
const assistantMessages = session.messages.filter(m => m.role === "assistant");
expect(assistantMessages.length).toBeGreaterThan(0);
}, 180000);
it("should persist compaction to session file", async () => {
await createSession();
@@ -206,9 +181,5 @@ describe.skipIf(!e2eApiKey("ANTHROPIC_API_KEY"))("AgentSession compaction e2e",
);
// Manual compaction doesn't emit auto_compaction events
expect(autoCompactionEvents.length).toBe(0);
// Regular events should have been emitted
const messageEndEvents = events.filter(e => e.type === "message_end");
expect(messageEndEvents.length).toBeGreaterThan(0);
}, 120000);
});
@@ -6,6 +6,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { scheduler } from "node:timers/promises";
import { Agent, AgentBusyError, type AgentTool } from "@oh-my-pi/pi-agent-core";
import { type AssistantMessage, getBundledModel, type Message, type ToolCall } from "@oh-my-pi/pi-ai";
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
@@ -25,6 +26,19 @@ import { createAssistantMessage } from "./helpers/agent-session-setup";
// Mock stream that mimics AssistantMessageEventStream
// AgentSession schedules its TTSR retry and context-promotion continuations
// through `scheduler.wait(delayMs, { signal })` (node:timers/promises), with
// blind 50ms/100ms "settle" delays. Tests that drive a continuation to
// completion would otherwise pay that wall-clock time on every run. This spy
// collapses the blind delay to a single macrotask hop (`scheduler.wait(0)`)
// while preserving the real abort-signal semantics, so the continuation still
// fires only after the aborted/overflowed turn has been recorded. Each test
// that opts in must run inside a block whose afterEach restores mocks.
const originalSchedulerWait = scheduler.wait.bind(scheduler);
function collapseSchedulerSettleDelays(): void {
vi.spyOn(scheduler, "wait").mockImplementation((_delayMs, options) => originalSchedulerWait(0, options));
}
describe("AgentSession concurrent prompt guard", () => {
let session: AgentSession;
let tempDir: string;
@@ -101,7 +115,7 @@ describe("AgentSession concurrent prompt guard", () => {
const deadline = Date.now() + timeoutMs;
while (Date.now() < deadline) {
if (predicate()) return;
await Bun.sleep(10);
await Bun.sleep(1);
}
throw new Error("Timed out waiting for condition");
@@ -131,7 +145,7 @@ describe("AgentSession concurrent prompt guard", () => {
await waitFor(() => session.isStreaming);
// steer should work while streaming
expect(() => session.steer("Steering message")).not.toThrow();
await session.steer("Steer while streaming");
expect(session.queuedMessageCount).toBe(1);
// Cleanup
@@ -139,6 +153,79 @@ describe("AgentSession concurrent prompt guard", () => {
await firstPrompt.catch(() => {});
});
it("interrupts active work and immediately sends queued steering messages", async () => {
const model = getBundledModel("anthropic", "claude-sonnet-4-5")!;
const callMessages: Message[][] = [];
const agent = new Agent({
getApiKey: () => "test-key",
initialState: {
model,
systemPrompt: ["Test"],
tools: [],
},
convertToLlm,
streamFn: (_model, context, options) => {
const callIndex = callMessages.length;
callMessages.push([...context.messages]);
const stream = new AssistantMessageEventStream();
queueMicrotask(() => {
stream.push({ type: "start", partial: createAssistantMessage("") });
if (callIndex > 0) {
stream.push({ type: "done", reason: "stop", message: createAssistantMessage("Handled steer") });
}
});
options?.signal?.addEventListener(
"abort",
() => {
stream.push({
type: "error",
reason: "aborted",
error: createAssistantMessage("Interrupted"),
});
},
{ once: true },
);
return stream;
},
});
const sessionManager = SessionManager.inMemory();
const settings = Settings.isolated();
const authStorage = await AuthStorage.create(path.join(tempDir, "testauth-interrupt-flush.db"));
authStorages.push(authStorage);
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models-interrupt-flush.yml"));
authStorage.setRuntimeApiKey("anthropic", "test-key");
session = new AgentSession({
agent,
sessionManager,
settings,
modelRegistry,
});
const firstPrompt = session.prompt("First message").catch(() => {});
await waitFor(() => session.isStreaming && callMessages.length === 1);
await session.steer("Send this now");
expect(session.getQueuedMessages().steering).toEqual(["Send this now"]);
await session.interruptAndFlushQueuedMessages({ reason: "Interrupted by user" });
await firstPrompt;
expect(callMessages).toHaveLength(2);
expect(
callMessages[1]?.some(message => {
if (typeof message.content === "string") {
return message.content.includes("Send this now");
}
return message.content.some(content => content.type === "text" && content.text.includes("Send this now"));
}),
).toBe(true);
expect(session.getQueuedMessages()).toEqual({ steering: [], followUp: [] });
});
it("should allow followUp() while streaming", async () => {
await createSession();
@@ -147,7 +234,7 @@ describe("AgentSession concurrent prompt guard", () => {
await waitFor(() => session.isStreaming);
// followUp should work while streaming
expect(() => session.followUp("Follow-up message")).not.toThrow();
await session.followUp("Follow-up while streaming");
expect(session.queuedMessageCount).toBe(1);
// Cleanup
@@ -595,13 +682,14 @@ describe("AgentSession TTSR resume gate", () => {
if (tempDir && fs.existsSync(tempDir)) {
fs.rmSync(tempDir, { recursive: true });
}
vi.restoreAllMocks();
});
async function waitFor(predicate: () => boolean, timeoutMs = 500): Promise<void> {
const deadline = Date.now() + timeoutMs;
while (Date.now() < deadline) {
if (predicate()) return;
await Bun.sleep(10);
await Bun.sleep(1);
}
throw new Error("Timed out waiting for condition");
@@ -674,6 +762,7 @@ describe("AgentSession TTSR resume gate", () => {
}
it("prompt() blocks until TTSR interrupt continuation completes", async () => {
collapseSchedulerSettleDelays();
const model = getBundledModel("anthropic", "claude-sonnet-4-5")!;
let streamCallCount = 0;
let continuationCompleted = false;
@@ -733,7 +822,114 @@ describe("AgentSession TTSR resume gate", () => {
expect(session.isStreaming).toBe(false);
});
it("labels aborted tool placeholders with the TTSR rule reason", async () => {
collapseSchedulerSettleDelays();
const model = getBundledModel("anthropic", "claude-sonnet-4-5")!;
let streamCallCount = 0;
const ttsrManager = new TtsrManager({
enabled: true,
contextMode: "discard",
interruptMode: "always",
repeatMode: "once",
repeatGap: 10,
});
ttsrManager.addRule(testRule);
const toolCallContent: ToolCall = {
type: "toolCall",
id: "call_ttsr_abort_reason",
name: "mock_edit",
arguments: { snippet: "let val = result.unwrap(" },
};
const makeToolCallMsg = (stopReason: "toolUse" | "aborted" = "toolUse"): AssistantMessage => ({
role: "assistant",
content: [toolCallContent],
api: "anthropic-messages",
provider: "anthropic",
model: "mock",
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason,
timestamp: Date.now(),
});
const agent = new Agent({
getApiKey: () => "test-key",
initialState: { model, systemPrompt: ["Test"], tools: [] },
streamFn: (_model, _context, options) => {
streamCallCount++;
const stream = new AssistantMessageEventStream();
const signal = options?.signal;
if (streamCallCount === 1) {
queueMicrotask(() => {
const partial = makeToolCallMsg();
if (signal) {
signal.addEventListener(
"abort",
() => {
stream.push({
type: "error",
reason: "aborted",
error: makeToolCallMsg("aborted"),
});
},
{ once: true },
);
}
stream.push({ type: "start", partial });
stream.push({ type: "toolcall_start", contentIndex: 0, partial });
stream.push({
type: "toolcall_delta",
contentIndex: 0,
delta: 'let val = result.unwrap("oops")',
partial,
});
});
} else {
pushContinuationStream(stream, () => {});
}
return stream;
},
});
const sessionManager = SessionManager.inMemory();
const settings = Settings.isolated();
const authStorage = await AuthStorage.create(path.join(tempDir, "testauth-abort-reason.db"));
authStorages.push(authStorage);
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml"));
authStorage.setRuntimeApiKey("anthropic", "test-key");
session = new AgentSession({ agent, sessionManager, settings, modelRegistry, ttsrManager });
await session.prompt("Write some Rust code");
const toolResult = sessionManager
.getEntries()
.find(
entry =>
entry.type === "message" &&
entry.message.role === "toolResult" &&
entry.message.toolCallId === toolCallContent.id,
);
expect(toolResult?.type).toBe("message");
const text =
toolResult?.type === "message" && toolResult.message.role === "toolResult"
? (toolResult.message.content.find((part): part is { type: "text"; text: string } => part.type === "text")
?.text ?? "")
: "";
expect(text).toContain("Tool execution was aborted: TTSR matched rule: no-unwrap");
expect(text).not.toContain("Request was aborted");
});
it("relativizes the rule file path in the TTSR interrupt injection (no absolute leak)", async () => {
collapseSchedulerSettleDelays();
const model = getBundledModel("anthropic", "claude-sonnet-4-5")!;
let streamCallCount = 0;
@@ -946,6 +1142,7 @@ describe("AgentSession TTSR resume gate", () => {
});
it("prompt() waits for TTSR continuation with tool calls to finish", async () => {
collapseSchedulerSettleDelays();
const model = getBundledModel("anthropic", "claude-sonnet-4-5")!;
let streamCallCount = 0;
let toolExecutionFinished = false;
@@ -1306,6 +1503,7 @@ describe("AgentSession TTSR resume gate", () => {
});
it("prompt() waits for context-promotion continuation to finish", async () => {
collapseSchedulerSettleDelays();
const authStorage = await AuthStorage.create(path.join(tempDir, "testauth-promo.db"));
authStorages.push(authStorage);
authStorage.setRuntimeApiKey("openai-codex", "test-key");
@@ -1,4 +1,4 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { Agent } from "@oh-my-pi/pi-agent-core";
import type { AssistantMessage, Model, ProviderSessionState } from "@oh-my-pi/pi-ai";
@@ -15,19 +15,26 @@ describe("AgentSession context promotion", () => {
let modelRegistry: ModelRegistry;
let authStorage: AuthStorage;
beforeEach(async () => {
beforeAll(async () => {
// ModelRegistry eagerly loads the immutable bundled model catalog in its
// constructor (~100ms). The catalog and auth fixture never change between
// tests here (tests only read models and add benign extra runtime keys),
// so build them once instead of paying ~950ms across the 9 cases.
tempDir = TempDir.createSync("@pi-context-promotion-");
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
authStorage.setRuntimeApiKey("openai-codex", "test-key");
modelRegistry = new ModelRegistry(authStorage);
});
afterAll(() => {
authStorage.close();
tempDir.removeSync();
});
afterEach(async () => {
if (session) {
await session.dispose();
}
authStorage.close();
tempDir.removeSync();
});
function createOverflowMessage(
@@ -113,6 +120,17 @@ describe("AgentSession context promotion", () => {
throw new Error("Timed out waiting for condition");
}
// Deterministically drain the fire-and-forget `agent_end` handler that
// `emitExternalEvent` dispatches. The handler's terminal maintenance work
// (`#checkCompaction`) is microtask-based on the no-promotion paths, so a
// single macrotask turn fully flushes it; `waitForIdle` then settles any
// tracked continuation. Used by the negative tests, which assert that *no*
// promotion happened and therefore need the handler to have actually run.
async function settle(): Promise<void> {
await new Promise(resolve => setTimeout(resolve, 0));
await session.waitForIdle();
}
it("promotes to a larger-context model on overflow and clears codex websocket session state", async () => {
const sparkModel = modelRegistry.find("openai-codex", "gpt-5.3-codex-spark");
const codexModel = modelRegistry.find("openai-codex", "gpt-5.5");
@@ -390,7 +408,7 @@ describe("AgentSession context promotion", () => {
session.agent.emitExternalEvent({ type: "message_end", message: overflowMessage });
session.agent.emitExternalEvent({ type: "agent_end", messages: [overflowMessage] });
await Bun.sleep(30);
await settle();
expect(session.model?.provider).toBe(sparkModel.provider);
expect(session.model?.id).toBe(sparkModel.id);
@@ -472,7 +490,7 @@ describe("AgentSession context promotion", () => {
session.agent.emitExternalEvent({ type: "message_end", message: staleIncomplete });
session.agent.emitExternalEvent({ type: "agent_end", messages: [staleIncomplete] });
await Bun.sleep(30);
await settle();
expect(session.model?.provider).toBe(codexModel.provider);
expect(session.model?.id).toBe(codexModel.id);
@@ -0,0 +1,101 @@
import { afterEach, describe, expect, it } from "bun:test";
import * as path from "node:path";
import { Agent, AppendOnlyContextManager } from "@oh-my-pi/pi-agent-core";
import type { ProviderSessionState } from "@oh-my-pi/pi-ai";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { TempDir } from "@oh-my-pi/pi-utils";
interface FreshHarness {
agent: Agent;
session: AgentSession;
sessionManager: SessionManager;
}
const cleanup: Array<() => Promise<void>> = [];
afterEach(async () => {
while (cleanup.length > 0) {
const run = cleanup.pop();
if (run) await run();
}
});
async function createFreshHarness(): Promise<FreshHarness> {
const tempDir = TempDir.createSync("@pi-agent-session-fresh-");
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "auth.db"));
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
const sessionManager = SessionManager.create(tempDir.path(), path.join(tempDir.path(), "sessions"));
const agent = new Agent({
initialState: {
systemPrompt: ["Test"],
tools: [],
messages: [],
},
});
const session = new AgentSession({
agent,
sessionManager,
settings: Settings.isolated(),
modelRegistry,
});
cleanup.push(async () => {
await session.dispose();
authStorage.close();
tempDir.removeSync();
});
return { agent, session, sessionManager };
}
describe("AgentSession fresh provider state", () => {
it("rotates only the provider-facing session id and prunes cached stream state", async () => {
const { agent, session, sessionManager } = await createFreshHarness();
const persistedSessionId = sessionManager.getSessionId();
const persistedSessionFile = sessionManager.getSessionFile();
const persistedHeaderId = sessionManager.getHeader()?.id;
let closeCount = 0;
const providerState: ProviderSessionState = {
close() {
closeCount += 1;
},
};
session.providerSessionState.set("websocket", providerState);
const appendOnlyContext = new AppendOnlyContextManager();
agent.setAppendOnlyContext(appendOnlyContext);
appendOnlyContext.syncMessages([{ role: "user", content: "cached context" }]);
appendOnlyContext.build({ systemPrompt: ["Test"], messages: [], tools: [] }, { intentTracing: false });
const result = session.freshSession();
expect(result).toBeDefined();
if (!result) return;
expect(result.previousSessionId).toBe(persistedSessionId);
expect(result.sessionId).not.toBe(persistedSessionId);
expect(result.closedProviderSessions).toBe(1);
expect(agent.sessionId).toBe(result.sessionId);
expect(session.sessionId).toBe(result.sessionId);
expect(sessionManager.getSessionId()).toBe(persistedSessionId);
expect(sessionManager.getHeader()?.id).toBe(persistedHeaderId);
expect(sessionManager.getSessionFile()).toBe(persistedSessionFile);
expect(closeCount).toBe(1);
expect(session.providerSessionState.size).toBe(0);
expect(appendOnlyContext.log.length).toBe(0);
expect(appendOnlyContext.prefix.built).toBe(false);
});
it("drops the transient provider id when a real new session starts", async () => {
const { session, sessionManager } = await createFreshHarness();
const freshResult = session.freshSession();
expect(freshResult).toBeDefined();
if (!freshResult) return;
await session.newSession();
expect(session.sessionId).toBe(sessionManager.getSessionId());
expect(session.sessionId).not.toBe(freshResult.sessionId);
});
});
@@ -1,8 +1,8 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { Agent } from "@oh-my-pi/pi-agent-core";
import * as compactionModule from "@oh-my-pi/pi-agent-core/compaction";
import type { AssistantMessage, ToolCall } from "@oh-my-pi/pi-ai";
import type { AssistantMessage, Model, ToolCall } from "@oh-my-pi/pi-ai";
import { getBundledModel } from "@oh-my-pi/pi-ai/models";
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
@@ -14,26 +14,67 @@ import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manage
import { TempDir } from "@oh-my-pi/pi-utils";
describe("AgentSession handoff", () => {
// Immutable across the whole file: the model registry's synchronous bundled-model
// load dominates per-test setup (~100ms each), and the auth store + bundled model
// never change. Build them once. Per-test mutable state (session, session file,
// emitted events) is rebuilt in beforeEach.
let sharedDir: TempDir;
let authStorage: AuthStorage;
let modelRegistry: ModelRegistry;
let model: Model;
let tempDir: TempDir;
let session: AgentSession;
let sessionManager: SessionManager;
let authStorage: AuthStorage;
let modelRegistry: ModelRegistry;
let events: AgentSessionEvent[];
/** Poll `predicate` until it holds (returns as soon as the state is reached) or the
* deadline elapses. Replaces blind settle sleeps for tests with a positive signal. */
async function waitFor(predicate: () => boolean, timeoutMs = 1_000): Promise<void> {
const deadline = Date.now() + timeoutMs;
while (!predicate()) {
if (Date.now() >= deadline) {
throw new Error("Timed out waiting for condition");
}
await Bun.sleep(1);
}
}
/** Drain post-turn maintenance deterministically for negative tests (those proving
* maintenance did NOT run, where there is no positive signal to poll on). Post-turn
* work is scheduled fire-and-forget: a single event-loop turn lets the handler run to
* its decision and register any compaction pass as a tracked post-prompt task, then
* `waitForIdle()` drains that task to completion. */
async function drainMaintenance(): Promise<void> {
await Bun.sleep(0);
await session.waitForIdle();
}
beforeAll(async () => {
sharedDir = TempDir.createSync("@pi-handoff-shared-");
authStorage = await AuthStorage.create(path.join(sharedDir.path(), "testauth.db"));
authStorage.setRuntimeApiKey("anthropic", "test-key");
modelRegistry = new ModelRegistry(authStorage);
const bundled = getBundledModel("anthropic", "claude-sonnet-4-5");
if (!bundled) {
throw new Error("Expected built-in anthropic model to exist");
}
model = bundled;
});
afterAll(async () => {
authStorage.close();
try {
await sharedDir.remove();
} catch {}
});
beforeEach(async () => {
tempDir = TempDir.createSync("@pi-handoff-");
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
authStorage.setRuntimeApiKey("anthropic", "test-key");
modelRegistry = new ModelRegistry(authStorage);
sessionManager = SessionManager.create(tempDir.path(), tempDir.path());
events = [];
const model = getBundledModel("anthropic", "claude-sonnet-4-5");
if (!model) {
throw new Error("Expected built-in anthropic model to exist");
}
const agent = new Agent({
initialState: {
model,
@@ -85,7 +126,6 @@ describe("AgentSession handoff", () => {
if (session) {
await session.dispose();
}
authStorage.close();
try {
await tempDir.remove();
} catch {}
@@ -97,7 +137,7 @@ describe("AgentSession handoff", () => {
const generateHandoffSpy = vi.spyOn(compactionModule, "generateHandoff").mockResolvedValue(handoffText);
const result = await session.handoff();
await Bun.sleep(20);
await drainMaintenance();
expect(generateHandoffSpy).toHaveBeenCalledTimes(1);
expect(result?.document).toBe(handoffText);
@@ -123,7 +163,11 @@ describe("AgentSession handoff", () => {
});
await session.prompt("pending prompt ".repeat(120));
await Bun.sleep(20);
await waitFor(
() =>
compactSpy.mock.calls.length === 1 &&
events.some(event => event.type === "auto_compaction_end" && event.aborted === false),
);
expect(compactSpy).toHaveBeenCalledTimes(1);
expect(promptSpy).toHaveBeenCalledTimes(1);
@@ -177,7 +221,7 @@ describe("AgentSession handoff", () => {
isError: false,
});
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] });
await Bun.sleep(20);
await drainMaintenance();
expect(handoffSpy).not.toHaveBeenCalled();
expect(events.filter(event => event.type === "auto_compaction_start")).toHaveLength(0);
@@ -259,7 +303,7 @@ describe("AgentSession handoff", () => {
const handoffSpy = vi.spyOn(session, "handoff");
session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage });
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] });
await Bun.sleep(20);
await drainMaintenance();
expect(handoffSpy).not.toHaveBeenCalled();
expect(events.filter(event => event.type === "auto_compaction_start")).toHaveLength(0);
@@ -307,7 +351,7 @@ describe("AgentSession handoff", () => {
session.agent.emitExternalEvent({ type: "message_end", message: overflowAssistant });
session.agent.emitExternalEvent({ type: "agent_end", messages: [overflowAssistant] });
await Bun.sleep(20);
await waitFor(() => events.filter(event => event.type === "auto_compaction_end").length === 1);
expect(handoffSpy).not.toHaveBeenCalled();
const startEvents = events.filter(event => event.type === "auto_compaction_start");
@@ -352,7 +396,11 @@ describe("AgentSession handoff", () => {
session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage });
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] });
await Bun.sleep(20);
await waitFor(
() =>
handoffSpy.mock.calls.length === 1 &&
events.filter(event => event.type === "auto_compaction_end").length === 1,
);
expect(handoffSpy).toHaveBeenCalledTimes(1);
expect(handoffSpy).toHaveBeenCalledWith(expect.stringContaining("Threshold-triggered maintenance"), {
@@ -500,7 +548,8 @@ describe("AgentSession handoff", () => {
session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage });
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] });
await Bun.sleep(20);
await waitFor(() => handoffSpy.mock.calls.length === 1);
await session.waitForIdle();
expect(handoffSpy).toHaveBeenCalledTimes(1);
// The bug surfaced as agent.continue() racing the deferred handoff. With the fix,
@@ -554,7 +603,7 @@ describe("AgentSession handoff", () => {
session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage });
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] });
// Let the deferred handoff post-prompt task enter the generateHandoff await.
await Bun.sleep(20);
await waitFor(() => session.isGeneratingHandoff);
expect(generateHandoffSpy).toHaveBeenCalledTimes(1);
expect(session.isGeneratingHandoff).toBe(true);
@@ -601,7 +650,7 @@ describe("AgentSession handoff", () => {
session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage });
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] });
await Bun.sleep(20);
await waitFor(() => events.filter(event => event.type === "auto_compaction_end").length === 1);
expect(handoffSpy).toHaveBeenCalledTimes(1);
const endEvents = events.filter(event => event.type === "auto_compaction_end");
@@ -1,4 +1,4 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test";
import * as path from "node:path";
import { Agent } from "@oh-my-pi/pi-agent-core";
import { type Api, Effort, getBundledModel, type Model } from "@oh-my-pi/pi-ai";
@@ -18,7 +18,24 @@ describe("AgentSession model persistence", () => {
let tempDir: TempDir;
let session: AgentSession | undefined;
let sessionSettings: Settings;
const authStorages: AuthStorage[] = [];
// Auth storage (SQLite DB) and the model registry are immutable across these tests:
// every test sets the same anthropic runtime key and only ever reads the bundled model
// list. Building them once avoids ~12 SQLite opens + registry constructions.
let sharedDir: TempDir;
let sharedAuthStorage: AuthStorage;
let sharedModelRegistry: ModelRegistry;
beforeAll(async () => {
sharedDir = TempDir.createSync("@pi-model-persistence-shared-");
sharedAuthStorage = await AuthStorage.create(path.join(sharedDir.path(), "auth.db"));
sharedAuthStorage.setRuntimeApiKey("anthropic", "test-key");
sharedModelRegistry = new ModelRegistry(sharedAuthStorage, path.join(sharedDir.path(), "models.yml"));
});
afterAll(() => {
sharedAuthStorage.close();
sharedDir.removeSync();
});
beforeEach(() => {
tempDir = TempDir.createSync("@pi-model-persistence-");
@@ -29,9 +46,6 @@ describe("AgentSession model persistence", () => {
await session.dispose();
session = undefined;
}
for (const authStorage of authStorages.splice(0)) {
authStorage.close();
}
tempDir.removeSync();
});
@@ -84,13 +98,7 @@ describe("AgentSession model persistence", () => {
modelRoles?: Record<string, string>;
persist?: boolean;
}): Promise<{ modelRegistry: ModelRegistry; settings: Settings; session: AgentSession }> {
const authStorage = await AuthStorage.create(path.join(tempDir.path(), `testauth-${authStorages.length}.db`));
authStorages.push(authStorage);
authStorage.setRuntimeApiKey("anthropic", "test-key");
const modelRegistry = new ModelRegistry(
authStorage,
path.join(tempDir.path(), `models-${authStorages.length}.yml`),
);
const modelRegistry = sharedModelRegistry;
const model =
options?.initialModel ??
options?.selectInitialModel?.(modelRegistry.getAvailable()) ??
@@ -118,7 +126,7 @@ describe("AgentSession model persistence", () => {
session = new AgentSession({
agent,
sessionManager: options?.persist
? SessionManager.create(tempDir.path(), path.join(tempDir.path(), `active-${authStorages.length}`))
? SessionManager.create(tempDir.path(), path.join(tempDir.path(), "active"))
: SessionManager.inMemory(),
settings: sessionSettings,
modelRegistry,
@@ -131,19 +139,12 @@ describe("AgentSession model persistence", () => {
targetSessionFile: string,
settings: Settings = Settings.isolated(),
): Promise<CreateAgentSessionResult> {
const authStorage = await AuthStorage.create(path.join(tempDir.path(), `testauth-${authStorages.length}.db`));
authStorages.push(authStorage);
authStorage.setRuntimeApiKey("anthropic", "test-key");
const modelRegistry = new ModelRegistry(
authStorage,
path.join(tempDir.path(), `models-${authStorages.length}.yml`),
);
const sessionManager = await SessionManager.open(targetSessionFile, path.join(tempDir.path(), "startup"));
const result = await createAgentSession({
cwd: tempDir.path(),
agentDir: tempDir.path(),
authStorage,
modelRegistry,
authStorage: sharedAuthStorage,
modelRegistry: sharedModelRegistry,
sessionManager,
settings,
disableExtensionDiscovery: true,
@@ -1,126 +0,0 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { Agent } from "@oh-my-pi/pi-agent-core";
import { getBundledModel } from "@oh-my-pi/pi-ai";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
import { TodoTool } from "@oh-my-pi/pi-coding-agent/tools";
import { Snowflake } from "@oh-my-pi/pi-utils";
/**
* Regression test: /new (AgentSession.newSession) must fully switch to a new session file
* before the call resolves.
*
* If it doesn't, UI code that reloads todos immediately after /new will read the old
* session artifact dir and keep showing stale todos.
*/
describe("AgentSession newSession clears todo artifacts", () => {
let tempDir: string;
let session: AgentSession;
let sessionManager: SessionManager;
let authStorage: AuthStorage | undefined;
beforeEach(async () => {
tempDir = path.join(os.tmpdir(), `pi-new-session-todos-test-${Snowflake.next()}`);
fs.mkdirSync(tempDir, { recursive: true });
sessionManager = SessionManager.create(tempDir, tempDir);
const settings = Settings.isolated();
authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db"));
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml"));
const model = getBundledModel("anthropic", "claude-sonnet-4-5");
if (!model) {
throw new Error("Test model not found in registry");
}
const toolSession: ToolSession = {
cwd: tempDir,
hasUI: false,
getSessionFile: () => sessionManager.getSessionFile() ?? null,
getSessionSpawns: () => "*",
settings,
};
const agent = new Agent({
getApiKey: () => "test",
initialState: {
model,
systemPrompt: ["test"],
tools: [new TodoTool(toolSession)],
},
});
session = new AgentSession({
agent,
sessionManager,
settings,
modelRegistry,
});
// Must subscribe to enable session persistence hooks
session.subscribe(() => {});
});
afterEach(async () => {
if (session) {
await session.dispose();
}
authStorage?.close();
authStorage = undefined;
if (tempDir && fs.existsSync(tempDir)) {
fs.rmSync(tempDir, { recursive: true });
}
});
it("should not carry over todo state to the new session branch", async () => {
const oldSessionFile = session.sessionFile;
expect(oldSessionFile).toBeDefined();
session.setTodoPhases([
{
name: "Tasks",
tasks: [{ content: "do the thing", status: "pending" }],
},
]);
expect(session.getTodoPhases()).toHaveLength(1);
expect(session.getTodoPhases()[0]?.tasks).toHaveLength(1);
await session.newSession();
const newSessionFile = session.sessionFile;
expect(newSessionFile).toBeDefined();
expect(newSessionFile).not.toBe(oldSessionFile);
expect(session.getTodoPhases()).toHaveLength(0);
});
it("should clear stale todo cache when branching from the first user message", async () => {
sessionManager.appendMessage({
role: "user",
content: "start task",
timestamp: Date.now(),
});
const branchCandidates = session.getUserMessagesForBranching();
expect(branchCandidates).toHaveLength(1);
session.setTodoPhases([
{
name: "Execution",
tasks: [{ content: "stale from old branch", status: "in_progress" }],
},
]);
expect(session.getTodoPhases()).toHaveLength(1);
const result = await session.branch(branchCandidates[0].entryId);
expect(result.cancelled).toBe(false);
expect(result.selectedText).toBe("start task");
expect(session.getTodoPhases()).toHaveLength(0);
});
});
@@ -1,4 +1,4 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
@@ -12,8 +12,11 @@ import type {
Usage,
} from "@oh-my-pi/pi-ai/types";
import { createOpenAIResponsesHistoryPayload } from "@oh-my-pi/pi-ai/utils";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk";
import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import type { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import {
type SessionEntry,
SessionManager,
@@ -209,20 +212,21 @@ async function createPersistedSession(
return { sessionFile, treeTargetId: result?.treeTargetId };
}
// ModelRegistry construction loads the bundled model catalog plus the on-disk
// cache (~100ms) and dominated this file's runtime when rebuilt once per test.
// The registry and its pinned AuthStorage are immutable across these tests (none
// mutate the catalog or stored credentials), so a single instance is shared via
// createAgentSession's `modelRegistry`/`authStorage` seam. The registry pins
// itself to its AuthStorage, so both MUST be the same shared instances.
let sharedModelRegistry: ModelRegistry;
let sharedRegistryDir: string;
async function createSessionHarness(
tempDir: string,
sessionManager: SessionManager,
options: { provider?: Parameters<typeof getBundledModel>[0]; modelId?: string } = {},
): Promise<{ session: AgentSession; authStorage: AuthStorage }> {
): Promise<{ session: AgentSession }> {
const { provider = "openai", modelId = "gpt-5-mini" } = options;
const [{ createAgentSession }, { Settings }, { AuthStorage }] = await Promise.all([
import("@oh-my-pi/pi-coding-agent/sdk"),
import("@oh-my-pi/pi-coding-agent/config/settings"),
import("@oh-my-pi/pi-coding-agent/session/auth-storage"),
]);
const authStorage = await AuthStorage.create(path.join(tempDir, `testauth-${Snowflake.next()}.db`));
authStorage.setRuntimeApiKey("openai", "test-key");
authStorage.setRuntimeApiKey("openai-codex", "test-key");
const model = getBundledModel(provider, modelId);
if (!model) {
throw new Error(`Expected bundled test model ${provider}/${modelId}`);
@@ -231,7 +235,8 @@ async function createSessionHarness(
const { session } = await createAgentSession({
cwd: tempDir,
agentDir: tempDir,
authStorage,
authStorage: sharedModelRegistry.authStorage,
modelRegistry: sharedModelRegistry,
sessionManager,
model,
settings: Settings.isolated(),
@@ -242,23 +247,42 @@ async function createSessionHarness(
slashCommands: [],
enableMCP: false,
enableLsp: false,
// These tests exercise session reload/sanitization/provider-state, never tool
// execution, rule resolution, or the workspace-tree render. A minimal tool set
// plus empty rules and a prebuilt (empty) workspace tree skip the per-call
// startup scans (native listWorkspace + rule capability discovery) without
// touching any asserted behavior.
rules: [],
workspaceTree: { rootPath: tempDir, rendered: "", truncated: false, totalLines: 0, agentsMdFiles: [] },
toolNames: ["read"],
});
return { session, authStorage };
return { session };
}
describe("AgentSession OpenAI Responses replay boundaries", () => {
const sessions: AgentSession[] = [];
const authStorages: AuthStorage[] = [];
const tempDirs: string[] = [];
beforeAll(async () => {
sharedRegistryDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-issue-505-registry-${Snowflake.next()}-`));
const authStorage = await AuthStorage.create(path.join(sharedRegistryDir, "auth.db"));
authStorage.setRuntimeApiKey("openai", "test-key");
authStorage.setRuntimeApiKey("openai-codex", "test-key");
sharedModelRegistry = new ModelRegistry(authStorage);
});
afterAll(() => {
sharedModelRegistry?.authStorage.close();
if (sharedRegistryDir && fs.existsSync(sharedRegistryDir)) {
fs.rmSync(sharedRegistryDir, { recursive: true, force: true });
}
});
afterEach(async () => {
while (sessions.length > 0) {
await sessions.pop()?.dispose();
}
while (authStorages.length > 0) {
authStorages.pop()?.close();
}
while (tempDirs.length > 0) {
const tempDir = tempDirs.pop();
if (tempDir && fs.existsSync(tempDir)) {
@@ -285,9 +309,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
});
const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir);
const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager);
const { session } = await createSessionHarness(tempDir, reloadedSessionManager);
sessions.push(session);
authStorages.push(authStorage);
const persistedUser = findPersistedMessageEntry(session.sessionManager, "user", "Preserved summary").message;
if (persistedUser.role !== "user") {
@@ -383,12 +406,11 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
});
const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir);
const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager, {
const { session } = await createSessionHarness(tempDir, reloadedSessionManager, {
provider: "openai-codex",
modelId: "gpt-5.2-codex",
});
sessions.push(session);
authStorages.push(authStorage);
const closeSpy = vi.fn();
session.providerSessionState.set("openai-codex-responses", { close: closeSpy } satisfies ProviderSessionState);
@@ -421,12 +443,11 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
});
const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir);
const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager, {
const { session } = await createSessionHarness(tempDir, reloadedSessionManager, {
provider: "openai-codex",
modelId: "gpt-5.2-codex",
});
sessions.push(session);
authStorages.push(authStorage);
const closeSpy = vi.fn();
session.providerSessionState.set("openai-codex-responses", { close: closeSpy } satisfies ProviderSessionState);
@@ -476,9 +497,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-issue-505-reload-proxy-${Snowflake.next()}-`));
tempDirs.push(tempDir);
const sessionManager = SessionManager.create(tempDir, tempDir);
const { session, authStorage } = await createSessionHarness(tempDir, sessionManager);
const { session } = await createSessionHarness(tempDir, sessionManager);
sessions.push(session);
authStorages.push(authStorage);
const proxyDetails = new Proxy({ ok: true, nested: { value: "preserved" } }, {});
await session.sendCustomMessage(
@@ -496,7 +516,6 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
await session.reload();
expect(() => session.sessionManager.captureState()).not.toThrow();
expect(session.sessionFile).toBe(originalSessionFile);
});
@@ -515,12 +534,11 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
});
const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir);
const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager, {
const { session } = await createSessionHarness(tempDir, reloadedSessionManager, {
provider: "openai-codex",
modelId: "gpt-5.2-codex",
});
sessions.push(session);
authStorages.push(authStorage);
const closeSpy = vi.fn();
session.providerSessionState.set("openai-codex-responses", { close: closeSpy } satisfies ProviderSessionState);
@@ -562,12 +580,11 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
});
const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir);
const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager, {
const { session } = await createSessionHarness(tempDir, reloadedSessionManager, {
provider: "openai-codex",
modelId: "gpt-5.2-codex",
});
sessions.push(session);
authStorages.push(authStorage);
const closeSpy = vi.fn();
session.providerSessionState.set("openai-codex-responses", { close: closeSpy } satisfies ProviderSessionState);
@@ -598,9 +615,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
});
const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir);
const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager);
const { session } = await createSessionHarness(tempDir, reloadedSessionManager);
sessions.push(session);
authStorages.push(authStorage);
const closeSpy = vi.fn();
session.providerSessionState.set("openai-responses:openai", { close: closeSpy } satisfies ProviderSessionState);
@@ -624,9 +640,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-issue-505-switch-fail-${Snowflake.next()}-`));
tempDirs.push(tempDir);
const currentSessionManager = SessionManager.create(tempDir, tempDir);
const { session, authStorage } = await createSessionHarness(tempDir, currentSessionManager);
const { session } = await createSessionHarness(tempDir, currentSessionManager);
sessions.push(session);
authStorages.push(authStorage);
const { sessionFile } = await createPersistedSession(tempDir, sessionManager => {
appendStaleAssistantTurn(sessionManager, "Unreadable assistant snapshot");
@@ -658,9 +673,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
const assistantText = "Switched assistant response";
const currentSessionManager = SessionManager.create(tempDir, tempDir);
const { session, authStorage } = await createSessionHarness(tempDir, currentSessionManager);
const { session } = await createSessionHarness(tempDir, currentSessionManager);
sessions.push(session);
authStorages.push(authStorage);
const { sessionFile } = await createPersistedSession(tempDir, sessionManager => {
sessionManager.appendMessage({
@@ -724,9 +738,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
}
const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir);
const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager);
const { session } = await createSessionHarness(tempDir, reloadedSessionManager);
sessions.push(session);
authStorages.push(authStorage);
const navigation = await session.navigateTree(treeTargetId, { summarize: false });
expect(navigation.cancelled).toBe(false);
@@ -747,9 +760,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => {
const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-issue-505-new-${Snowflake.next()}-`));
tempDirs.push(tempDir);
const sessionManager = SessionManager.create(tempDir, tempDir);
const { session, authStorage } = await createSessionHarness(tempDir, sessionManager);
const { session } = await createSessionHarness(tempDir, sessionManager);
sessions.push(session);
authStorages.push(authStorage);
const closeSpy = vi.fn();
session.providerSessionState.set("live-provider-session", { close: closeSpy } satisfies ProviderSessionState);
@@ -116,6 +116,7 @@ const createSession = async (
disableExtensionDiscovery: true,
extensions: options.extensions,
skills: [],
rules: [],
contextFiles: [],
promptTemplates: [],
workspaceTree: emptyWorkspaceTree(cwd),
@@ -195,6 +196,7 @@ describe("AgentSession python cleanup", () => {
disableExtensionDiscovery: true,
extensions: [throwingExtension],
skills: [],
rules: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
@@ -261,6 +263,7 @@ describe("AgentSession python cleanup", () => {
model: getModel(),
disableExtensionDiscovery: true,
skills: [],
rules: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
@@ -2,7 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { scheduler } from "node:timers/promises";
import { Agent } from "@oh-my-pi/pi-agent-core";
import { type AssistantMessage, getBundledModel } from "@oh-my-pi/pi-ai";
import { type ApiKeyResolveContext, type AssistantMessage, getBundledModel } from "@oh-my-pi/pi-ai";
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
@@ -22,6 +22,16 @@ function lastAssistant(session: AgentSession): AssistantMessage {
return message as AssistantMessage;
}
function resolveInitialApiKey(
apiKey: string | ((ctx: ApiKeyResolveContext) => string | Promise<string | undefined> | undefined) | undefined,
): string {
const resolved = typeof apiKey === "function" ? apiKey({ lastChance: false, error: undefined }) : apiKey;
if (typeof resolved !== "string") {
throw new Error("Expected API key to be resolved before streaming");
}
return resolved;
}
/**
* Contract: when the provider asks us to wait longer than `retry.maxDelayMs`
* and we have no credential/model fallback to switch to, the auto-retry
@@ -155,10 +165,7 @@ describe("AgentSession retry delay cap", () => {
messages: [],
},
streamFn: (requestedModel, context, options) => {
const apiKey = options?.apiKey;
if (typeof apiKey !== "string") {
throw new Error("Expected API key to be resolved before streaming");
}
const apiKey = resolveInitialApiKey(options?.apiKey);
requestedKeys.push(apiKey);
if (requestedKeys.length === 1) {
mock.push({ throw: rateLimitError });
@@ -1,4 +1,4 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { scheduler } from "node:timers/promises";
import { Agent } from "@oh-my-pi/pi-agent-core";
@@ -66,16 +66,34 @@ function createFallbackAgent(primaryModel: Model, requestedModels: string[]): Ag
describe("AgentSession retry fallback", () => {
let tempDir: TempDir;
let authStorage: AuthStorage;
let sharedRegistry: ModelRegistry;
let modelRegistry: ModelRegistry;
let session: AgentSession | undefined;
beforeEach(async () => {
// The model registry is an immutable fixture whose construction builds a
// canonical index over ~2.7k bundled models (~100ms). Build it (and the
// auth DB) once for the whole file instead of per-test; reset only the
// mutable retry-fallback cooldown state between tests.
beforeAll(async () => {
tempDir = TempDir.createSync("@pi-retry-fallback-");
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
authStorage.setRuntimeApiKey("anthropic", "anthropic-test-key");
authStorage.setRuntimeApiKey("openai", "openai-test-key");
authStorage.setRuntimeApiKey("google", "google-test-key");
modelRegistry = new ModelRegistry(authStorage);
sharedRegistry = new ModelRegistry(authStorage);
});
afterAll(() => {
authStorage.close();
tempDir.removeSync();
});
beforeEach(() => {
// Reset to the shared registry (a few tests reassign it to a scoped
// instance) and clear cooldown suppressions left by fallback-path tests
// (default 5-minute suppression) so state never leaks between tests.
modelRegistry = sharedRegistry;
modelRegistry.clearSuppressedSelectors();
});
afterEach(async () => {
@@ -83,8 +101,6 @@ describe("AgentSession retry fallback", () => {
await session.dispose();
session = undefined;
}
authStorage.close();
tempDir.removeSync();
vi.restoreAllMocks();
});
@@ -257,6 +273,87 @@ describe("AgentSession retry fallback", () => {
expect(lastAssistant.content).toContainEqual({ type: "text", text: "Recovered after Google quota retry" });
});
it("keeps retry on the primary model when retry model fallback is disabled", async () => {
const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5");
const fallbackModel = getBundledModel("openai", "gpt-4o-mini");
if (!primaryModel || !fallbackModel) {
throw new Error("Expected bundled test models to exist");
}
const requestedModels: string[] = [];
const fallbackAppliedEvents: Array<Extract<AgentSessionEvent, { type: "retry_fallback_applied" }>> = [];
const fallbackSucceededEvents: Array<Extract<AgentSessionEvent, { type: "retry_fallback_succeeded" }>> = [];
const mock = createMockModel({
responses: [{ throw: "rate limit exceeded retry-after-ms=200" }, { content: ["Recovered on primary retry"] }],
});
const agent = new Agent({
getApiKey: provider => `${provider}-test-key`,
initialState: {
model: primaryModel,
systemPrompt: ["Test"],
tools: [],
messages: [],
},
streamFn: (requestedModel, context, options) => {
requestedModels.push(`${requestedModel.provider}/${requestedModel.id}`);
return mock.stream(requestedModel, context, options);
},
});
const settings = Settings.isolated({
"compaction.enabled": false,
"retry.baseDelayMs": 5,
"retry.maxRetries": 1,
"retry.modelFallback": false,
"retry.fallbackChains": {
default: [`${fallbackModel.provider}/${fallbackModel.id}`],
},
});
settings.setModelRole("default", `${primaryModel.provider}/${primaryModel.id}`);
session = new AgentSession({
agent,
sessionManager: SessionManager.inMemory(),
settings,
modelRegistry,
});
const waitSpy = vi.spyOn(scheduler, "wait").mockResolvedValue(undefined);
const { retryStartEvents, retryEndEvents } = trackRetryEvents(session);
session.subscribe(event => {
if (event.type === "retry_fallback_applied") {
fallbackAppliedEvents.push(event);
}
if (event.type === "retry_fallback_succeeded") {
fallbackSucceededEvents.push(event);
}
});
await session.prompt("Retry rate limit without switching models");
await session.waitForIdle();
expect(requestedModels).toEqual([
`${primaryModel.provider}/${primaryModel.id}`,
`${primaryModel.provider}/${primaryModel.id}`,
]);
expect(retryStartEvents).toHaveLength(1);
expect(retryStartEvents[0]).toMatchObject({
attempt: 1,
maxAttempts: 1,
delayMs: 200,
errorMessage: "rate limit exceeded retry-after-ms=200",
});
expect(waitSpy).toHaveBeenCalledWith(200, { signal: expect.any(AbortSignal) });
expect(retryEndEvents).toHaveLength(1);
expect(retryEndEvents[0]).toMatchObject({ success: true, attempt: 1 });
expect(fallbackAppliedEvents).toHaveLength(0);
expect(fallbackSucceededEvents).toHaveLength(0);
expect(session.model?.provider).toBe(primaryModel.provider);
expect(session.model?.id).toBe(primaryModel.id);
const lastAssistant = getLastAssistantMessage(session);
expect(lastAssistant.stopReason).toBe("stop");
expect(lastAssistant.content).toContainEqual({ type: "text", text: "Recovered on primary retry" });
});
it("auto-retries preserved OpenAI first-event timeout errors", async () => {
const model = getBundledModel("openai", "gpt-4o-mini");
if (!model) {
@@ -68,16 +68,16 @@ function createPiHarness(initialTools: string[] = []): PiHarness {
return { api, activeTools, appendEntries, setActiveToolsCalls };
}
async function initGitRepo(dir: string): Promise<{ baselineCommit: string; mainBranch: string }> {
await $`git init --initial-branch=main`.cwd(dir).quiet();
await $`git config user.email tester@example.com`.cwd(dir).quiet();
await $`git config user.name Tester`.cwd(dir).quiet();
async function initGitRepo(dir: string): Promise<{ baselineCommit: string }> {
await Bun.write(path.join(dir, "README.md"), "# baseline\n");
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m baseline`.cwd(dir).quiet();
// One shell invocation instead of five: the git processes are unavoidable
// (config identity is read by the production tool's own commits), but
// chaining collapses the per-call Node↔shell spawn overhead.
await $`git init --initial-branch=main && git config user.email tester@example.com && git config user.name Tester && git add -A && git commit -m baseline`
.cwd(dir)
.quiet();
const sha = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
const branch = (await $`git rev-parse --abbrev-ref HEAD`.cwd(dir).text()).trim();
return { baselineCommit: sha, mainBranch: branch };
return { baselineCommit: sha };
}
async function checkoutBranch(dir: string, name: string): Promise<void> {
@@ -585,8 +585,7 @@ describe("log_experiment", () => {
// Commit `src/edit-me.ts` to baseline so it is tracked, not in pre-run dirty paths.
fs.mkdirSync(path.join(dir, "src"), { recursive: true });
await Bun.write(path.join(dir, "src", "edit-me.ts"), "export const v = 1;\n");
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m seed`.cwd(dir).quiet();
await $`git add -A && git commit -m seed`.cwd(dir).quiet();
const runtime = createSessionRuntime();
const harness = createPiHarness();
const init = createInitExperimentTool({
@@ -638,8 +637,7 @@ describe("log_experiment", () => {
await initGitRepo(dir);
// Commit the harness on main so it is part of the autoresearch branch's baseline.
await writeHarnessStub(dir);
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m harness`.cwd(dir).quiet();
await $`git add -A && git commit -m harness`.cwd(dir).quiet();
await checkoutBranch(dir, "autoresearch/test-20260501");
const runtime = createSessionRuntime();
const harness = createPiHarness();
@@ -651,8 +649,7 @@ describe("log_experiment", () => {
await init.execute("i", { name: "x", primary_metric: "m" }, undefined, undefined, createCtx(dir));
// Simulate a previously kept iteration by committing it directly on the branch.
await Bun.write(path.join(dir, "src", "kept.ts"), "export const v = 1;\n");
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m "kept iteration"`.cwd(dir).quiet();
await $`git add -A && git commit -m "kept iteration"`.cwd(dir).quiet();
const headBeforeDiscard = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
const run = createRunExperimentTool({
@@ -691,13 +688,11 @@ describe("log_experiment", () => {
const dir = makeTempDir();
await initGitRepo(dir);
await writeHarnessStub(dir);
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m harness`.cwd(dir).quiet();
await $`git add -A && git commit -m harness`.cwd(dir).quiet();
// Seed a tracked file that the agent will edit during the iteration.
fs.mkdirSync(path.join(dir, "src"), { recursive: true });
await Bun.write(path.join(dir, "src", "store.ts"), "export const v = 1;\n");
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m seed`.cwd(dir).quiet();
await $`git add -A && git commit -m seed`.cwd(dir).quiet();
await checkoutBranch(dir, "autoresearch/keep-test");
const runtime = createSessionRuntime();
const harness = createPiHarness();
@@ -747,8 +742,7 @@ describe("log_experiment", () => {
const dir = makeTempDir();
await initGitRepo(dir);
await writeHarnessStub(dir);
await $`git add -A`.cwd(dir).quiet();
await $`git commit -m harness`.cwd(dir).quiet();
await $`git add -A && git commit -m harness`.cwd(dir).quiet();
await checkoutBranch(dir, "autoresearch/scope-test");
const runtime = createSessionRuntime();
const harness = createPiHarness();
@@ -216,14 +216,6 @@ describe("BashExecutionComponent #clampDisplayLine", () => {
expect(output).toContain(`[1 visible columns omitted]`);
});
it("handles string with 0 visible width (empty after ANSI removal)", () => {
const onlyAnsi = "\x1b[0m\x1b[1m\x1b[2m";
const component = createComponentWithOutput(onlyAnsi);
const output = component.getOutput();
expect(output).toBeDefined();
});
it("handles empty string", () => {
const component = createComponentWithOutput("");
const output = component.getOutput();
@@ -14,12 +14,27 @@ import * as piNatives from "@oh-my-pi/pi-natives";
const ARTIFACT_HEAD_BYTES_DEFAULT = 20 * 1024;
const BACKGROUND_COMPLETION_RACE_MS = 750;
const KILL_MARKER_DELAY_SECONDS = "0.4";
const KILL_MARKER_ASSERTION_WAIT_MS = 900;
const KILL_MARKER_DELAY_MS = 400;
// We prove a killed process never wrote its marker by observing until the
// wall-clock instant the marker WOULD have appeared (spawn + delay) plus a
// margin. Anchoring the deadline to a pre-spawn timestamp — instead of blindly
// sleeping a fixed amount after executeBash returns — keeps the wait bounded
// without shrinking the kill-propagation margin: the timeout/abort fires at
// ~100ms, well before the 400ms marker write, so the margin between kill and
// write is unchanged; only the redundant observation tail goes away.
const KILL_MARKER_OBSERVE_MARGIN_MS = 300;
function makeTempDir(): string {
return fs.mkdtempSync(path.join(os.tmpdir(), "omp-bash-exec-"));
}
/** Spin-wait until the wall-clock deadline, polling rather than blind-sleeping. */
async function waitUntil(deadlineMs: number): Promise<void> {
while (Date.now() < deadlineMs) {
await Bun.sleep(20);
}
}
describe("executeBash", () => {
let tempDir: string;
@@ -532,6 +547,7 @@ describe("executeBash", () => {
const markerEscaped = marker.replace(/'/g, "'\\''");
// Command creates marker after a short delay, but we timeout before then.
const start = Date.now();
const result = await executeBash(`sleep ${KILL_MARKER_DELAY_SECONDS} && echo done > '${markerEscaped}'`, {
cwd: tempDir,
timeout: 100,
@@ -539,10 +555,9 @@ describe("executeBash", () => {
expect(result.cancelled).toBe(true);
// Wait longer than the command would have needed to create the marker.
await Bun.sleep(KILL_MARKER_ASSERTION_WAIT_MS);
// If process was killed (not orphaned), marker should NOT exist
// Observe past the instant the marker would have been written had the
// process survived. If it was killed (not orphaned), it never appears.
await waitUntil(start + KILL_MARKER_DELAY_MS + KILL_MARKER_OBSERVE_MARGIN_MS);
expect(fs.existsSync(marker)).toBe(false);
});
@@ -552,6 +567,7 @@ describe("executeBash", () => {
const marker = path.join(tempDir, "marker-bg.txt");
const markerEscaped = marker.replace(/'/g, "'\\''");
const start = Date.now();
const result = await executeBash(
`{ sleep ${KILL_MARKER_DELAY_SECONDS}; echo done > '${markerEscaped}'; } & sleep 10`,
{
@@ -562,7 +578,7 @@ describe("executeBash", () => {
expect(result.cancelled).toBe(true);
await Bun.sleep(KILL_MARKER_ASSERTION_WAIT_MS);
await waitUntil(start + KILL_MARKER_DELAY_MS + KILL_MARKER_OBSERVE_MARGIN_MS);
expect(fs.existsSync(marker)).toBe(false);
});
@@ -597,9 +613,13 @@ describe("executeBash", () => {
expect(result.cancelled).toBe(true);
expect(result.output).toContain("Command cancelled");
await Bun.sleep(KILL_MARKER_ASSERTION_WAIT_MS);
// The backgrounded subshell only writes its marker once `release` exists.
// If abort failed to kill the process group, the orphan is still polling
// for `release` every 50ms — touching it makes a survivor react within one
// poll. A short settle first lets the kill signal propagate before we probe.
await Bun.sleep(100);
fs.writeFileSync(release, "");
await Bun.sleep(150);
await Bun.sleep(200);
expect(fs.existsSync(marker)).toBe(false);
});
@@ -611,6 +631,7 @@ describe("executeBash", () => {
const controller = new AbortController();
// Command creates marker after a short delay.
const start = Date.now();
const promise = executeBash(`sleep ${KILL_MARKER_DELAY_SECONDS} && echo done > '${markerEscaped}'`, {
cwd: tempDir,
timeout: 10000,
@@ -625,10 +646,9 @@ describe("executeBash", () => {
expect(result.cancelled).toBe(true);
expect(result.output).toContain("Command cancelled");
// Wait longer than the command would have needed to create the marker.
await Bun.sleep(KILL_MARKER_ASSERTION_WAIT_MS);
// If process was killed (not orphaned), marker should NOT exist
// Observe past the instant the marker would have been written had the
// process survived. If it was killed (not orphaned), it never appears.
await waitUntil(start + KILL_MARKER_DELAY_MS + KILL_MARKER_OBSERVE_MARGIN_MS);
expect(fs.existsSync(marker)).toBe(false);
});
});
@@ -96,4 +96,23 @@ describe("bucketRules", () => {
expect(mgr.checkDelta("contains FORBIDDEN token", { source: "text" }).map(r => r.name)).toEqual(["builtin-foo"]);
});
it("falls condition rules through to the rulebook when ttsr is disabled on the manager", () => {
const mgr = new TtsrManager({
enabled: false,
contextMode: "discard",
interruptMode: "always",
repeatMode: "once",
repeatGap: 10,
});
const ttsr = makeRule({ name: "no-foo", condition: ["FORBIDDEN"], description: "blocks foo" });
const { rulebookRules, alwaysApplyRules } = bucketRules([ttsr], mgr);
// Manager refused to register; condition rule degrades to its rulebook shape.
expect(mgr.hasRules()).toBe(false);
expect(mgr.checkDelta("contains FORBIDDEN token", { source: "text" })).toEqual([]);
expect(alwaysApplyRules.map(r => r.name)).toEqual([]);
expect(rulebookRules.map(r => r.name)).toEqual(["no-foo"]);
});
});
@@ -157,24 +157,6 @@ describe("serverSupportsResourceSubscriptions", () => {
});
describe("subscribeToResources", () => {
it("no-ops on empty URI array", async () => {
const transport = createMockTransport(new Map());
const conn = createMockConnection({ resources: { subscribe: true } }, transport);
await subscribeToResources(conn, []);
});
it("no-ops when server lacks subscribe capability", async () => {
const transport = createMockTransport(new Map());
const conn = createMockConnection({ resources: {} }, transport);
await subscribeToResources(conn, ["test://a"]);
});
it("sends resources/subscribe for each URI", async () => {
const transport = createMockTransport(new Map([["resources/subscribe", [{}, {}]]]));
const conn = createMockConnection({ resources: { subscribe: true } }, transport);
await subscribeToResources(conn, ["test://a", "test://b"]);
});
it("does not throw when one subscription fails", async () => {
const transport: MCPTransport = {
connected: true,
@@ -191,24 +173,6 @@ describe("subscribeToResources", () => {
});
describe("unsubscribeFromResources", () => {
it("no-ops on empty URI array", async () => {
const transport = createMockTransport(new Map());
const conn = createMockConnection({ resources: { subscribe: true } }, transport);
await unsubscribeFromResources(conn, []);
});
it("no-ops when server lacks subscribe capability", async () => {
const transport = createMockTransport(new Map());
const conn = createMockConnection({ resources: {} }, transport);
await unsubscribeFromResources(conn, ["test://a"]);
});
it("sends resources/unsubscribe for each URI", async () => {
const transport = createMockTransport(new Map([["resources/unsubscribe", [{}, {}]]]));
const conn = createMockConnection({ resources: { subscribe: true } }, transport);
await unsubscribeFromResources(conn, ["test://a", "test://b"]);
});
it("does not throw when one unsubscription fails", async () => {
const transport: MCPTransport = {
connected: true,
@@ -38,6 +38,9 @@ describe("commit role thinking selection", () => {
const registry = {
getAvailable: () => [defaultModel, commitModel],
getApiKey: async () => "test-key",
getApiKeyForProvider: async () => "test-key",
authStorage: { rotateSessionCredential: async () => false as const },
resolver: () => async () => "test-key",
};
const primary = await resolvePrimaryModel(undefined, settings, registry);
@@ -164,15 +164,12 @@ describe.skipIf(!e2eApiKey("ANTHROPIC_API_KEY"))("Compaction hooks", () => {
expect(beforeEvent.preparation).toBeDefined();
expect(beforeEvent.preparation.messagesToSummarize).toBeDefined();
expect(beforeEvent.preparation.turnPrefixMessages).toBeDefined();
expect(beforeEvent.preparation.tokensBefore).toBeGreaterThanOrEqual(0);
expect(typeof beforeEvent.preparation.isSplitTurn).toBe("boolean");
expect(beforeEvent.branchEntries).toBeDefined();
// sessionManager, modelRegistry, and model are now on ctx, not event
const afterEvent = compactEvents[0];
expect(afterEvent.compactionEntry).toBeDefined();
expect(afterEvent.compactionEntry.summary.length).toBeGreaterThan(0);
expect(afterEvent.compactionEntry.tokensBefore).toBeGreaterThanOrEqual(0);
expect(afterEvent.fromExtension).toBe(false);
}, 120000);
@@ -285,7 +282,6 @@ describe.skipIf(!e2eApiKey("ANTHROPIC_API_KEY"))("Compaction hooks", () => {
const result = await session.compact();
expect(result.summary).toBeDefined();
expect(result.summary.length).toBeGreaterThan(0);
const compactEvents = capturedEvents.filter((e): e is SessionCompactEvent => e.type === "session_compact");
expect(compactEvents.length).toBe(1);
@@ -396,10 +392,6 @@ describe.skipIf(!e2eApiKey("ANTHROPIC_API_KEY"))("Compaction hooks", () => {
// Verify they're accessible via session
expect(typeof session.sessionManager.getEntries).toBe("function");
expect(typeof session.modelRegistry.getApiKey).toBe("function");
const entries = session.sessionManager.getEntries();
expect(Array.isArray(entries)).toBe(true);
expect(entries.length).toBeGreaterThan(0);
}, 120000);
it("should use hook compaction even with different values", async () => {
@@ -8,18 +8,10 @@
* Reproduces issue where compact fails when maxTokens < thinkingBudget.
*/
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import { afterEach, beforeEach, describe } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { Agent } from "@oh-my-pi/pi-agent-core";
import { Effort, getBundledModel, type Model, type Effort as ThinkingLevelType } from "@oh-my-pi/pi-ai";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { createTools, type ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
import { Snowflake } from "@oh-my-pi/pi-utils";
import { e2eApiKey } from "./utilities";
@@ -28,9 +20,9 @@ const HAS_ANTIGRAVITY_AUTH = false; // OAuth not available in test environment
const HAS_ANTHROPIC_AUTH = !!e2eApiKey("ANTHROPIC_API_KEY");
describe.skipIf(!HAS_ANTIGRAVITY_AUTH)("Compaction with thinking models (Antigravity)", () => {
let session: AgentSession;
let session: { dispose: () => Promise<void> } | undefined;
let tempDir: string;
let authStorage: AuthStorage | undefined;
let authStorage: { close: () => void } | undefined;
beforeEach(() => {
tempDir = path.join(os.tmpdir(), `pi-thinking-compaction-test-${Snowflake.next()}`);
@@ -47,104 +39,15 @@ describe.skipIf(!HAS_ANTIGRAVITY_AUTH)("Compaction with thinking models (Antigra
fs.rmSync(tempDir, { recursive: true });
}
});
async function createSession(
modelId: "claude-opus-4-5-thinking" | "claude-sonnet-4-5",
thinkingLevel: ThinkingLevelType = Effort.High,
) {
const toolSession: ToolSession = {
cwd: tempDir,
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => "*",
settings: Settings.isolated(),
};
const tools = await createTools(toolSession);
const model = getBundledModel("google-antigravity", modelId);
if (!model) {
throw new Error(`Model not found: google-antigravity/${modelId}`);
}
const agent = new Agent({
getApiKey: () => e2eApiKey("ANTHROPIC_API_KEY"),
initialState: {
model,
systemPrompt: ["You are a helpful assistant. Be concise."],
tools,
thinkingLevel,
},
});
const sessionManager = SessionManager.inMemory();
const settings = Settings.isolated();
authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db"));
const modelRegistry = new ModelRegistry(authStorage);
session = new AgentSession({
agent,
sessionManager,
settings,
modelRegistry,
});
session.subscribe(() => {});
return session;
}
it("should compact successfully with claude-opus-4-5-thinking and thinking level high", async () => {
await createSession("claude-opus-4-5-thinking", Effort.High);
// Send a simple prompt
await session.prompt("Write down the first 10 prime numbers.");
await session.agent.waitForIdle();
// Verify we got a response
const messages = session.messages;
expect(messages.length).toBeGreaterThan(0);
const assistantMessages = messages.filter(m => m.role === "assistant");
expect(assistantMessages.length).toBeGreaterThan(0);
// Now try to compact - this should not throw
const result = await session.compact();
expect(result.summary).toBeDefined();
expect(result.summary.length).toBeGreaterThan(0);
expect(result.tokensBefore).toBeGreaterThan(0);
// Verify session is still usable after compaction
const messagesAfterCompact = session.messages;
expect(messagesAfterCompact.length).toBeGreaterThan(0);
expect(messagesAfterCompact[0].role).toBe("compactionSummary");
}, 180000);
it("should compact successfully with claude-sonnet-4-5 (non-thinking) for comparison", async () => {
await createSession("claude-sonnet-4-5");
await session.prompt("Write down the first 10 prime numbers.");
await session.agent.waitForIdle();
const messages = session.messages;
expect(messages.length).toBeGreaterThan(0);
const result = await session.compact();
expect(result.summary).toBeDefined();
expect(result.summary.length).toBeGreaterThan(0);
}, 180000);
});
// ============================================================================
// Real Anthropic API tests (for comparison)
// ============================================================================
describe.skipIf(!HAS_ANTHROPIC_AUTH)("Compaction with thinking models (Anthropic)", () => {
let session: AgentSession;
let session: { dispose: () => Promise<void> } | undefined;
let tempDir: string;
let authStorage: AuthStorage | undefined;
let authStorage: { close: () => void } | undefined;
beforeEach(() => {
tempDir = path.join(os.tmpdir(), `pi-thinking-compaction-anthropic-test-${Snowflake.next()}`);
@@ -161,70 +64,4 @@ describe.skipIf(!HAS_ANTHROPIC_AUTH)("Compaction with thinking models (Anthropic
fs.rmSync(tempDir, { recursive: true });
}
});
async function createSession(model: Model, thinkingLevel: ThinkingLevelType = Effort.High) {
const toolSession: ToolSession = {
cwd: tempDir,
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => "*",
settings: Settings.isolated(),
};
const tools = await createTools(toolSession);
const agent = new Agent({
getApiKey: () => e2eApiKey("ANTHROPIC_API_KEY"),
initialState: {
model,
systemPrompt: ["You are a helpful assistant. Be concise."],
tools,
thinkingLevel,
},
});
const sessionManager = SessionManager.inMemory();
const settings = Settings.isolated();
authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db"));
const modelRegistry = new ModelRegistry(authStorage);
session = new AgentSession({
agent,
sessionManager,
settings,
modelRegistry,
});
session.subscribe(() => {});
return session;
}
it("should compact successfully with claude-3-7-sonnet and thinking level high", async () => {
const model = getBundledModel("anthropic", "claude-3-7-sonnet-latest")!;
await createSession(model, Effort.High);
// Send a simple prompt
await session.prompt("Write down the first 10 prime numbers.");
await session.agent.waitForIdle();
// Verify we got a response
const messages = session.messages;
expect(messages.length).toBeGreaterThan(0);
const assistantMessages = messages.filter(m => m.role === "assistant");
expect(assistantMessages.length).toBeGreaterThan(0);
// Now try to compact - this should not throw
const result = await session.compact();
expect(result.summary).toBeDefined();
expect(result.summary.length).toBeGreaterThan(0);
expect(result.tokensBefore).toBeGreaterThan(0);
// Verify session is still usable after compaction
const messagesAfterCompact = session.messages;
expect(messagesAfterCompact.length).toBeGreaterThan(0);
expect(messagesAfterCompact[0].role).toBe("compactionSummary");
}, 180000);
});
@@ -887,26 +887,6 @@ describe("Large session fixture", () => {
// ============================================================================
describe.skipIf(!e2eApiKey("ANTHROPIC_API_KEY"))("LLM summarization", () => {
it("should generate a compaction result for the large session", async () => {
const entries = await loadLargeSessionEntries();
const model = getBundledModel("anthropic", "claude-sonnet-4-5")!;
const preparation = prepareCompaction(entries, DEFAULT_COMPACTION_SETTINGS);
expect(preparation).toBeDefined();
const compactionResult = await compact(preparation!, model, e2eApiKey("ANTHROPIC_API_KEY")!);
expect(compactionResult.summary.length).toBeGreaterThan(100);
expect(compactionResult.firstKeptEntryId).toBeTruthy();
expect(compactionResult.tokensBefore).toBeGreaterThan(0);
console.log("Summary length:", compactionResult.summary.length);
console.log("First kept entry ID:", compactionResult.firstKeptEntryId);
console.log("Tokens before:", compactionResult.tokensBefore);
console.log("\n--- SUMMARY ---\n");
console.log(compactionResult.summary);
}, 60000);
it("should produce valid session after compaction", async () => {
const entries = await loadLargeSessionEntries();
const loaded = buildSessionContext(entries);
@@ -184,7 +184,7 @@ describe("hashline normalization", () => {
describe("hashline parser — range-anchor syntax", () => {
it("keeps parsed edits reusable across different target snapshots", () => {
const section = Patch.parseSingle(["a.ts", `insert after ${tag(2, "bbb")}:`, repl("tail")].join("\n"));
const section = Patch.parseSingle(["[a.ts]", `insert after ${tag(2, "bbb")}:`, repl("tail")].join("\n"));
expect(section.applyTo("aaa\nbbb").text).toBe("aaa\nbbb\ntail");
expect(section.applyTo("aaa\nbbb\nccc").text).toBe("aaa\nbbb\ntail\nccc");
@@ -546,9 +546,9 @@ describe("hashline — snapshot tag binding", () => {
});
});
describe("splitHashlineInput — @ headers", () => {
it("extracts path, snapshot tag, and diff body from @path#tag header", () => {
const input = [`src/foo.ts#1A2B`, `${sameLineRange(tag(2, "bbb"))}`, repl("BBB")].join("\n");
describe("splitHashlineInput — bracket headers", () => {
it("extracts path, snapshot tag, and diff body from [path#tag] header", () => {
const input = [`[src/foo.ts#1A2B]`, `${sameLineRange(tag(2, "bbb"))}`, repl("BBB")].join("\n");
expect(splitHashlineInput(input)).toEqual({
path: "src/foo.ts",
fileHash: "1A2B",
@@ -557,7 +557,7 @@ describe("splitHashlineInput — @ headers", () => {
});
it("strips leading blank lines", () => {
expect(splitHashlineInput(`\nfoo.ts\ninsert head:\n${repl("x")}`)).toEqual({
expect(splitHashlineInput(`\n[foo.ts]\ninsert head:\n${repl("x")}`)).toEqual({
path: "foo.ts",
diff: `insert head:\n${repl("x")}`,
});
@@ -566,7 +566,7 @@ describe("splitHashlineInput — @ headers", () => {
it("normalizes cwd-prefixed absolute paths to cwd-relative paths", () => {
const cwd = process.cwd();
const absolute = path.join(cwd, "src", "foo.ts");
expect(splitHashlineInput(`${absolute}\ninsert head:\n${repl("x")}`, { cwd }).path).toBe("src/foo.ts");
expect(splitHashlineInput(`[${absolute}]\ninsert head:\n${repl("x")}`, { cwd }).path).toBe("src/foo.ts");
});
it("uses explicit fallback path only when input has recognizable operations", () => {
@@ -578,7 +578,7 @@ describe("splitHashlineInput — @ headers", () => {
});
it("splits multiple edit sections", () => {
const input = ["a.ts", "insert head:", repl("a"), "b.ts", "insert tail:", repl("b")].join("\n");
const input = ["[a.ts]", "insert head:", repl("a"), "[b.ts]", "insert tail:", repl("b")].join("\n");
expect(splitHashlineInputs(input)).toEqual([
{ path: "a.ts", diff: `insert head:\n${repl("a")}` },
{ path: "b.ts", diff: `insert tail:\n${repl("b")}` },
@@ -595,7 +595,7 @@ describe("splitHashlineInput — @ headers", () => {
});
it("silently drops a trailing header with no operations", () => {
const input = ["a.ts", "insert head:", repl("a"), "b.ts"].join("\n");
const input = ["[a.ts]", "insert head:", repl("a"), "[b.ts]"].join("\n");
expect(splitHashlineInputs(input)).toEqual([{ path: "a.ts", diff: `insert head:\n${repl("a")}` }]);
});
});
@@ -630,7 +630,7 @@ it("preflights write policy for every section before committing a batch", async
describe("hashline executor", () => {
it("rejects file creation and directs to the write tool", async () => {
await withTempDir(async tempDir => {
const input = `new.ts\ninsert head:\n${repl("export const x = 1;")}\n`;
const input = `[new.ts]\ninsert head:\n${repl("export const x = 1;")}\n`;
await expect(executeHashlineSingle(hashlineExecuteOptions(tempDir, input))).rejects.toThrow(/write tool/);
expect(await Bun.file(path.join(tempDir, "new.ts")).exists()).toBe(false);
});
@@ -681,7 +681,7 @@ describe("hashline executor", () => {
await Bun.write(bPath, "bbb\n");
const session = makeHashlineSession(tempDir);
const aTag = recordFullSnapshot(getFileReadCache(session), aPath, "aaa\n");
const bHeader = "b.ts#FFFF";
const bHeader = "[b.ts#FFFF]";
const input = [
header("a.ts", aTag),
`${sameLineRange(tag(1, "aaa"))}`,
@@ -795,14 +795,14 @@ describe("hashlineEditParamsSchema — payload shape", () => {
it("tolerates provider extra fields without declaring `path`", () => {
expect(
hashlineEditParamsSchema.safeParse({ path: "x.ts", input: `x.ts\ninsert head:\n${repl("x")}` }).success,
hashlineEditParamsSchema.safeParse({ path: "x.ts", input: `[x.ts]\ninsert head:\n${repl("x")}` }).success,
).toBe(true);
});
it("accepts `_input` as a provider-emitted alias for `input`", () => {
const parsed = hashlineEditParamsSchema.safeParse({ _input: `x.ts\ninsert head:\n${repl("x")}` });
const parsed = hashlineEditParamsSchema.safeParse({ _input: `[x.ts]\ninsert head:\n${repl("x")}` });
expect(parsed.success).toBe(true);
if (parsed.success) expect(parsed.data.input).toBe(`x.ts\ninsert head:\n${repl("x")}`);
if (parsed.success) expect(parsed.data.input).toBe(`[x.ts]\ninsert head:\n${repl("x")}`);
});
it("still requires `input`", () => {
@@ -1088,11 +1088,11 @@ describe("hashline *** Abort recovery sentinel (harmony-leak mitigation)", () =>
it("splitter respects *** Abort like *** End Patch", () => {
const input = [
`a.ts`,
`[a.ts]`,
`insert after ${tag(1, "alpha")}:`,
repl("a-payload"),
sentinel,
`b.ts`,
`[b.ts]`,
`insert after ${tag(1, "beta")}:`,
repl("never-emitted"),
].join("\n");
@@ -1112,33 +1112,33 @@ describe("hashline *** Abort recovery sentinel (harmony-leak mitigation)", () =>
describe("hashline parser — delete and empty-block semantics", () => {
it("inline delete deletes a single line", () => {
const text = "line1\nline2\nline3\n";
const { diff } = splitHashlineInput(`a.ts\ndelete 2\n`);
const { diff } = splitHashlineInput(`[a.ts]\ndelete 2\n`);
expect(applyDiff(text, diff)).toBe("line1\nline3\n");
});
it("inline delete deletes the range", () => {
const text = "line1\nline2\nline3\nline4\n";
const { diff } = splitHashlineInput(`a.ts\ndelete 2..3\n`);
const { diff } = splitHashlineInput(`[a.ts]\ndelete 2..3\n`);
expect(applyDiff(text, diff)).toBe("line1\nline4\n");
});
it("empty replace removes the range", () => {
const text = "line1\nline2\nline3\n";
const { diff } = splitHashlineInput(`a.ts\nreplace 2..2:\n`);
const { diff } = splitHashlineInput(`[a.ts]\nreplace 2..2:\n`);
expect(applyDiff(text, diff)).toBe("line1\nline3\n");
});
it("`2..2=replacement` (old format) parses as orphan body, not as inline payload", () => {
const { diff } = splitHashlineInput(`a.ts\n2..2=replacement\n`);
const { diff } = splitHashlineInput(`[a.ts]\n2..2=replacement\n`);
expect(() => parseHashline(diff)).toThrow(/payload line has no preceding hunk header/);
});
it("explicit empty literal rows insert blank lines when the anchor is repeated", () => {
const text = "line1\nline2\nline3\n";
const aboveDiff = splitHashlineInput(`a.ts\ninsert before 2:\n${repl("")}\n`).diff;
const aboveDiff = splitHashlineInput(`[a.ts]\ninsert before 2:\n${repl("")}\n`).diff;
expect(applyDiff(text, aboveDiff)).toBe("line1\n\nline2\nline3\n");
const belowDiff = splitHashlineInput(`a.ts\ninsert after 2:\n${repl("")}\n`).diff;
const belowDiff = splitHashlineInput(`[a.ts]\ninsert after 2:\n${repl("")}\n`).diff;
expect(applyDiff(text, belowDiff)).toBe("line1\nline2\n\nline3\n");
});
});
@@ -1146,28 +1146,28 @@ describe("hashline parser — delete and empty-block semantics", () => {
describe("hashline parser — explicit blank payload rows", () => {
it("raw blank lines between ops are ignored", () => {
const text = "a\nb\nc\nd\ne\n";
const ops = `a.ts\nreplace 1..1:\n${repl("A")}\n\nreplace 3..3:\n${repl("C")}\n`;
const ops = `[a.ts]\nreplace 1..1:\n${repl("A")}\n\nreplace 3..3:\n${repl("C")}\n`;
const { diff } = splitHashlineInput(ops);
expect(applyDiff(text, diff)).toBe("A\nb\nC\nd\ne\n");
});
it("empty replacement payload rows are appended as blank payload lines", () => {
const text = "a\nb\nc\nd\ne\n";
const ops = `a.ts\nreplace 1..1:\n${repl("A")}\n${repl("")}\n${repl("")}\nreplace 3..3:\n${repl("C")}\n`;
const ops = `[a.ts]\nreplace 1..1:\n${repl("A")}\n${repl("")}\n${repl("")}\nreplace 3..3:\n${repl("C")}\n`;
const { diff } = splitHashlineInput(ops);
expect(applyDiff(text, diff)).toBe("A\n\n\nb\nC\nd\ne\n");
});
it("`replace N..N:` followed by two empty replace rows replaces the line with two blanks", () => {
const text = "a\nb\nc\nd\ne\n";
const ops = `a.ts\nreplace 2..2:\n${repl("")}\n${repl("")}\nreplace 4..4:\n${repl("D")}\n`;
const ops = `[a.ts]\nreplace 2..2:\n${repl("")}\n${repl("")}\nreplace 4..4:\n${repl("D")}\n`;
const { diff } = splitHashlineInput(ops);
expect(applyDiff(text, diff)).toBe("a\n\n\nc\nD\ne\n");
});
it("empty replace row inside payload between two content lines is preserved", () => {
const text = "a\nb\nc\n";
const ops = `a.ts\nreplace 2..2:\n${repl("first")}\n${repl("")}\n${repl("second")}\n`;
const ops = `[a.ts]\nreplace 2..2:\n${repl("first")}\n${repl("")}\n${repl("second")}\n`;
const { diff } = splitHashlineInput(ops);
expect(applyDiff(text, diff)).toBe("a\nfirst\n\nsecond\nc\n");
});
@@ -69,13 +69,11 @@ describe("executePython session lifecycle", () => {
return kernel as unknown as PythonKernel;
};
const first = await executePython("print('one')", { sessionId: "session-1" });
const second = await executePython("print('two')", { sessionId: "session-1" });
await executePython("print('one')", { sessionId: "session-1" });
await executePython("print('two')", { sessionId: "session-1" });
expect(startCount).toBe(1);
expect(kernel.executeCalls).toEqual(["print('one')", "print('two')"]);
expect(first.output).toContain("ok");
expect(second.output).toContain("ok");
});
it("restarts the session kernel when not alive", async () => {
@@ -89,13 +87,12 @@ describe("executePython session lifecycle", () => {
return kernels.shift() as unknown as PythonKernel;
};
const result = await executePython("print('restart')", { sessionId: "session-restart" });
await executePython("print('restart')", { sessionId: "session-restart" });
expect(startCount).toBe(2);
expect(deadKernel.shutdownCalls).toBe(1);
expect(deadKernel.executeCalls).toEqual([]);
expect(liveKernel.executeCalls).toEqual(["print('restart')"]);
expect(result.output).toContain("live");
});
it("resets the session kernel when requested", async () => {
@@ -1,6 +1,6 @@
import { describe, expect, it, vi } from "bun:test";
import { defaultEditorTheme } from "../../tui/test/test-themes";
import { CustomEditor } from "../src/modes/components/custom-editor";
import { CustomEditor, extractBracketedImagePastePath } from "../src/modes/components/custom-editor";
function ctrl(key: string): string {
return String.fromCharCode(key.toLowerCase().charCodeAt(0) & 31);
@@ -20,6 +20,23 @@ describe("CustomEditor literal question mark input", () => {
});
});
describe("CustomEditor bracketed image path paste", () => {
it("routes a single pasted image path to the image-path handler", () => {
const editor = createEditor();
const paths: string[] = [];
editor.onPasteImagePath = path => paths.push(path);
editor.handleInput("\x1b[200~/tmp/screenshot.png\x1b[201~");
expect(paths).toEqual(["/tmp/screenshot.png"]);
expect(editor.getText()).toBe("");
});
it("leaves ordinary bracketed paste text on the editor path", () => {
expect(extractBracketedImagePastePath("\x1b[200~not an image.txt\x1b[201~")).toBeUndefined();
});
});
describe("CustomEditor temporary model selector keybinding", () => {
it("triggers the temporary selector from a remapped action key instead of Alt+P", () => {
const editor = createEditor();
@@ -48,6 +65,39 @@ describe("CustomEditor temporary model selector keybinding", () => {
});
});
describe("CustomEditor model selector and display reset keybindings", () => {
it("uses Alt+M for the model selector and Ctrl+L for display reset by default", () => {
const editor = createEditor();
const onSelectModel = vi.fn();
const onDisplayReset = vi.fn();
editor.onSelectModel = onSelectModel;
editor.onDisplayReset = onDisplayReset;
editor.handleInput("\x1bm");
expect(onSelectModel).toHaveBeenCalledTimes(1);
expect(onDisplayReset).not.toHaveBeenCalled();
editor.handleInput(ctrl("l"));
expect(onSelectModel).toHaveBeenCalledTimes(1);
expect(onDisplayReset).toHaveBeenCalledTimes(1);
});
it("lets display reset win when an old model remap also uses Ctrl+L", () => {
const editor = createEditor();
const onSelectModel = vi.fn();
const onDisplayReset = vi.fn();
editor.onSelectModel = onSelectModel;
editor.onDisplayReset = onDisplayReset;
editor.setActionKeys("app.model.select", ["ctrl+l"]);
editor.setActionKeys("app.display.reset", ["ctrl+l"]);
editor.handleInput(ctrl("l"));
expect(onDisplayReset).toHaveBeenCalledTimes(1);
expect(onSelectModel).not.toHaveBeenCalled();
});
});
describe("CustomEditor escape key dispatch", () => {
function installAutocompleteProvider(editor: CustomEditor) {
editor.setAutocompleteProvider({
@@ -3,6 +3,7 @@ import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { Settings } from "../../src/config/settings";
import * as dapModule from "../../src/dap";
import { DapClient } from "../../src/dap/client";
import { DapSessionManager } from "../../src/dap/session";
import type { DapCapabilities, DapClientState, DapEventMessage, DapResolvedAdapter } from "../../src/dap/types";
@@ -20,8 +21,35 @@ const TEST_ADAPTER: DapResolvedAdapter = {
launchDefaults: {},
attachDefaults: {},
connectMode: "stdio",
acceptsDirectoryProgram: false,
};
const DELAYED_UNIX_SOCKET_ADAPTER = `
const listenPrefix = "--listen=unix:";
const listenArg = process.argv.find(arg => arg.startsWith(listenPrefix));
if (!listenArg) {
throw new Error("missing --listen=unix argument");
}
const socketPath = listenArg.slice(listenPrefix.length);
let server;
process.on("SIGTERM", () => {
server?.stop();
process.exit(0);
});
await Bun.sleep(100);
server = Bun.listen({
unix: socketPath,
socket: {
open() {},
data() {},
close() {},
error() {},
},
});
await Bun.sleep(2_000);
server.stop();
`;
type DapEventHandler = (body: unknown, event: DapEventMessage) => void | Promise<void>;
class FakeDapClient {
@@ -286,32 +314,179 @@ describe("DAP launch failure handling", () => {
expect(message).toContain("launch: 'C:\\repo\\program' is not a valid executable");
expect(message).toContain("configurationDone: Expected process to be stopped.");
});
it("waits for delayed Unix socket adapters before connecting on Linux", async () => {
if (process.platform !== "linux") return;
const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-dlv-socket-"));
const adapterPath = path.join(cwd, "delayed-unix-socket-adapter.mjs");
await fs.writeFile(adapterPath, DELAYED_UNIX_SOCKET_ADAPTER);
const adapter: DapResolvedAdapter = {
...TEST_ADAPTER,
name: "dlv",
command: process.execPath,
args: [adapterPath],
resolvedCommand: process.execPath,
connectMode: "socket",
};
let client: DapClient | undefined;
try {
client = await DapClient.spawn({ adapter, cwd });
expect(client.isAlive()).toBe(true);
} finally {
await client?.dispose();
await fs.rm(cwd, { recursive: true, force: true });
}
});
});
describe("DebugTool launch validation", () => {
it("rejects directory-valued launch programs before adapter selection", async () => {
const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-program-"));
it("rejects directory programs when the selected adapter cannot debug a directory", async () => {
const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue(TEST_ADAPTER);
try {
await fs.mkdir(path.join(cwd, "python"));
const session: ToolSession = {
cwd,
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => "*",
settings: Settings.isolated({ "debug.enabled": true }),
};
const tool = new DebugTool(session);
const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-program-"));
try {
await fs.mkdir(path.join(cwd, "python"));
const session: ToolSession = {
cwd,
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => "*",
settings: Settings.isolated({ "debug.enabled": true }),
};
const tool = new DebugTool(session);
await expect(tool.execute("call", { action: "launch", program: "python" })).rejects.toThrow(
/launch program resolves to a directory.*python/,
);
await expect(tool.execute("call", { action: "launch", program: "python" })).rejects.toThrow(
/launch program resolves to a directory.*python/,
);
} finally {
await fs.rm(cwd, { recursive: true, force: true });
}
} finally {
await fs.rm(cwd, { recursive: true, force: true });
launchSpy.mockRestore();
}
});
it("allows directory programs when the selected adapter accepts them (dlv on a Go package)", async () => {
const dlvAdapter: DapResolvedAdapter = {
...TEST_ADAPTER,
name: "dlv",
command: "dlv",
resolvedCommand: "dlv",
launchDefaults: { request: "launch", mode: "debug", stopOnEntry: true },
acceptsDirectoryProgram: true,
};
const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue(dlvAdapter);
const sessionLaunchSpy = spyOn(dapModule.dapSessionManager, "launch").mockImplementation(async opts => {
throw Object.assign(new Error("captured launch"), { capturedOptions: opts });
});
try {
const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-dlv-dir-"));
try {
await fs.mkdir(path.join(cwd, "cmd"));
const session: ToolSession = {
cwd,
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => "*",
settings: Settings.isolated({ "debug.enabled": true }),
};
const tool = new DebugTool(session);
// Validation must pass and propagate to dapSessionManager.launch; we
// stop the actual spawn there and inspect the launch arguments.
await expect(tool.execute("call", { action: "launch", program: "cmd", adapter: "dlv" })).rejects.toThrow(
/captured launch/,
);
expect(sessionLaunchSpy).toHaveBeenCalledTimes(1);
const [opts] = sessionLaunchSpy.mock.calls[0]!;
expect(opts.extraLaunchArguments).toEqual({ mode: "debug" });
expect(opts.program).toBe(path.join(cwd, "cmd"));
} finally {
await fs.rm(cwd, { recursive: true, force: true });
}
} finally {
sessionLaunchSpy.mockRestore();
launchSpy.mockRestore();
}
});
it("prefers directory-capable dlv over native adapters for extensionless Go package directories", async () => {
const sessionLaunchSpy = spyOn(dapModule.dapSessionManager, "launch").mockImplementation(async opts => {
throw Object.assign(new Error("captured launch"), { capturedOptions: opts });
});
try {
const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-dlv-mixed-roots-"));
try {
await fs.writeFile(path.join(cwd, "go.mod"), "module hello\n\ngo 1.22\n");
await fs.writeFile(path.join(cwd, "Makefile"), "all:\n\tgo build ./...\n");
await fs.mkdir(path.join(cwd, "bin"));
await fs.writeFile(path.join(cwd, "bin", "dlv"), "");
await fs.writeFile(path.join(cwd, "bin", "gdb"), "");
await fs.mkdir(path.join(cwd, "cmd", "hello"), { recursive: true });
const session: ToolSession = {
cwd,
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => "*",
settings: Settings.isolated({ "debug.enabled": true }),
};
const tool = new DebugTool(session);
await expect(tool.execute("call", { action: "launch", program: "cmd/hello" })).rejects.toThrow(
/captured launch/,
);
const [opts] = sessionLaunchSpy.mock.calls[0]!;
expect(opts.adapter.name).toBe("dlv");
expect(opts.extraLaunchArguments).toEqual({ mode: "debug" });
} finally {
await fs.rm(cwd, { recursive: true, force: true });
}
} finally {
sessionLaunchSpy.mockRestore();
}
});
it("dlv launch with a compiled binary switches mode from debug to exec", async () => {
const dlvAdapter: DapResolvedAdapter = {
...TEST_ADAPTER,
name: "dlv",
command: "dlv",
resolvedCommand: "dlv",
launchDefaults: { request: "launch", mode: "debug", stopOnEntry: true },
acceptsDirectoryProgram: true,
};
const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue(dlvAdapter);
const sessionLaunchSpy = spyOn(dapModule.dapSessionManager, "launch").mockImplementation(async opts => {
throw Object.assign(new Error("captured launch"), { capturedOptions: opts });
});
try {
const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-dlv-exec-"));
try {
await fs.writeFile(path.join(cwd, "hello"), "#!/usr/bin/env sh\necho hi\n");
const session: ToolSession = {
cwd,
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => "*",
settings: Settings.isolated({ "debug.enabled": true }),
};
const tool = new DebugTool(session);
await expect(tool.execute("call", { action: "launch", program: "hello", adapter: "dlv" })).rejects.toThrow(
/captured launch/,
);
const [opts] = sessionLaunchSpy.mock.calls[0]!;
expect(opts.extraLaunchArguments).toEqual({ mode: "exec" });
} finally {
await fs.rm(cwd, { recursive: true, force: true });
}
} finally {
sessionLaunchSpy.mockRestore();
launchSpy.mockRestore();
}
});
it("throws targeted 'python not found in PATH' when adapter:'debugpy' is unresolvable for launch", async () => {
const dapModule = await import("../../src/dap");
const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue(null);
try {
const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-debugpy-"));
@@ -338,7 +513,6 @@ describe("DebugTool launch validation", () => {
});
it("throws targeted 'python not found in PATH' when adapter:'debugpy' is unresolvable for attach", async () => {
const dapModule = await import("../../src/dap");
const attachSpy = spyOn(dapModule, "selectAttachAdapter").mockReturnValue(null);
try {
const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-debugpy-attach-"));
@@ -364,7 +538,6 @@ describe("DebugTool launch validation", () => {
});
it("falls back to the generic 'No debugger adapter' error when adapter is unspecified", async () => {
const dapModule = await import("../../src/dap");
const launchSpy = spyOn(dapModule, "selectLaunchAdapter").mockReturnValue(null);
try {
const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-noadapter-"));
+2 -2
View File
@@ -276,7 +276,7 @@ describe("computeHashlineDiff", () => {
// preview/diff path MUST emit the SAME rejection so a successful preview
// never precedes a failing apply.
const result = await computeHashlineDiff(
{ input: `${relativePath}\ninsert tail:\n+second` },
{ input: `[${relativePath}]\ninsert tail:\n+second` },
tempDir,
new InMemorySnapshotStore(),
);
@@ -287,7 +287,7 @@ describe("computeHashlineDiff", () => {
});
test("returns a handled error when the source path is a local URL", async () => {
const result = await computeHashlineDiff(
{ input: "local://PLAN.md\ninsert tail:\n+x" },
{ input: "[local://PLAN.md]\ninsert tail:\n+x" },
tempDir,
new InMemorySnapshotStore(),
);
@@ -136,7 +136,7 @@ describe("hashline streaming preview (single-op trailing payload)", () => {
});
test("does not surface stale hash errors while streaming", async () => {
const input = "a.ts#FFFF\nreplace 2..2:\n+const b = 22";
const input = "[a.ts#FFFF]\nreplace 2..2:\n+const b = 22";
const previews = await strategy.computeDiffPreview({ input } as never, ctx(tmpDir) as never);
expect(previews).toHaveLength(1);
expect(previews?.[0]?.error).toBeUndefined();
@@ -170,7 +170,7 @@ describe("hashline streaming preview (single-op trailing payload)", () => {
});
test("surfaces stale hash errors once streaming is complete", async () => {
const input = "a.ts#FFFF\nreplace 2..2:\n+const b = 22\n";
const input = "[a.ts#FFFF]\nreplace 2..2:\n+const b = 22\n";
const previews = await strategy.computeDiffPreview({ input } as never, ctx(tmpDir, false) as never);
expect(previews).toHaveLength(1);
expect(previews?.[0]?.error).toContain("not from this session");
@@ -245,12 +245,12 @@ describe("apply_patch streaming preview (trailing partial line)", () => {
describe("matcherDigest", () => {
test("hashline: digests stripped `+` body rows only, never headers or op lines", () => {
const input = ["a.ts#AB12", "replace 1..2:", "+const x = 1;", "+const y = 2;", "delete 5", ""].join("\n");
const input = ["[a.ts#AB12]", "replace 1..2:", "+const x = 1;", "+const y = 2;", "delete 5", ""].join("\n");
expect(EDIT_MODE_STRATEGIES.hashline.matcherDigest({ input })).toBe("const x = 1;\nconst y = 2;");
});
test("hashline: grammar-only payload digests to empty, missing input to undefined", () => {
expect(EDIT_MODE_STRATEGIES.hashline.matcherDigest({ input: "a.ts#AB12\ndelete 3\n" })).toBe("");
expect(EDIT_MODE_STRATEGIES.hashline.matcherDigest({ input: "[a.ts#AB12]\ndelete 3\n" })).toBe("");
expect(EDIT_MODE_STRATEGIES.hashline.matcherDigest({})).toBeUndefined();
});
@@ -7,9 +7,11 @@
* C1 errorMessage = SILENT_ABORT_MARKER + aborted
* → `updateContent` receives a message with `stopReason: "stop"`;
* `errorMessage` is NOT overwritten.
* C2 errorMessage = undefined + aborted + no TTSR flag
* → `streamingMessage.errorMessage` is set to "Operation aborted";
* `updateContent` receives the original message ref.
* C2 errorMessage = undefined (no threaded reason) + aborted + no TTSR flag
* → `streamingMessage.errorMessage` is set to the generic "Operation
* aborted"; `updateContent` receives the original message ref.
* C2b errorMessage = USER_INTERRUPT_LABEL (threaded via AbortController) + aborted
* → the threaded reason is preserved verbatim, NOT replaced by the generic.
* C3 isTtsrAbortPending = true + aborted
* → `updateContent` receives a message with `stopReason: "stop"`;
* `errorMessage` is NOT set (TTSR existing behavior unchanged).
@@ -19,7 +21,7 @@ import type { AssistantMessage } from "@oh-my-pi/pi-ai";
import { EventController } from "@oh-my-pi/pi-coding-agent/modes/controllers/event-controller";
import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types";
import type { AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { SILENT_ABORT_MARKER } from "@oh-my-pi/pi-coding-agent/session/messages";
import { SILENT_ABORT_MARKER, USER_INTERRUPT_LABEL } from "@oh-my-pi/pi-coding-agent/session/messages";
function makeAssistantMessage(overrides: Partial<AssistantMessage> = {}): AssistantMessage {
return {
@@ -103,7 +105,7 @@ describe("EventController #handleMessageEnd abort labeling", () => {
expect(ctx.streamingMessage).toBeUndefined();
});
it("C2: errorMessage undefined + aborted + no TTSR -> errorMessage='Operation aborted', updateContent receives original ref", async () => {
it("C2: errorMessage undefined (no threaded reason) + aborted + no TTSR -> errorMessage='Operation aborted', updateContent receives original ref", async () => {
const message = makeAssistantMessage({ stopReason: "aborted", errorMessage: undefined });
const { controller, streamingComponent } = createFixture({
streamingMessage: message,
@@ -112,7 +114,7 @@ describe("EventController #handleMessageEnd abort labeling", () => {
await controller.handleEvent({ type: "message_end", message });
// Operator-facing label was stamped in-place on the streaming message ref.
// No threaded reason -> generic operator-facing label stamped in-place.
expect(message.errorMessage).toBe("Operation aborted");
// `updateContent` saw the original streaming message ref (no `{...streamingMessage, stopReason:"stop"}` spread).
@@ -123,6 +125,23 @@ describe("EventController #handleMessageEnd abort labeling", () => {
expect(arg.errorMessage).toBe("Operation aborted");
});
it("C2b: threaded user-interrupt reason on aborted message is preserved, not replaced by the generic label", async () => {
const message = makeAssistantMessage({ stopReason: "aborted", errorMessage: USER_INTERRUPT_LABEL });
const { controller, streamingComponent } = createFixture({
streamingMessage: message,
isTtsrAbortPending: false,
});
await controller.handleEvent({ type: "message_end", message });
// The Esc-interrupt reason rode the AbortController onto errorMessage; the
// controller must surface it verbatim instead of overwriting with "Operation aborted".
expect(message.errorMessage).toBe(USER_INTERRUPT_LABEL);
const arg = streamingComponent.updateContent.mock.calls[0]![0] as AssistantMessage;
expect(arg.errorMessage).toBe(USER_INTERRUPT_LABEL);
expect(arg.stopReason).toBe("aborted");
});
it("C3: isTtsrAbortPending=true + aborted -> updateContent stopReason='stop', errorMessage NOT set", async () => {
const message = makeAssistantMessage({ stopReason: "aborted", errorMessage: undefined });
const { controller, streamingComponent } = createFixture({
@@ -7,8 +7,10 @@
* `agent_start` via `ctx.clearPinnedError`. Aborts and normal stops must NOT
* pin a banner.
*/
import { beforeAll, describe, expect, it, vi } from "bun:test";
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { AssistantMessageComponent } from "@oh-my-pi/pi-coding-agent/modes/components/assistant-message";
import { ErrorBannerComponent } from "@oh-my-pi/pi-coding-agent/modes/components/error-banner";
import { EventController } from "@oh-my-pi/pi-coding-agent/modes/controllers/event-controller";
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
@@ -40,12 +42,22 @@ beforeAll(async () => {
await initTheme(false);
});
beforeEach(async () => {
resetSettingsForTest();
await Settings.init({ inMemory: true });
});
afterEach(() => {
resetSettingsForTest();
});
function createFixture(streamingMessage?: AssistantMessage) {
const streamingComponent = {
updateContent: vi.fn(),
setUsageInfo: vi.fn(),
setComplete: vi.fn(),
markTranscriptBlockFinalized: vi.fn(),
setErrorPinned: vi.fn(),
};
const showPinnedError = vi.fn();
const clearPinnedError = vi.fn();
@@ -67,14 +79,14 @@ function createFixture(streamingMessage?: AssistantMessage) {
} as unknown as InteractiveModeContext;
const controller = new EventController(ctx);
return { controller, ctx, showPinnedError, clearPinnedError };
return { controller, ctx, showPinnedError, clearPinnedError, streamingComponent };
}
describe("EventController error banner", () => {
it("pins the provider error above the editor when an assistant turn ends on stopReason error", async () => {
const errorMessage = "Output blocked by content filtering policy";
const message = makeAssistantMessage({ stopReason: "error", errorMessage });
const { controller, showPinnedError } = createFixture(message);
const { controller, showPinnedError, streamingComponent } = createFixture(message);
await controller.handleEvent({ type: "message_end", message } as Extract<
AgentSessionEvent,
@@ -83,6 +95,26 @@ describe("EventController error banner", () => {
expect(showPinnedError).toHaveBeenCalledTimes(1);
expect(showPinnedError).toHaveBeenCalledWith(errorMessage);
// The same error is mirrored in the banner, so the transcript's inline
// `Error: …` line is suppressed to avoid a duplicate render.
expect(streamingComponent.setErrorPinned).toHaveBeenCalledWith(true);
});
it("restores the transcript inline error when the next turn starts", async () => {
const errorMessage = "Output blocked by content filtering policy";
const message = makeAssistantMessage({ stopReason: "error", errorMessage });
const { controller, clearPinnedError, streamingComponent } = createFixture(message);
await controller.handleEvent({ type: "message_end", message } as Extract<
AgentSessionEvent,
{ type: "message_end" }
>);
streamingComponent.setErrorPinned.mockClear();
await controller.handleEvent({ type: "agent_start" } as Extract<AgentSessionEvent, { type: "agent_start" }>);
expect(clearPinnedError).toHaveBeenCalledTimes(1);
expect(streamingComponent.setErrorPinned).toHaveBeenCalledWith(false);
});
it("does not pin a banner for a normal assistant stop", async () => {
@@ -135,3 +167,22 @@ describe("ErrorBannerComponent", () => {
expect(detailLines.length).toBeGreaterThan(0);
});
});
describe("AssistantMessageComponent error pinning", () => {
it("hides the inline error while pinned and restores it afterwards", () => {
const message = makeAssistantMessage({
content: [],
stopReason: "error",
errorMessage: "400 invalid reasoning value",
});
const component = new AssistantMessageComponent(message);
expect(Bun.stripANSI(component.render(120).join("\n"))).toContain("Error: 400 invalid reasoning value");
component.setErrorPinned(true);
expect(Bun.stripANSI(component.render(120).join("\n"))).not.toContain("Error: 400 invalid reasoning value");
component.setErrorPinned(false);
expect(Bun.stripANSI(component.render(120).join("\n"))).toContain("Error: 400 invalid reasoning value");
});
});
@@ -31,11 +31,4 @@ describe("HTML export template script inlining", () => {
expect(script).not.toMatch(/<\/body>/i);
expect(script).not.toMatch(/<\/html>/i);
});
it("produces a syntactically valid inlined script", () => {
const script = extractScript();
// `new Function(body)` parses without executing. Throws SyntaxError on
// the spliced-tag corruption the substitution-pattern bug produces.
expect(() => new Function(script)).not.toThrow();
});
});
@@ -2,7 +2,7 @@
* Tests for ExtensionRunner - conflict detection, error handling, tool wrapping.
*/
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as path from "node:path";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
@@ -20,21 +20,34 @@ describe("ExtensionRunner", () => {
let tempDir: TempDir;
let extensionsDir: string;
let sessionManager: SessionManager;
// Shared immutable fixtures. ModelRegistry's constructor synchronously loads
// every bundled model and rebuilds the canonical index (~100ms); these tests
// never mutate the registry or auth storage, so build them once per file
// instead of paying that cost in every beforeEach.
let sharedTempDir: TempDir;
let modelRegistry: ModelRegistry;
let authStorage: AuthStorage;
beforeEach(async () => {
beforeAll(async () => {
sharedTempDir = TempDir.createSync("@pi-runner-shared-");
authStorage = await AuthStorage.create(path.join(sharedTempDir.path(), "testauth.db"));
modelRegistry = new ModelRegistry(authStorage);
});
afterAll(() => {
authStorage.close();
sharedTempDir.removeSync();
});
beforeEach(() => {
tempDir = TempDir.createSync("@pi-runner-test-");
extensionsDir = path.join(getProjectAgentDir(tempDir.path()), "extensions");
fs.mkdirSync(extensionsDir, { recursive: true });
sessionManager = SessionManager.inMemory();
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
modelRegistry = new ModelRegistry(authStorage);
});
afterEach(() => {
testSetExtensionHandlerTimeoutMs(EXTENSION_HANDLER_TIMEOUT_MS);
authStorage.close();
tempDir.removeSync();
});
@@ -86,6 +99,68 @@ describe("ExtensionRunner", () => {
warnSpy.mockRestore();
});
it("rejects ctrl+q so it cannot shadow the app.message.followUp default (#1903)", async () => {
const extCode = `
export default function(pi) {
pi.registerShortcut("ctrl+q", {
description: "Tries to bind the follow-up chord",
handler: async () => {},
});
}
`;
fs.writeFileSync(path.join(extensionsDir, "conflict-q.ts"), extCode);
const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {});
const result = await loadTestExtensions();
const runner = new ExtensionRunner(
result.extensions,
result.runtime,
tempDir.path(),
sessionManager,
modelRegistry,
);
const shortcuts = runner.getShortcuts();
// Contract: ctrl+q is reserved because it is now a default chord for
// app.message.followUp. Without this guard, InputController registers
// the extension shortcut first and the follow-up handler silently
// overwrites it in the editor's custom-key map.
expect(warnSpy).toHaveBeenCalledWith(expect.stringContaining("conflicts with built-in"), expect.any(Object));
expect(shortcuts.has("ctrl+q")).toBe(false);
warnSpy.mockRestore();
});
it("rejects Alt+M so it cannot shadow the app.model.select default", async () => {
const extCode = `
export default function(pi) {
pi.registerShortcut("alt+m", {
description: "Tries to bind model select",
handler: async () => {},
});
}
`;
fs.writeFileSync(path.join(extensionsDir, "conflict-model.ts"), extCode);
const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {});
const result = await loadTestExtensions();
const runner = new ExtensionRunner(
result.extensions,
result.runtime,
tempDir.path(),
sessionManager,
modelRegistry,
);
const shortcuts = runner.getShortcuts();
expect(warnSpy).toHaveBeenCalledWith(expect.stringContaining("conflicts with built-in"), expect.any(Object));
expect(shortcuts.has("alt+m")).toBe(false);
warnSpy.mockRestore();
});
it("warns when two extensions register same shortcut", async () => {
// Use a non-reserved shortcut
const extCode1 = `
@@ -0,0 +1,79 @@
import { beforeAll, describe, expect, it } from "bun:test";
import { GALLERY_STATES, renderGalleryState, resolveFixture } from "../src/cli/gallery-cli";
import type { GalleryFixture } from "../src/cli/gallery-fixtures";
import { resetSettingsForTest, Settings } from "../src/config/settings";
import { initTheme } from "../src/modes/theme/theme";
import { toolRenderers } from "../src/tools/renderers";
beforeAll(async () => {
resetSettingsForTest();
await Settings.init({ inMemory: true });
await initTheme(false, undefined, undefined, "dark", "light");
});
describe("gallery harness", () => {
it("renders every registered tool in every lifecycle state without throwing", async () => {
for (const name in toolRenderers) {
const fixture = resolveFixture(name);
for (const state of GALLERY_STATES) {
const lines = await renderGalleryState(name, fixture, state, 100);
// A renderer that produces no lines for a state is a regression: the
// component should always emit at least the call header or result.
expect(lines.length, `${name}/${state} rendered nothing`).toBeGreaterThan(0);
}
}
});
it("routes each state to the matching args/result (streaming args vs result, success vs error)", async () => {
const fixture: GalleryFixture = {
label: "Bash",
streamingArgs: { command: "echo STREAM_MARK" },
args: { command: "echo PROGRESS_MARK" },
result: { content: [{ type: "text", text: "SUCCESS_OUT" }], details: { exitCode: 0 } },
errorResult: { content: [{ type: "text", text: "ERROR_OUT" }], isError: true, details: { exitCode: 1 } },
};
const render = async (state: (typeof GALLERY_STATES)[number]) =>
Bun.stripANSI((await renderGalleryState("bash", fixture, state, 100)).join("\n"));
const streaming = await render("streaming");
expect(streaming).toContain("STREAM_MARK");
expect(streaming).not.toContain("PROGRESS_MARK");
expect(streaming).not.toContain("SUCCESS_OUT");
const progress = await render("progress");
expect(progress).toContain("PROGRESS_MARK");
expect(progress).not.toContain("SUCCESS_OUT");
const success = await render("success");
expect(success).toContain("SUCCESS_OUT");
expect(success).not.toContain("ERROR_OUT");
const error = await render("error");
expect(error).toContain("ERROR_OUT");
expect(error).not.toContain("SUCCESS_OUT");
});
it("routes customRendered tools (lsp, task) through the custom-tool branch", async () => {
// `lsp`/`task` attach their renderers on the real AgentTool, so the gallery
// must reproduce that path. With a result present and mergeCallAndResult, the
// custom branch must NOT emit a redundant tool-name line above the result box
// (regression guard for tool-execution's custom-branch fallback label).
const lsp = resolveFixture("lsp");
expect(lsp.customRendered).toBe(true);
const lines = await renderGalleryState("lsp", lsp, "error", 100);
const stripped = lines.map(line => Bun.stripANSI(line).trim());
// The framed result header is present...
expect(stripped.some(line => line.includes("LSP references"))).toBe(true);
// ...but no standalone "LSP" label line precedes it.
expect(stripped).not.toContain("LSP");
});
it("falls back to a generic fixture for registry tools without curated sample data", () => {
// resolveFixture never returns undefined for a registry tool, even one
// missing from the curated fixtures, so the gallery cannot crash on a newly
// added renderer.
const fixture = resolveFixture("a-tool-that-has-no-fixture");
expect(fixture.args).toBeDefined();
expect(fixture.result.content.length).toBeGreaterThan(0);
});
});
@@ -1,4 +1,4 @@
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { Agent } from "@oh-my-pi/pi-agent-core";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
@@ -25,7 +25,6 @@ function createToolSession(cwd: string, settings: Settings, overrides: Partial<T
type GoalHarness = {
tempDir: TempDir;
authStorage: AuthStorage;
settings: Settings;
session: AgentSession;
mode: InteractiveMode;
@@ -33,16 +32,35 @@ type GoalHarness = {
cleanup: () => Promise<void>;
};
async function createGoalHarness(): Promise<GoalHarness> {
resetSettingsForTest();
const tempDir = TempDir.createSync("@pi-goal-mode-");
await Settings.init({ inMemory: true, cwd: tempDir.path() });
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
// Immutable, expensive fixtures shared across every test. `new ModelRegistry`
// alone is ~110ms (loads + parses the bundled model catalog), which dominated
// this file's wall time when rebuilt per test. The registry, its auth storage,
// and the resolved model are never mutated by goal-mode flows, and
// AgentSession.dispose() never closes authStorage — so a single shared instance
// is safe and drops ~8×110ms of pure setup overhead.
type SharedFixture = {
authStorage: AuthStorage;
modelRegistry: ModelRegistry;
model: NonNullable<ReturnType<ModelRegistry["find"]>>;
baseDir: TempDir;
};
async function createSharedFixture(): Promise<SharedFixture> {
const baseDir = TempDir.createSync("@pi-goal-mode-shared-");
const authStorage = await AuthStorage.create(path.join(baseDir.path(), "testauth.db"));
const modelRegistry = new ModelRegistry(authStorage);
const model = modelRegistry.find("anthropic", "claude-sonnet-4-5");
if (!model) {
throw new Error("Expected claude-sonnet-4-5 to exist in registry");
}
return { authStorage, modelRegistry, model, baseDir };
}
async function createGoalHarness(shared: SharedFixture): Promise<GoalHarness> {
resetSettingsForTest();
const tempDir = TempDir.createSync("@pi-goal-mode-");
await Settings.init({ inMemory: true, cwd: tempDir.path() });
const { modelRegistry, model } = shared;
const settings = Settings.isolated({
"compaction.enabled": false,
@@ -77,7 +95,6 @@ async function createGoalHarness(): Promise<GoalHarness> {
return {
tempDir,
authStorage,
settings,
session,
mode,
@@ -85,7 +102,6 @@ async function createGoalHarness(): Promise<GoalHarness> {
cleanup: async () => {
mode.stop();
await session.dispose();
authStorage.close();
tempDir.removeSync();
resetSettingsForTest();
},
@@ -98,13 +114,20 @@ async function toolNamesFor(harness: GoalHarness): Promise<string[]> {
describe("InteractiveMode goal mode integration", () => {
let harness: GoalHarness;
let shared: SharedFixture;
beforeAll(() => {
beforeAll(async () => {
initTheme();
shared = await createSharedFixture();
});
afterAll(() => {
shared.authStorage.close();
shared.baseDir.removeSync();
});
beforeEach(async () => {
harness = await createGoalHarness();
harness = await createGoalHarness(shared);
});
afterEach(async () => {
@@ -1,6 +1,7 @@
import { describe, expect, it, type Mock, vi } from "bun:test";
import { InputController } from "@oh-my-pi/pi-coding-agent/modes/controllers/input-controller";
import type { InteractiveModeContext, SubmittedUserInput } from "@oh-my-pi/pi-coding-agent/modes/types";
import { USER_INTERRUPT_LABEL } from "@oh-my-pi/pi-coding-agent/session/messages";
type Spy = Mock<(...args: unknown[]) => unknown>;
type StartPendingSubmissionSpy = Mock<InteractiveModeContext["startPendingSubmission"]>;
@@ -34,10 +35,12 @@ type FakeEditor = {
function createSubmission(input: {
text: string;
images?: InteractiveModeContext["pendingImages"];
imageLinks?: InteractiveModeContext["pendingImageLinks"];
}): SubmittedUserInput {
return {
text: input.text,
images: input.images,
imageLinks: input.imageLinks,
cancelled: false,
started: false,
};
@@ -80,10 +83,16 @@ function createContext(): {
const hasActiveBtw = vi.fn(() => false);
const handleOmfgEscape = vi.fn(() => true);
const hasActiveOmfg = vi.fn(() => false);
const startPendingSubmission = vi.fn((input: { text: string; images?: InteractiveModeContext["pendingImages"] }) => {
ensureLoadingAnimation();
return createSubmission(input);
});
const startPendingSubmission = vi.fn(
(input: {
text: string;
images?: InteractiveModeContext["pendingImages"];
imageLinks?: InteractiveModeContext["pendingImageLinks"];
}) => {
ensureLoadingAnimation();
return createSubmission(input);
},
);
const editor: FakeEditor = {
setText(text: string) {
editorText = text;
@@ -104,7 +113,11 @@ function createContext(): {
ctx = {
editor: editor as unknown as InteractiveModeContext["editor"],
ui: { requestRender } as unknown as InteractiveModeContext["ui"],
ui: {
requestRender,
addInputListener: vi.fn(),
addStartListener: vi.fn(),
} as unknown as InteractiveModeContext["ui"],
loadingAnimation: undefined,
autoCompactionLoader: undefined,
retryLoader: undefined,
@@ -132,6 +145,7 @@ function createContext(): {
getKeys: () => [],
} as unknown as InteractiveModeContext["keybindings"],
pendingImages: [],
pendingImageLinks: [],
isBashMode: false,
isPythonMode: false,
optimisticUserMessageSignature: undefined,
@@ -197,7 +211,11 @@ describe("InputController escape behavior", () => {
controller.setupEditorSubmitHandler();
await editor.onSubmit?.("hello");
expect(spies.startPendingSubmission).toHaveBeenCalledWith({ text: "hello", images: undefined });
expect(spies.startPendingSubmission).toHaveBeenCalledWith({
text: "hello",
images: undefined,
imageLinks: undefined,
});
expect(spies.onInputCallback).toHaveBeenCalledWith(submission);
editor.onEscape?.();
@@ -232,6 +250,9 @@ describe("InputController escape behavior", () => {
expect(spies.cancelPendingSubmission).toHaveBeenCalledTimes(1);
expect(spies.clearQueue).toHaveBeenCalledTimes(1);
expect(spies.abort).toHaveBeenCalledTimes(1);
// The Esc interrupt threads a user-facing reason so the aborted turn and its
// synthetic tool results read as a deliberate interrupt, not "Request was aborted".
expect(spies.abort).toHaveBeenCalledWith({ reason: USER_INTERRUPT_LABEL });
});
it("prefers aborting bash before aborting an overlapping stream", () => {
@@ -6,6 +6,7 @@ type FakeEditor = {
onEscape?: () => void;
onClear?: () => void;
onExit?: () => void;
onDisplayReset?: () => void;
onSuspend?: () => void;
onCycleThinkingLevel?: () => void;
onCycleModelForward?: () => void;
@@ -20,23 +21,41 @@ type FakeEditor = {
onExternalEditor?: () => void;
onDequeue?: () => void;
onChange?: (text: string) => void;
onSubmit?: (text: string) => Promise<void>;
setText(text: string): void;
getText(): string;
addToHistory(text: string): void;
setActionKeys(action: string, keys: string[]): void;
setCustomKeyHandler(key: string, handler: () => void): void;
clearCustomKeyHandlers(): void;
pasteText(text: string): void;
};
async function createContext() {
let editorText = "";
const keyMap: Record<string, string[]> = {
"app.display.reset": ["ctrl+l"],
"app.model.selectTemporary": ["ctrl+y"],
"app.model.select": ["ctrl+l"],
"app.model.select": ["alt+m"],
};
const customHandlers = new Map<string, () => void>();
const setActionKeys = vi.fn();
const setCustomKeyHandler = vi.fn((key: string, handler: () => void) => {
customHandlers.set(key, handler);
});
const clearCustomKeyHandlers = vi.fn(() => {
customHandlers.clear();
});
const resetDisplay = vi.fn();
const showModelSelector = vi.fn();
const requestRender = vi.fn();
const addInputListener = vi.fn();
const addStartListener = vi.fn();
const terminalWrite = vi.fn();
const prompt = vi.fn(async () => {});
const abort = vi.fn(async () => {});
const interruptAndFlushQueuedMessages = vi.fn(async () => {});
const getQueuedMessages = vi.fn(() => ({ steering: [] as string[], followUp: [] as string[] }));
const updatePendingMessagesDisplay = vi.fn();
const editor: FakeEditor = {
setText(text: string) {
@@ -46,13 +65,22 @@ async function createContext() {
return editorText;
},
addToHistory: vi.fn(),
pasteText(text: string) {
editorText += text;
},
setActionKeys,
setCustomKeyHandler: vi.fn(),
clearCustomKeyHandlers: vi.fn(),
setCustomKeyHandler,
clearCustomKeyHandlers,
};
const ctx = {
editor: editor as unknown as InteractiveModeContext["editor"],
ui: { requestRender: vi.fn() } as unknown as InteractiveModeContext["ui"],
ui: {
requestRender,
resetDisplay,
addInputListener,
addStartListener,
terminal: { write: terminalWrite },
} as unknown as InteractiveModeContext["ui"],
loadingAnimation: undefined,
autoCompactionLoader: undefined,
retryLoader: undefined,
@@ -66,6 +94,10 @@ async function createContext() {
isEvalRunning: false,
extensionRunner: undefined,
prompt,
queuedMessageCount: 0,
getQueuedMessages,
abort,
interruptAndFlushQueuedMessages,
} as unknown as InteractiveModeContext["session"],
keybindings: {
getKeys(action: string) {
@@ -122,33 +154,61 @@ async function createContext() {
InputController,
ctx,
editor,
customHandlers,
spies: {
setActionKeys,
showModelSelector,
prompt,
updatePendingMessagesDisplay,
requestRender,
abort,
interruptAndFlushQueuedMessages,
getQueuedMessages,
resetDisplay,
},
};
}
describe("InputController keybinding setup", () => {
it("registers temporary and persisted model selector actions separately", async () => {
it("registers model selector and display reset actions separately", async () => {
const { InputController, ctx, editor, spies } = await createContext();
const controller = new InputController(ctx);
controller.setupKeyHandlers();
expect(spies.setActionKeys).toHaveBeenCalledWith("app.display.reset", ["ctrl+l"]);
expect(spies.setActionKeys).toHaveBeenCalledWith("app.model.selectTemporary", ["ctrl+y"]);
expect(spies.setActionKeys).toHaveBeenCalledWith("app.model.select", ["ctrl+l"]);
expect(spies.setActionKeys).toHaveBeenCalledWith("app.model.select", ["alt+m"]);
expect(editor.onDisplayReset).toBeDefined();
expect(editor.onSelectModelTemporary).toBeDefined();
expect(editor.onSelectModel).toBeDefined();
expect(editor.onSelectModelTemporary).not.toBe(editor.onSelectModel);
editor.onDisplayReset?.();
editor.onSelectModelTemporary?.();
editor.onSelectModel?.();
expect(spies.showModelSelector).toHaveBeenNthCalledWith(1, { temporaryOnly: true });
expect(spies.showModelSelector).toHaveBeenNthCalledWith(2);
expect(spies.resetDisplay).toHaveBeenCalledTimes(1);
});
it("empty Enter interrupts and sends a queued steering message", async () => {
const { InputController, ctx, editor, spies } = await createContext();
const session = ctx.session as unknown as { isStreaming: boolean; queuedMessageCount: number };
session.isStreaming = true;
session.queuedMessageCount = 1;
spies.getQueuedMessages.mockReturnValue({ steering: ["Send this now"], followUp: [] });
const controller = new InputController(ctx);
controller.setupEditorSubmitHandler();
await editor.onSubmit?.("");
expect(spies.interruptAndFlushQueuedMessages).toHaveBeenCalledWith({ reason: "Interrupted by user" });
expect(spies.abort).not.toHaveBeenCalled();
expect(spies.updatePendingMessagesDisplay).toHaveBeenCalledTimes(1);
expect(spies.requestRender).toHaveBeenCalledTimes(1);
expect(spies.prompt).not.toHaveBeenCalled();
});
it("marks streaming follow-up submissions as local", async () => {
@@ -94,7 +94,7 @@ function createStubInputControllerContext(opts: { skillCommands: Map<string, str
isBashMode: false,
isPythonMode: false,
pendingImages: [],
isBackgrounded: false,
pendingImageLinks: [],
loopModeEnabled: false,
compactionQueuedMessages: [],
locallySubmittedUserSignatures: new Set<string>(),
@@ -6,8 +6,7 @@ import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config
import { resolveLocalUrlToPath } from "@oh-my-pi/pi-coding-agent/internal-urls";
import { AssistantMessageComponent } from "@oh-my-pi/pi-coding-agent/modes/components/assistant-message";
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import { SILENT_ABORT_MARKER } from "@oh-my-pi/pi-coding-agent/session/messages";
import { Text } from "@oh-my-pi/pi-tui";
import { SILENT_ABORT_MARKER, USER_INTERRUPT_LABEL } from "@oh-my-pi/pi-coding-agent/session/messages";
import { formatNumber, TempDir } from "@oh-my-pi/pi-utils";
import { ModelRegistry } from "../src/config/model-registry";
import type { HookSelectorSlider } from "../src/modes/components/hook-selector";
@@ -68,7 +67,6 @@ describe("InteractiveMode plan review rendering", () => {
});
beforeEach(async () => {
Bun.gc(true);
resetSettingsForTest();
tempDir = TempDir.createSync("@pi-plan-review-");
await Settings.init({ inMemory: true, cwd: tempDir.path() });
@@ -111,10 +109,9 @@ describe("InteractiveMode plan review rendering", () => {
currentAuthStorage?.close();
currentTempDir?.removeSync();
resetSettingsForTest();
Bun.gc(true);
});
it("appends each submitted plan review preview to preserve scrollback", async () => {
it("forwards each submitted plan to the review overlay", async () => {
const planFilePath = "local://PLAN.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
@@ -124,38 +121,119 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(mode, "showHookSelector").mockResolvedValue("Refine plan");
const review = vi.spyOn(mode, "showPlanReview").mockResolvedValue("Refine plan");
await mode.handlePlanApproval({
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath: "local://PLAN.md",
});
const firstPreview = mode.chatContainer.children.at(-1);
expect(firstPreview).toBeDefined();
expect(firstPreview!.render(120).join("\n")).toContain("First plan");
expect(review.mock.calls[0]?.[0]).toContain("First plan");
const marker = new Text("MARKER", 0, 0);
mode.chatContainer.addChild(marker);
await Bun.write(resolvedPlanPath, "# Second plan\n\nbeta");
await mode.handlePlanApproval({
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath: "local://PLAN.md",
});
const secondPreview = mode.chatContainer.children.at(-1);
expect(secondPreview).toBeDefined();
expect(secondPreview).not.toBe(firstPreview);
expect(mode.chatContainer.children.at(-2)).toBe(marker);
expect(mode.chatContainer.children.at(-3)).toBe(firstPreview);
expect(firstPreview!.render(120).join("\n")).toContain("First plan");
expect(firstPreview!.render(120).join("\n")).not.toContain("Second plan");
expect(secondPreview!.render(120).join("\n")).toContain("Second plan");
// Each approval shows the current plan in the overlay, not a stale one.
expect(review.mock.calls[1]?.[0]).toContain("Second plan");
expect(review.mock.calls[1]?.[0]).not.toContain("First plan");
});
it("re-prompts the model with annotation feedback when Refine is chosen", async () => {
const planFilePath = "local://PLAN.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
});
await Bun.write(resolvedPlanPath, "# Plan\n\nbody");
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
const feedback = "Refinement feedback on the plan:\n\n## Goal\n- needs detail\n";
// The overlay reports annotation feedback through onFeedbackChange before the
// operator picks "Refine plan".
vi.spyOn(mode, "showPlanReview").mockImplementation(async (_plan, _title, _options, dialogOptions) => {
dialogOptions?.onFeedbackChange?.(feedback);
return "Refine plan";
});
const startSpy = vi
.spyOn(mode, "startPendingSubmission")
.mockReturnValue({ text: feedback, cancelled: false, started: false });
const onInput = vi.fn();
mode.onInputCallback = onInput;
await mode.handlePlanApproval({
planFilePath,
planExists: true,
title: "PLAN",
});
expect(startSpy).toHaveBeenCalledWith(expect.objectContaining({ text: expect.stringContaining("needs detail") }));
expect(onInput).toHaveBeenCalledTimes(1);
});
it("Refine with no annotations does not re-prompt the model", async () => {
const planFilePath = "local://PLAN.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
});
await Bun.write(resolvedPlanPath, "# Plan\n\nbody");
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(mode, "showPlanReview").mockResolvedValue("Refine plan");
const startSpy = vi.spyOn(mode, "startPendingSubmission");
const onInput = vi.fn();
mode.onInputCallback = onInput;
await mode.handlePlanApproval({
planFilePath,
planExists: true,
title: "PLAN",
});
expect(startSpy).not.toHaveBeenCalled();
expect(onInput).not.toHaveBeenCalled();
});
it("approves with in-overlay edits and mirrors them to the plan file", async () => {
const planFilePath = "local://PLAN.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
});
await Bun.write(resolvedPlanPath, "# Plan\n\noriginal body\n");
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
const edited = "# Plan\n\nedited body\n";
vi.spyOn(mode, "showPlanReview").mockImplementation(async (_plan, _title, _options, dialogOptions) => {
dialogOptions?.onPlanEdited?.(edited);
return "Approve and execute";
});
vi.spyOn(mode, "handleClearCommand").mockResolvedValue();
const promptSpy = vi.spyOn(session, "prompt").mockResolvedValue(undefined as never);
await mode.handlePlanApproval({
planFilePath,
planExists: true,
title: "PLAN",
});
// The synthetic plan-approved prompt carries the in-overlay edit, not the
// stale on-disk content (preferring editedContent avoids the write race).
const call = promptSpy.mock.calls.find(isPlanApprovedCall);
expect(call).toBeDefined();
expect(call?.[0] as string).toContain("edited body");
expect(call?.[0] as string).not.toContain("original body");
// onPlanEdited mirrored the edit to the plan file.
expect(await Bun.file(resolvedPlanPath).text()).toContain("edited body");
});
it("offers approve-and-keep-context as a distinct plan approval path", async () => {
@@ -169,16 +247,16 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(session, "getContextUsage").mockReturnValue({ tokens: 7320, contextWindow: 10000, percent: 73.2 });
const selector = vi.spyOn(mode, "showHookSelector").mockResolvedValue("Refine plan");
const selector = vi.spyOn(mode, "showPlanReview").mockResolvedValue("Refine plan");
await mode.handlePlanApproval({
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath: "local://APPROVED.md",
});
expect(selector).toHaveBeenCalledWith(
expect.any(String),
"Plan mode - next step",
[
"Approve and execute",
@@ -246,17 +324,16 @@ describe("InteractiveMode plan review rendering", () => {
percent: (tokens / contextWindow) * 100,
};
});
const selector = vi.spyOn(mode, "showHookSelector").mockResolvedValue("Refine plan");
const selector = vi.spyOn(mode, "showPlanReview").mockResolvedValue("Refine plan");
await mode.handlePlanApproval({
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath: "local://APPROVED.md",
});
expect(contextSpy).toHaveBeenCalledWith({ contextWindow: executionModel.contextWindow });
expect(selector.mock.calls[0]?.[1]).toEqual([
expect(selector.mock.calls[0]?.[2]).toEqual([
"Approve and execute",
"Approve and compact context",
`Approve and keep context (~${compactNumber(tokens)} / ${compactNumber(executionModel.contextWindow)})`,
@@ -275,16 +352,15 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(session, "getContextUsage").mockReturnValue({ tokens: 9600, contextWindow: 10000, percent: 96 });
const selector = vi.spyOn(mode, "showHookSelector").mockResolvedValue("Refine plan");
const selector = vi.spyOn(mode, "showPlanReview").mockResolvedValue("Refine plan");
await mode.handlePlanApproval({
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath: "local://APPROVED.md",
});
expect(selector.mock.calls[0]?.[2]).toEqual(
expect(selector.mock.calls[0]?.[3]).toEqual(
expect.objectContaining({
disabledIndices: [2],
}),
@@ -302,16 +378,15 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(session, "getContextUsage").mockReturnValue({ tokens: 9500, contextWindow: 10000, percent: 95 });
const selector = vi.spyOn(mode, "showHookSelector").mockResolvedValue("Refine plan");
const selector = vi.spyOn(mode, "showPlanReview").mockResolvedValue("Refine plan");
await mode.handlePlanApproval({
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath: "local://APPROVED.md",
});
expect(selector.mock.calls[0]?.[2]).toEqual(
expect(selector.mock.calls[0]?.[3]).toEqual(
expect.objectContaining({
disabledIndices: undefined,
}),
@@ -330,16 +405,16 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModePlanFilePath = planFilePath;
// Post-compaction: tokens unknown until the next LLM response.
vi.spyOn(session, "getContextUsage").mockReturnValue({ tokens: null, contextWindow: 200000, percent: null });
const selector = vi.spyOn(mode, "showHookSelector").mockResolvedValue("Refine plan");
const selector = vi.spyOn(mode, "showPlanReview").mockResolvedValue("Refine plan");
await mode.handlePlanApproval({
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath: "local://APPROVED.md",
});
expect(selector).toHaveBeenCalledWith(
expect.any(String),
"Plan mode - next step",
["Approve and execute", "Approve and compact context", "Approve and keep context", "Refine plan"],
expect.any(Object),
@@ -349,12 +424,11 @@ describe("InteractiveMode plan review rendering", () => {
it("approves a plan without clearing the session when keeping context", async () => {
const planFilePath = "local://PLAN.md";
const finalPlanFilePath = "local://APPROVED.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
});
const resolvedFinalPlanPath = resolveLocalUrlToPath(finalPlanFilePath, {
const resolvedFinalPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
});
@@ -363,7 +437,7 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(session, "getContextUsage").mockReturnValue({ tokens: null, contextWindow: 200000, percent: null });
vi.spyOn(mode, "showHookSelector").mockResolvedValue("Approve and keep context");
vi.spyOn(mode, "showPlanReview").mockResolvedValue("Approve and keep context");
const clear = vi.spyOn(mode, "handleClearCommand").mockResolvedValue();
const prompt = vi.spyOn(session, "prompt").mockResolvedValue(undefined as never);
@@ -371,7 +445,6 @@ describe("InteractiveMode plan review rendering", () => {
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath,
});
expect(clear).not.toHaveBeenCalled();
@@ -383,7 +456,6 @@ describe("InteractiveMode plan review rendering", () => {
it("keeps the existing approve-and-execute path clearing the session", async () => {
const planFilePath = "local://PLAN.md";
const finalPlanFilePath = "local://APPROVED.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
@@ -392,7 +464,7 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(mode, "showHookSelector").mockResolvedValue("Approve and execute");
vi.spyOn(mode, "showPlanReview").mockResolvedValue("Approve and execute");
const clear = vi.spyOn(mode, "handleClearCommand").mockResolvedValue();
const prompt = vi.spyOn(session, "prompt").mockResolvedValue(undefined as never);
@@ -400,7 +472,6 @@ describe("InteractiveMode plan review rendering", () => {
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath,
});
expect(clear).toHaveBeenCalledTimes(1);
@@ -428,7 +499,6 @@ describe("InteractiveMode plan review rendering", () => {
session.settings.setModelRole("plan", "anthropic/claude-sonnet-4-5");
const planFilePath = "local://PLAN.md";
const finalPlanFilePath = "local://APPROVED.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
@@ -445,8 +515,8 @@ describe("InteractiveMode plan review rendering", () => {
vi.spyOn(session, "prompt").mockResolvedValue(undefined as never);
let observedSegments: string[] = [];
vi.spyOn(mode, "showHookSelector").mockImplementation(
async (_title, _options, _dialogOptions, extra?: { slider?: HookSelectorSlider }) => {
vi.spyOn(mode, "showPlanReview").mockImplementation(
async (_planContent, _title, _options, _dialogOptions, extra?: { slider?: HookSelectorSlider }) => {
const slider = extra?.slider;
expect(slider).toBeDefined();
observedSegments = slider!.segments.map(segment => segment.label);
@@ -462,7 +532,6 @@ describe("InteractiveMode plan review rendering", () => {
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath,
});
expect(observedSegments).toEqual(["default", "slow"]);
@@ -473,7 +542,6 @@ describe("InteractiveMode plan review rendering", () => {
it("re-enters plan mode on the approved titled artifact after approve-and-execute", async () => {
const planFilePath = "local://PLAN.md";
const finalPlanFilePath = "local://APPROVED.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
@@ -482,7 +550,7 @@ describe("InteractiveMode plan review rendering", () => {
await mode.handlePlanModeCommand();
vi.spyOn(mode, "showHookSelector").mockResolvedValue("Approve and execute");
vi.spyOn(mode, "showPlanReview").mockResolvedValue("Approve and execute");
vi.spyOn(mode, "handleClearCommand").mockResolvedValue();
vi.spyOn(session, "prompt").mockResolvedValue(undefined as never);
@@ -490,23 +558,21 @@ describe("InteractiveMode plan review rendering", () => {
planFilePath,
planExists: true,
title: "APPROVED",
finalPlanFilePath,
});
expect(mode.planModeEnabled).toBe(false);
expect(session.getPlanReferencePath()).toBe(finalPlanFilePath);
expect(session.getPlanReferencePath()).toBe(planFilePath);
await mode.handlePlanModeCommand();
expect(session.getPlanModeState()).toMatchObject({
enabled: true,
planFilePath: finalPlanFilePath,
planFilePath,
reentry: true,
});
});
it("Approve and compact context: ok outcome dispatches plan-approved after compaction", async () => {
const planFilePath = "local://PLAN.md";
const finalPlanFilePath = "local://APPROVED.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
@@ -515,7 +581,7 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(mode, "showHookSelector").mockResolvedValue("Approve and compact context");
vi.spyOn(mode, "showPlanReview").mockResolvedValue("Approve and compact context");
const compactSpy = vi.spyOn(mode, "handleCompactCommand").mockResolvedValue("ok");
const markSentSpy = vi.spyOn(session, "markPlanReferenceSent");
const promptSpy = vi.spyOn(session, "prompt").mockResolvedValue(undefined as never);
@@ -524,14 +590,13 @@ describe("InteractiveMode plan review rendering", () => {
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath,
});
// Compaction was run with the rendered planning-specific custom instruction.
expect(compactSpy).toHaveBeenCalledTimes(1);
const [compactInstruction] = compactSpy.mock.calls[0]!;
expect(typeof compactInstruction).toBe("string");
expect(compactInstruction as string).toContain(finalPlanFilePath);
expect(compactInstruction as string).toContain(planFilePath);
// Plan-approved synthetic prompt was dispatched.
const planApprovedIdx = promptSpy.mock.calls.findIndex(isPlanApprovedCall);
@@ -549,7 +614,6 @@ describe("InteractiveMode plan review rendering", () => {
// CompactionOutcome boundary; the underlying executeCompaction → sentinel
// classification path is producer-layer and not under T3's contract.)
const planFilePath = "local://PLAN.md";
const finalPlanFilePath = "local://APPROVED.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
@@ -558,7 +622,7 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(mode, "showHookSelector").mockResolvedValue("Approve and compact context");
vi.spyOn(mode, "showPlanReview").mockResolvedValue("Approve and compact context");
vi.spyOn(mode, "handleCompactCommand").mockResolvedValue("cancelled");
const showWarningSpy = vi.spyOn(mode, "showWarning");
const setPlanRefSpy = vi.spyOn(session, "setPlanReferencePath");
@@ -569,7 +633,6 @@ describe("InteractiveMode plan review rendering", () => {
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath,
});
// Operator was told the dispatch was deferred.
@@ -578,7 +641,7 @@ describe("InteractiveMode plan review rendering", () => {
);
// Plan reference path was recorded so the session knows about the approved
// plan at its final destination …
expect(setPlanRefSpy).toHaveBeenCalledWith(finalPlanFilePath);
expect(setPlanRefSpy).toHaveBeenCalledWith(planFilePath);
// … but markPlanReferenceSent was NOT called, so the next operator turn
// will inject the reference fresh via #buildPlanReferenceMessage. This is
// the load-bearing assertion that the cancel path leaves the executor
@@ -592,7 +655,6 @@ describe("InteractiveMode plan review rendering", () => {
// Mock `handleCompactCommand` to surface the "failed" outcome directly.
// Failure → approval intent stands → synthetic dispatch fires.
const planFilePath = "local://PLAN.md";
const finalPlanFilePath = "local://APPROVED.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
@@ -601,7 +663,7 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(mode, "showHookSelector").mockResolvedValue("Approve and compact context");
vi.spyOn(mode, "showPlanReview").mockResolvedValue("Approve and compact context");
vi.spyOn(mode, "handleCompactCommand").mockResolvedValue("failed");
const markSentSpy = vi.spyOn(session, "markPlanReferenceSent");
const promptSpy = vi.spyOn(session, "prompt").mockResolvedValue(undefined as never);
@@ -610,7 +672,6 @@ describe("InteractiveMode plan review rendering", () => {
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath,
});
// Plan-approved synthetic prompt WAS dispatched despite the failure.
@@ -625,7 +686,6 @@ describe("InteractiveMode plan review rendering", () => {
// hit #buildPlanReferenceMessage with the stale plan-mode path. Pin it
// before the compaction await.
const planFilePath = "local://PLAN.md";
const finalPlanFilePath = "local://APPROVED.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
@@ -634,13 +694,13 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(mode, "showHookSelector").mockResolvedValue("Approve and compact context");
vi.spyOn(mode, "showPlanReview").mockResolvedValue("Approve and compact context");
vi.spyOn(session, "prompt").mockResolvedValue(undefined as never);
const setPlanRefSpy = vi.spyOn(session, "setPlanReferencePath");
let planRefSetWhenCompactionRan = false;
vi.spyOn(mode, "handleCompactCommand").mockImplementation(async () => {
planRefSetWhenCompactionRan = setPlanRefSpy.mock.calls.some(call => call[0] === finalPlanFilePath);
planRefSetWhenCompactionRan = setPlanRefSpy.mock.calls.some(call => call[0] === planFilePath);
return "ok";
});
@@ -648,7 +708,6 @@ describe("InteractiveMode plan review rendering", () => {
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath,
});
// The contract: by the time handleCompactCommand runs (and flushes the
@@ -678,7 +737,6 @@ describe("InteractiveMode plan review rendering", () => {
throwError?: Error,
): Promise<void> {
const planFilePath = "local://PLAN.md";
const finalPlanFilePath = "local://APPROVED.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
@@ -687,7 +745,7 @@ describe("InteractiveMode plan review rendering", () => {
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(mode, "showHookSelector").mockResolvedValue("Approve and compact context");
vi.spyOn(mode, "showPlanReview").mockResolvedValue("Approve and compact context");
if (compactOutcome === "throw") {
vi.spyOn(mode, "handleCompactCommand").mockRejectedValue(throwError ?? new Error("compact boom"));
} else {
@@ -699,7 +757,6 @@ describe("InteractiveMode plan review rendering", () => {
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath,
});
}
@@ -737,7 +794,6 @@ describe("InteractiveMode plan review rendering", () => {
it("B5: Approve and execute (no compact) → markPlanCompactAbortPending never called; flag stays false", async () => {
const planFilePath = "local://PLAN.md";
const finalPlanFilePath = "local://APPROVED.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
@@ -745,7 +801,7 @@ describe("InteractiveMode plan review rendering", () => {
await Bun.write(resolvedPlanPath, "# Plan\n\nBody.");
mode.planModeEnabled = true;
mode.planModePlanFilePath = planFilePath;
vi.spyOn(mode, "showHookSelector").mockResolvedValue("Approve and execute");
vi.spyOn(mode, "showPlanReview").mockResolvedValue("Approve and execute");
const markSpy = vi.spyOn(session, "markPlanCompactAbortPending");
vi.spyOn(session, "prompt").mockResolvedValue(undefined as never);
@@ -753,7 +809,6 @@ describe("InteractiveMode plan review rendering", () => {
planFilePath,
planExists: true,
title: "PLAN",
finalPlanFilePath,
});
expect(markSpy).not.toHaveBeenCalled();
@@ -762,7 +817,6 @@ describe("InteractiveMode plan review rendering", () => {
it("re-enters plan mode on the approved titled artifact after approval", async () => {
const planFilePath = "local://PLAN.md";
const finalPlanFilePath = "local://APPROVED.md";
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
@@ -773,7 +827,7 @@ describe("InteractiveMode plan review rendering", () => {
expect(session.getPlanModeState()?.planFilePath).toBe(planFilePath);
vi.spyOn(session, "getContextUsage").mockReturnValue({ tokens: null, contextWindow: 200000, percent: null });
const selector = vi.spyOn(mode, "showHookSelector").mockResolvedValue("Approve and keep context");
const selector = vi.spyOn(mode, "showPlanReview").mockResolvedValue("Approve and keep context");
const showError = vi.spyOn(mode, "showError");
vi.spyOn(session, "prompt").mockResolvedValue(undefined as never);
@@ -781,24 +835,22 @@ describe("InteractiveMode plan review rendering", () => {
planFilePath,
planExists: true,
title: "APPROVED",
finalPlanFilePath,
});
expect(mode.planModeEnabled).toBe(false);
expect(session.getPlanReferencePath()).toBe(finalPlanFilePath);
expect(session.getPlanReferencePath()).toBe(planFilePath);
await mode.handlePlanModeCommand();
expect(session.getPlanModeState()).toMatchObject({
enabled: true,
planFilePath: finalPlanFilePath,
planFilePath,
reentry: true,
});
await mode.handlePlanApproval({
planFilePath: finalPlanFilePath,
planFilePath,
planExists: true,
title: "APPROVED",
finalPlanFilePath,
});
expect(selector).toHaveBeenCalledTimes(2);
@@ -808,9 +860,10 @@ describe("InteractiveMode plan review rendering", () => {
// ==========================================================================
// Phase 6 — D layer: replay-side render branches in AssistantMessageComponent.
//
// D1 asserts that the persisted `SILENT_ABORT_MARKER` suppresses the red
// "Operation aborted" line. D2 is the over-suppression regression guard —
// an aborted message with NO marker must still render the line.
// D1 asserts that the persisted `SILENT_ABORT_MARKER` suppresses the red abort
// line. D2 is the over-suppression regression guard — an aborted message with
// NO marker still renders the generic label. D3 covers a threaded interrupt
// reason rendering verbatim.
// ==========================================================================
function renderAssistant(message: AssistantMessage, width = 120): string {
@@ -840,20 +893,30 @@ describe("InteractiveMode plan review rendering", () => {
};
}
it("D1: Replay of an assistant message with SILENT_ABORT_MARKER + aborted: rendered component contains no /Operation aborted/", () => {
it("D1: Replay of an assistant message with SILENT_ABORT_MARKER + aborted: rendered component contains no abort line", () => {
const message = buildAbortedAssistantMessage({ errorMessage: SILENT_ABORT_MARKER });
const rendered = renderAssistant(message);
expect(rendered).not.toMatch(/Operation aborted/);
expect(rendered).not.toContain("Operation aborted");
expect(rendered).not.toContain(USER_INTERRUPT_LABEL);
// The marker itself MUST NOT leak into rendered output either.
expect(rendered).not.toContain(SILENT_ABORT_MARKER);
});
it("D2: Replay of an aborted message with no marker + empty content: rendered component DOES contain 'Operation aborted'", () => {
it("D2: Replay of an aborted message with no threaded reason + empty content: rendered component DOES contain the generic label", () => {
// Over-suppression regression guard: silent path is opt-in via the
// persisted marker. A user-cancel abort with no marker and no content
// still surfaces the standard label.
// persisted marker. An abort with no marker and no threaded reason still
// surfaces the generic operator-facing label.
const message = buildAbortedAssistantMessage({ content: [], errorMessage: undefined });
const rendered = renderAssistant(message);
expect(rendered).toContain("Operation aborted");
});
it("D3: Replay of an aborted message carrying a user-interrupt reason renders it verbatim", () => {
// The Esc-interrupt reason persisted on errorMessage must render as-is,
// not collapse into the generic label.
const message = buildAbortedAssistantMessage({ content: [], errorMessage: USER_INTERRUPT_LABEL });
const rendered = renderAssistant(message);
expect(rendered).toContain(USER_INTERRUPT_LABEL);
expect(rendered).not.toContain("Operation aborted");
});
});
@@ -1,67 +0,0 @@
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { Agent } from "@oh-my-pi/pi-agent-core";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { InteractiveMode } from "@oh-my-pi/pi-coding-agent/modes/interactive-mode";
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { postmortem, TempDir } from "@oh-my-pi/pi-utils";
describe("InteractiveMode shutdown", () => {
let authStorage: AuthStorage;
let mode: InteractiveMode;
let session: AgentSession;
let tempDir: TempDir;
beforeAll(() => {
initTheme();
});
beforeEach(async () => {
resetSettingsForTest();
tempDir = TempDir.createSync("@pi-shutdown-");
await Settings.init({ inMemory: true, cwd: tempDir.path() });
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
const modelRegistry = new ModelRegistry(authStorage);
const model = modelRegistry.find("anthropic", "claude-sonnet-4-5");
if (!model) throw new Error("Expected claude-sonnet-4-5 test model");
session = new AgentSession({
agent: new Agent({ initialState: { model, systemPrompt: ["Test"], tools: [], messages: [] } }),
sessionManager: SessionManager.create(tempDir.path(), tempDir.path()),
settings: Settings.isolated(),
modelRegistry,
});
mode = new InteractiveMode(session, "test");
});
afterEach(async () => {
mode?.stop();
vi.restoreAllMocks();
await session?.dispose();
authStorage?.close();
tempDir?.removeSync();
resetSettingsForTest();
});
it("stops from the last committed TUI frame without forcing a teardown repaint", async () => {
const requestRenderSpy = vi.spyOn(mode.ui, "requestRender").mockImplementation(() => {});
const stopSpy = vi.spyOn(mode.ui, "stop").mockImplementation(() => {});
const drainSpy = vi.spyOn(mode.ui.terminal, "drainInput").mockResolvedValue(undefined);
const disposeSpy = vi.spyOn(session, "dispose").mockResolvedValue(undefined);
const quitSpy = vi.spyOn(postmortem, "quit").mockResolvedValue(undefined);
vi.spyOn(session.sessionManager, "getSessionId").mockReturnValue("");
mode.isInitialized = true;
await mode.shutdown();
expect(disposeSpy).toHaveBeenCalled();
expect(requestRenderSpy.mock.calls.some(call => call[0] === true)).toBe(false);
expect(drainSpy).toHaveBeenCalledWith(1000);
expect(stopSpy).toHaveBeenCalled();
expect(quitSpy).toHaveBeenCalledWith(0);
});
});
@@ -4,7 +4,7 @@ import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types";
import { UiHelpers } from "@oh-my-pi/pi-coding-agent/modes/utils/ui-helpers";
import { buildSessionContext, type SessionContext } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { Container } from "@oh-my-pi/pi-tui";
import { type Component, Container } from "@oh-my-pi/pi-tui";
function renderLastLine(container: Container, width = 120): string {
const last = container.children[container.children.length - 1];
@@ -25,7 +25,11 @@ function createInitialRenderHarness(): { ctx: InteractiveModeContext; helpers: U
pendingPythonComponents: [],
pendingTools: new Map(),
ui: { requestRender: vi.fn() },
isBackgrounded: false,
present: (content: Component | readonly Component[]) => {
const items = Array.isArray(content) ? content : [content];
for (const item of items) ctx.chatContainer.addChild(item);
ctx.ui.requestRender();
},
sessionManager: {
buildSessionContext: () => buildSessionContext([]),
getEntries: () => [],
@@ -59,7 +63,11 @@ describe("InteractiveMode.showStatus", () => {
const ctx = {
chatContainer: new Container(),
ui: { requestRender: vi.fn() },
isBackgrounded: false,
present: (content: Component | readonly Component[]) => {
const items = Array.isArray(content) ? content : [content];
for (const item of items) ctx.chatContainer.addChild(item);
ctx.ui.requestRender();
},
lastStatusSpacer: undefined,
lastStatusText: undefined,
} as unknown as InteractiveModeContext;
@@ -80,7 +88,11 @@ describe("InteractiveMode.showStatus", () => {
const ctx = {
chatContainer: new Container(),
ui: { requestRender: vi.fn() },
isBackgrounded: false,
present: (content: Component | readonly Component[]) => {
const items = Array.isArray(content) ? content : [content];
for (const item of items) ctx.chatContainer.addChild(item);
ctx.ui.requestRender();
},
lastStatusSpacer: undefined,
lastStatusText: undefined,
} as unknown as InteractiveModeContext;
@@ -37,6 +37,7 @@ interface ModelRegistryLike {
find: (...args: unknown[]) => Model;
getAll: () => Model[];
getApiKey: (...args: unknown[]) => Promise<string>;
resolver: (...args: unknown[]) => () => Promise<string>;
}
const createdDirs = new Set<string>();
@@ -62,6 +63,7 @@ function createModelRegistry(model: Model): ModelRegistryLike {
find: vi.fn(() => model),
getAll: vi.fn(() => [model]),
getApiKey: vi.fn(async () => "test-api-key"),
resolver: vi.fn(() => async () => "test-api-key"),
};
}
@@ -45,7 +45,6 @@ describe("issue #899 — sync git metadata reads must survive EINTR", () => {
}) as typeof fs.readFileSync);
// On main this throws EINTR; after fix it must return null (metadata unavailable).
expect(() => head.resolveSync(tempDir)).not.toThrow();
expect(head.resolveSync(tempDir)).toBeNull();
expect(spy).toHaveBeenCalled();
});
@@ -80,6 +80,10 @@ describe("issue #956: interactive /mcp test", () => {
const disconnectServer = vi.spyOn(mcpClient, "disconnectServer").mockResolvedValue();
const controller = new MCPCommandController({
chatContainer: { addChild },
present: (content: unknown) => {
for (const item of Array.isArray(content) ? content : [content]) addChild(item);
requestRender();
},
ui: { requestRender },
editor: {},
showError,
@@ -106,4 +106,71 @@ describe("KeybindingsManager.create", () => {
await fs.rm(agentDir, { recursive: true, force: true });
}
});
it("defaults model selection to Alt+M and display reset to Ctrl+L", () => {
const manager = KeybindingsManager.inMemory();
expect(manager.getKeys("app.model.select")).toEqual(["alt+m"]);
expect(manager.getKeys("app.display.reset")).toEqual(["ctrl+l"]);
});
it("keeps the Ctrl+L display reset default when an old model remap still claims Ctrl+L", () => {
const manager = KeybindingsManager.inMemory({
"app.model.select": "ctrl+l",
});
expect(manager.getKeys("app.model.select")).toEqual(["ctrl+l"]);
expect(manager.getKeys("app.display.reset")).toEqual(["ctrl+l"]);
expect(manager.getEffectiveConfig()["app.display.reset"]).toBe("ctrl+l");
});
it("keeps Ctrl+L when the user explicitly assigns it to display reset", () => {
const manager = KeybindingsManager.inMemory({
"app.display.reset": "ctrl+l",
});
expect(manager.getKeys("app.display.reset")).toEqual(["ctrl+l"]);
});
it("defaults the follow-up shortcut to both Ctrl+Q and Ctrl+Enter (#1903)", async () => {
const agentDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-keybindings-"));
try {
const manager = KeybindingsManager.create(agentDir);
// Both chords must be registered so Windows Terminal users (which swallow
// Ctrl+Enter at the terminal layer) get a working follow-up binding out
// of the box, without breaking users on Kitty/iTerm2/WezTerm/Ghostty.
expect(manager.getKeys("app.message.followUp")).toEqual(["ctrl+q", "ctrl+enter"]);
} finally {
await fs.rm(agentDir, { recursive: true, force: true });
}
});
it("removes the Ctrl+Q follow-up default when a user remap already claims it (#1903)", () => {
const manager = KeybindingsManager.inMemory({
"app.plan.toggle": "ctrl+q",
});
expect(manager.getKeys("app.plan.toggle")).toEqual(["ctrl+q"]);
expect(manager.getKeys("app.message.followUp")).toEqual(["ctrl+enter"]);
expect(manager.getDisplayString("app.message.followUp")).toBe("Ctrl+Enter");
expect(manager.getEffectiveConfig()["app.message.followUp"]).toBe("ctrl+enter");
});
it("keeps the Ctrl+Q follow-up default when only an unknown config key claims it (#1903)", () => {
const manager = KeybindingsManager.inMemory({
"unknown.action": "ctrl+q",
});
expect(manager.getKeys("app.message.followUp")).toEqual(["ctrl+q", "ctrl+enter"]);
});
it("keeps Ctrl+Q when the user explicitly assigns it to follow-up (#1903)", () => {
const manager = KeybindingsManager.inMemory({
"app.message.followUp": "ctrl+q",
});
expect(manager.getKeys("app.message.followUp")).toEqual(["ctrl+q"]);
});
});
@@ -1,4 +1,4 @@
import { afterEach, beforeAll, describe, expect, it } from "bun:test";
import { afterEach, beforeAll, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
@@ -86,8 +86,15 @@ async function createHistoryStorage(prompts: string[]): Promise<HistoryStorage>
tempDirs.push(dir);
HistoryStorage.resetInstance();
const storage = HistoryStorage.open(path.join(dir, "history.db"));
for (const prompt of prompts) {
await storage.add(prompt);
// add() batches writes behind a 100ms AsyncDrain timer. Drive that timer with
// fake timers so the flush is instant instead of waiting real wall-clock time.
vi.useFakeTimers();
try {
const writes = prompts.map(prompt => storage.add(prompt));
vi.advanceTimersByTime(100);
await Promise.all(writes);
} finally {
vi.useRealTimers();
}
return storage;
}
@@ -16,7 +16,6 @@ describe("/loop slash command", () => {
handleLoopCommand,
editor: { setText: vi.fn() },
},
handleBackgroundCommand: vi.fn(),
} as unknown as BuiltinSlashCommandRuntime;
const result = await executeBuiltinSlashCommand("/loop 10min", runtime);
@@ -91,13 +91,6 @@ describe("OMP registry path contract", () => {
const expected = path.join(tmpHome, ".omp", "plugins", "installed_plugins.json");
expect(ompRegistryPath).toBe(expected);
});
it("OMP config dir name is .omp", () => {
// Validate our hardcoded constant matches getConfigDirName().
// If getConfigDirName() ever changes, this assertion will fail and
// we'll know the path constant here must be updated too.
expect(OMP_CONFIG_DIR).toBe(".omp");
});
});
// ── Format compatibility ───────────────────────────────────────────────────────
@@ -230,21 +223,4 @@ describe("OMP precedence contract (registry structure)", () => {
expect(id.slice(0, atIndex)).toBe("shared-plugin");
expect(id.slice(atIndex + 1)).toBe("common-mkt");
});
it("installPath deduplication: same path → one entry", () => {
// Mirrors the deduplication check: roots.some(r => r.id === pluginId && r.path === entry.installPath)
const id = buildPluginId("dup-plugin", "mkt");
const sharedPath = "/tmp/shared-install-path";
// Simulate what listClaudePluginRoots would do:
const roots: Array<{ id: string; path: string }> = [{ id, path: sharedPath }];
// Second entry with same installPath should be deduplicated
const isDuplicate = roots.some(r => r.id === id && r.path === sharedPath);
expect(isDuplicate).toBe(true);
// Entry with different installPath should NOT be deduplicated
const isDifferent = roots.some(r => r.id === id && r.path === "/tmp/other-path");
expect(isDifferent).toBe(false);
});
});
@@ -52,10 +52,19 @@ describe("MCP reconnect storm (issue #1592)", () => {
try {
await manager.connectServers({ crashy: config }, {});
// Give the reconnect loop generous time to fire. With the bug this
// produced thousands of processes within a second; with the fix the
// circuit breaker caps the per-server spawn budget.
await Bun.sleep(3000);
// Wait for the circuit breaker to trip rather than blind-sleeping a
// fixed budget. During the storm `getConnectionStatus` is always
// "connected" or "connecting" (`#pendingReconnections` is set
// synchronously before any await in `#doReconnect`); it only reports
// "disconnected" once `#tripReconnectBreaker` opens, tears down the
// stale connection, and detaches `onClose` so no further spawns fire.
// That makes the terminal state a race-free signal: poll for it and
// return the instant the storm is capped instead of waiting out a
// fixed 3s. Generous deadline stays well under the 15s test timeout.
const deadline = Date.now() + 10_000;
while (manager.getConnectionStatus("crashy") !== "disconnected" && Date.now() < deadline) {
await Bun.sleep(5);
}
const spawns = countSpawns();
// `RECONNECT_BURST_LIMIT` (5) is the per-server reconnect cap inside
@@ -46,6 +46,7 @@ function createModelRegistry(model: Model): any {
find: vi.fn(() => model),
getAll: vi.fn(() => [model]),
getApiKey: vi.fn(async () => "test-api-key"),
resolver: vi.fn(() => async () => "test-api-key"),
};
}
@@ -11,10 +11,10 @@ describe("resolveMemoryBackend", () => {
resetSettingsForTest();
});
it("returns the hindsight backend when memory.backend is hindsight, regardless of legacy memories.enabled", () => {
it("returns the hindsight backend when memory.backend is hindsight, regardless of legacy memories.enabled", async () => {
const a = Settings.isolated({ "memory.backend": "hindsight", "memories.enabled": false });
const b = Settings.isolated({ "memory.backend": "hindsight", "memories.enabled": true });
expect(resolveMemoryBackend(a).id).toBe("hindsight");
expect(resolveMemoryBackend(b).id).toBe("hindsight");
expect((await resolveMemoryBackend(a)).id).toBe("hindsight");
expect((await resolveMemoryBackend(b)).id).toBe("hindsight");
});
});
@@ -0,0 +1,66 @@
import { describe, expect, test } from "bun:test";
import {
getBracketStrippedModelIdCandidates,
getLongestModelLikeIdSegment,
getModelLikeIdSegments,
stripBracketedModelIdAffixes,
} from "../src/config/model-id-affixes";
describe("getModelLikeIdSegments", () => {
test("keeps only family-prefixed segments that carry a digit, deduped", () => {
expect(getModelLikeIdSegments("openrouter/anthropic/claude-3.5-sonnet")).toEqual(["claude-3.5-sonnet"]);
// `random-text` lacks a family prefix; `claude` (no digit) is dropped.
expect(getModelLikeIdSegments("random-text claude gemini-2")).toEqual(["gemini-2"]);
});
test("orders longest first with lexicographic tie-break", () => {
expect(getModelLikeIdSegments("claude-3 claude-3-5-haiku claude-2")).toEqual([
"claude-3-5-haiku",
"claude-2",
"claude-3",
]);
});
test("normalizes whitespace and case before matching", () => {
expect(getModelLikeIdSegments(" GLM-4.5-Air GEMINI-2 ")).toEqual(["glm-4.5-air", "gemini-2"]);
});
test("returns empty for ids with no model-like segment", () => {
expect(getModelLikeIdSegments("")).toEqual([]);
expect(getModelLikeIdSegments("just some words")).toEqual([]);
});
});
describe("getLongestModelLikeIdSegment", () => {
test("matches getModelLikeIdSegments[0]", () => {
const id = "[Kiro] claude-3 claude-3-5-sonnet";
expect(getLongestModelLikeIdSegment(id)).toBe(getModelLikeIdSegments(id)[0]);
expect(getLongestModelLikeIdSegment(id)).toBe("claude-3-5-sonnet");
});
test("is undefined when nothing matches", () => {
expect(getLongestModelLikeIdSegment("vendor/unknown-tag")).toBeUndefined();
});
});
describe("getBracketStrippedModelIdCandidates", () => {
test("no brackets yields no candidates", () => {
expect(getBracketStrippedModelIdCandidates("claude-opus-4-8")).toEqual([]);
});
test("strips leading reseller tag", () => {
expect(getBracketStrippedModelIdCandidates("[Kiro] claude-opus-4-8")).toEqual(["claude-opus-4-8"]);
});
test("strips both ends first, then each side, in preference order", () => {
expect(getBracketStrippedModelIdCandidates("[gcli转] gemini-3.1-pro-preview [假流]")).toEqual([
"gemini-3.1-pro-preview",
"gemini-3.1-pro-preview [假流]",
"[gcli转] gemini-3.1-pro-preview",
]);
});
test("supports full-width brackets", () => {
expect(stripBracketedModelIdAffixes("【供应商】 deepseek-v3 【限时】")).toBe("deepseek-v3");
});
});
@@ -1,4 +1,4 @@
import { afterEach, beforeEach, describe, expect, test } from "bun:test";
import { afterEach, beforeEach, describe, expect, type Mock, spyOn, test } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
@@ -13,6 +13,9 @@ describe("ModelRegistry runtime provider registration", () => {
let tempDir: string;
let modelsJsonPath: string;
let authStorage: AuthStorage;
// Neutralizes real network egress during "online" refresh tests so the merge
// path runs without wall-clock-bound DNS/socket latency. Restored in afterEach.
let fetchSpy: Mock<typeof fetch> | undefined;
const sourceIds = ["ext://atomic", "ext://runtime", "ext://oauth"];
@@ -24,6 +27,8 @@ describe("ModelRegistry runtime provider registration", () => {
});
afterEach(() => {
fetchSpy?.mockRestore();
fetchSpy = undefined;
clearCustomApis();
for (const sourceId of sourceIds) {
unregisterOAuthProviders(sourceId);
@@ -192,6 +197,11 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("extension-registered models survive refresh('online') cycle", async () => {
// The contract is overlay survival through the full online refresh path
// (static reload + discovery + merge), not discovery success. Stub fetch so
// the online branch runs identically to production-with-no-reachable-providers
// without paying real network latency (~400ms of DNS/socket time otherwise).
fetchSpy = spyOn(globalThis, "fetch").mockRejectedValue(new Error("network disabled in test"));
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const config: ProviderConfigInput = {
baseUrl: "https://runtime.example.com/v1",
@@ -29,7 +29,10 @@ describe("ModelRegistry", () => {
fs.mkdirSync(tempDir, { recursive: true });
modelsJsonPath = path.join(tempDir, "models.json");
cacheDbPath = path.join(tempDir, "models.db");
authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db"));
// In-memory auth DB: tests need a fresh, isolated credential store per case but
// never reopen it from disk, so :memory: avoids the WAL/chmod disk-open cost
// (~3ms/test) while preserving per-test isolation.
authStorage = await AuthStorage.create(":memory:");
});
afterEach(() => {
@@ -47,6 +47,47 @@ function createOllamaCloudModel(id: string): Model {
maxTokens: 8192,
};
}
function createContextTestModel(id: string, contextWindow: number): Model {
return {
id,
name: id,
api: "ollama-chat",
baseUrl: "https://example.com",
reasoning: false,
provider: "test",
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow,
maxTokens: 1024,
};
}
function createScopedSelector(
models: Model[],
settings: Settings,
onSelect: (model: Model) => void,
options?: { temporaryOnly?: boolean; currentContextTokens?: number },
): ModelSelectorComponent {
const modelRegistry = {
getAll: () => models,
getDiscoverableProviders: () => [],
getCanonicalModels: () => [],
resolveCanonicalModel: () => undefined,
} as unknown as ModelRegistry;
const ui = {
requestRender: vi.fn(),
} as unknown as TUI;
return new ModelSelectorComponent(
ui,
undefined,
settings,
modelRegistry,
models.map(model => ({ model })),
model => onSelect(model),
() => {},
options,
);
}
let testTheme = await getThemeByName("dark");
function installTestTheme(): void {
@@ -96,6 +137,48 @@ describe("ModelSelector role badge thinking display", () => {
expect(menuRendered).toContain("Set as SMOL (Quick)");
});
test("dims and disables models below the current context size", async () => {
installTestTheme();
const settings = Settings.isolated({});
const small = createContextTestModel("a-small", 4096);
const large = createContextTestModel("b-large", 128_000);
const selected: string[] = [];
const selector = createScopedSelector([small, large], settings, model => selected.push(model.id), {
temporaryOnly: true,
currentContextTokens: 6000,
});
await Bun.sleep(0);
installTestTheme();
const rendered = normalizeRenderedText(selector.render(220).join("\n"));
expect(rendered).toContain("a-small");
expect(rendered).toContain("context>4.1k");
selector.handleInput("\n");
expect(selected).toEqual(["b-large"]);
});
test("does not open the model menu when every candidate is disabled", async () => {
installTestTheme();
const settings = Settings.isolated({});
const small = createContextTestModel("only-small", 4096);
const onSelect = vi.fn();
const selector = createScopedSelector([small], settings, onSelect, {
currentContextTokens: 6000,
});
await Bun.sleep(0);
installTestTheme();
const rendered = normalizeRenderedText(selector.render(220).join("\n"));
expect(rendered).toContain("only-small");
expect(rendered).toContain("current context 6k > 4.1k limit");
selector.handleInput("\n");
const afterEnter = normalizeRenderedText(selector.render(220).join("\n"));
expect(afterEnter).not.toContain("Action for");
expect(onSelect).not.toHaveBeenCalled();
});
test("refreshes Ollama Cloud using provider id instead of tab label", async () => {
installTestTheme();
const settings = Settings.isolated({});
@@ -0,0 +1,143 @@
import { beforeEach, describe, expect, it } from "bun:test";
import { ChatBlock, type ChatBlockHost } from "@oh-my-pi/pi-coding-agent/modes/components/chat-block";
import type { Component } from "@oh-my-pi/pi-tui";
/** Concrete subclass exposing the protected lifecycle seams for assertions. */
class TestBlock extends ChatBlock {
mountCount = 0;
cleanupCount = 0;
protected override onMount(): void {
this.mountCount++;
this.onCleanup(() => {
this.cleanupCount++;
});
}
/** Public proxy for the protected requestRender. */
ping(): void {
this.requestRender();
}
/** Public proxy for the protected onCleanup. */
register(cleanup: () => void): void {
this.onCleanup(cleanup);
}
}
describe("ChatBlock lifecycle", () => {
let renders: number;
let host: ChatBlockHost;
beforeEach(() => {
renders = 0;
host = {
requestRender: () => {
renders++;
},
};
});
it("runs onMount exactly once; a second mount is a no-op", () => {
const block = new TestBlock();
expect(block.mountCount).toBe(0);
block.mount(host);
block.mount(host);
expect(block.mountCount).toBe(1);
});
it("is finalized until mounted, live while active, finalized after finish", () => {
const block = new TestBlock();
expect(block.isTranscriptBlockFinalized()).toBe(true);
block.mount(host);
expect(block.isTranscriptBlockFinalized()).toBe(false);
block.finish();
expect(block.isTranscriptBlockFinalized()).toBe(true);
});
it("finish runs cleanups once and requests one render", () => {
const block = new TestBlock();
block.mount(host);
const before = renders;
block.finish();
expect(block.cleanupCount).toBe(1);
expect(renders).toBe(before + 1);
block.finish();
expect(block.cleanupCount).toBe(1);
});
it("dispose runs cleanups once, is idempotent, and finalizes", () => {
const block = new TestBlock();
block.mount(host);
block.dispose();
expect(block.cleanupCount).toBe(1);
expect(block.isTranscriptBlockFinalized()).toBe(true);
block.dispose();
expect(block.cleanupCount).toBe(1);
});
it("finish then dispose does not double-run cleanups", () => {
const block = new TestBlock();
block.mount(host);
block.finish();
block.dispose();
expect(block.cleanupCount).toBe(1);
});
it("requestRender routes to the host only between mount and dispose", () => {
const block = new TestBlock();
block.ping();
expect(renders).toBe(0);
block.mount(host);
block.ping();
expect(renders).toBe(1);
block.dispose();
const after = renders;
block.ping();
expect(renders).toBe(after);
});
it("onCleanup registered after dispose runs immediately so callers never leak", () => {
const block = new TestBlock();
block.mount(host);
block.dispose();
let ran = false;
block.register(() => {
ran = true;
});
expect(ran).toBe(true);
});
it("dispose propagates to child components", () => {
const block = new TestBlock();
let childDisposed = 0;
const child: Component = {
render: () => [],
invalidate: () => {},
dispose: () => {
childDisposed++;
},
};
block.addChild(child);
block.mount(host);
block.dispose();
expect(childDisposed).toBe(1);
});
it("tears down a timer effect started in onMount when finished", async () => {
class TimerBlock extends ChatBlock {
protected override onMount(): void {
const id = setInterval(() => this.requestRender(), 5);
this.onCleanup(() => clearInterval(id));
}
}
const block = new TimerBlock();
block.mount(host);
await Bun.sleep(25);
expect(renders).toBeGreaterThan(0); // timer fired while active
block.finish();
const settled = renders; // includes finish()'s own render
await Bun.sleep(25);
expect(renders).toBe(settled); // interval torn down — no further ticks
});
});
@@ -0,0 +1,135 @@
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import { stripVTControlCharacters } from "node:util";
import { KeybindingsManager } from "@oh-my-pi/pi-coding-agent/config/keybindings";
import { CopySelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/copy-selector";
import { getThemeByName, setThemeInstance } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import type { CopyTarget } from "@oh-my-pi/pi-coding-agent/modes/utils/copy-targets";
import { setKeybindings } from "@oh-my-pi/pi-tui";
const UP = "\x1b[A";
const DOWN = "\x1b[B";
const ENTER = "\n";
const CANCEL = "\x07"; // ctrl+g, remapped to tui.select.cancel below
let darkTheme = await getThemeByName("dark");
// Flatten order (always expanded): msg:1, Block 1, Block 2, msg:2.
function makeRoots(): CopyTarget[] {
return [
{
id: "msg:1",
label: "Newest message",
hint: "5 lines · 2 code",
preview: "newest-preview-text",
content: "FULL_MESSAGE",
copyMessage: "Copied last message to clipboard",
children: [
{
id: "msg:1:code:0",
label: "Block 1",
hint: "ts",
language: "ts",
preview: "alpha()",
content: "BLOCK0",
copyMessage: "Copied block 1",
},
{
id: "msg:1:code:1",
label: "Block 2",
hint: "py",
language: "python",
preview: "beta()",
content: "BLOCK1",
copyMessage: "Copied block 2",
},
],
},
{
id: "msg:2",
label: "Older message",
hint: "3 lines",
preview: "older-text",
content: "OLDER",
copyMessage: "Copied message",
},
];
}
function render(component: CopySelectorComponent): string {
return stripVTControlCharacters(component.render(80).join("\n"));
}
describe("CopySelectorComponent", () => {
beforeAll(async () => {
darkTheme = await getThemeByName("dark");
if (!darkTheme) throw new Error("Failed to load dark theme");
});
beforeEach(() => {
setThemeInstance(darkTheme!);
setKeybindings(KeybindingsManager.inMemory({ "tui.select.cancel": "ctrl+g" }));
});
afterEach(() => {
setKeybindings(KeybindingsManager.inMemory());
vi.restoreAllMocks();
});
it("renders an outlined tree with code blocks nested under their message", () => {
const out = render(new CopySelectorComponent(makeRoots(), { onPick: vi.fn(), onCancel: vi.fn() }));
expect(out).toContain("┌");
expect(out).toContain("│");
expect(out).toContain("Copy to clipboard");
// Messages and their nested blocks are all visible (always expanded),
// connected with /tree-style branch glyphs.
expect(out).toContain("Newest message");
expect(out).toContain("Block 1");
expect(out).toContain("Block 2");
expect(out).toContain("Older message");
expect(out).toMatch(/[├└]/);
});
it("copies the message node itself on Enter", () => {
const onPick = vi.fn();
const component = new CopySelectorComponent(makeRoots(), { onPick, onCancel: vi.fn() });
component.handleInput(ENTER); // cursor starts on the message node
expect(onPick).toHaveBeenCalledTimes(1);
expect(onPick.mock.calls[0]![0].content).toBe("FULL_MESSAGE");
});
it("navigates into a nested code block and copies it", () => {
const onPick = vi.fn();
const component = new CopySelectorComponent(makeRoots(), { onPick, onCancel: vi.fn() });
component.handleInput(DOWN); // onto "Block 1"
component.handleInput(ENTER);
expect(onPick).toHaveBeenCalledTimes(1);
expect(onPick.mock.calls[0]![0].content).toBe("BLOCK0");
});
it("traverses past nested blocks to the older message, with the preview tracking the cursor", () => {
const component = new CopySelectorComponent(makeRoots(), { onPick: vi.fn(), onCancel: vi.fn() });
component.handleInput(DOWN); // Block 1
expect(render(component)).toContain("alpha()");
component.handleInput(DOWN); // Block 2
component.handleInput(DOWN); // Older message
expect(render(component)).toContain("older-text");
component.handleInput(UP); // back onto Block 2
expect(render(component)).toContain("beta()");
});
it("quits on the cancel key", () => {
const onCancel = vi.fn();
const component = new CopySelectorComponent(makeRoots(), { onPick: vi.fn(), onCancel });
component.handleInput(CANCEL);
expect(onCancel).toHaveBeenCalledTimes(1);
});
});
@@ -0,0 +1,447 @@
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import { stripVTControlCharacters } from "node:util";
import { KeybindingsManager } from "@oh-my-pi/pi-coding-agent/config/keybindings";
import type { HookSelectorSlider } from "@oh-my-pi/pi-coding-agent/modes/components/hook-selector";
import { PlanReviewOverlay } from "@oh-my-pi/pi-coding-agent/modes/components/plan-review-overlay";
import { getThemeByName, setThemeInstance } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import { setKeybindings } from "@oh-my-pi/pi-tui";
const UP = "\x1b[A";
const DOWN = "\x1b[B";
const LEFT = "\x1b[D";
const RIGHT = "\x1b[C";
const ENTER = "\r";
const TAB = "\t";
const SHIFT_DOWN = "\x1b[1;2B";
const CANCEL = "\x07"; // ctrl+g, remapped to tui.select.cancel below
let darkTheme = await getThemeByName("dark");
function render(component: PlanReviewOverlay): string {
return stripVTControlCharacters(component.render(80).join("\n"));
}
const APPROVAL_OPTIONS = [
"Approve and execute",
"Approve and compact context",
"Approve and keep context",
"Refine plan",
];
describe("PlanReviewOverlay", () => {
beforeAll(async () => {
darkTheme = await getThemeByName("dark");
if (!darkTheme) throw new Error("Failed to load dark theme");
});
beforeEach(() => {
setThemeInstance(darkTheme!);
setKeybindings(KeybindingsManager.inMemory({ "tui.select.cancel": "ctrl+g" }));
});
afterEach(() => {
setKeybindings(KeybindingsManager.inMemory());
vi.restoreAllMocks();
});
it("renders the plan body, prompt, options and footer inside one outlined box", () => {
const overlay = new PlanReviewOverlay(
"# My Plan\n\nstep one then step two",
{ promptTitle: "Plan mode - next step", options: APPROVAL_OPTIONS, helpText: "esc cancel" },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
const out = render(overlay);
expect(out).toContain("Plan Review");
expect(out).toContain("My Plan");
expect(out).toContain("step one then step two");
expect(out).toContain("Plan mode - next step");
for (const option of APPROVAL_OPTIONS) expect(out).toContain(option);
expect(out).toContain("esc cancel");
// Outlined like the /copy overlay.
expect(out).toContain("┌");
expect(out).toContain("│");
expect(out).toContain("└");
});
it("confirms the highlighted option on Enter", () => {
const onPick = vi.fn();
const overlay = new PlanReviewOverlay(
"plan",
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick, onCancel: vi.fn() },
);
overlay.handleInput(ENTER);
expect(onPick).toHaveBeenCalledTimes(1);
expect(onPick).toHaveBeenCalledWith("Approve and execute");
});
it("moves the option cursor with up/down and confirms the new target", () => {
const onPick = vi.fn();
const overlay = new PlanReviewOverlay(
"plan",
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick, onCancel: vi.fn() },
);
overlay.handleInput(DOWN);
overlay.handleInput(ENTER);
expect(onPick).toHaveBeenCalledWith("Approve and compact context");
onPick.mockClear();
overlay.handleInput(UP);
overlay.handleInput(ENTER);
expect(onPick).toHaveBeenCalledWith("Approve and execute");
});
it("skips disabled options and never confirms them", () => {
const onPick = vi.fn();
// Disable index 2 ("Approve and keep context").
const overlay = new PlanReviewOverlay(
"plan",
{ promptTitle: "next", options: APPROVAL_OPTIONS, disabledIndices: [2] },
{ onPick, onCancel: vi.fn() },
);
// 0 -> 1 -> (skip 2) -> 3.
overlay.handleInput(DOWN);
overlay.handleInput(DOWN);
overlay.handleInput(ENTER);
expect(onPick).toHaveBeenCalledTimes(1);
expect(onPick).toHaveBeenCalledWith("Refine plan");
});
it("cancels on the cancel key", () => {
const onCancel = vi.fn();
const overlay = new PlanReviewOverlay(
"plan",
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel },
);
overlay.handleInput(CANCEL);
expect(onCancel).toHaveBeenCalledTimes(1);
});
it("drives the model-tier slider with left/right without changing the option cursor", () => {
const changes: number[] = [];
const slider: HookSelectorSlider = {
caption: "continue with",
index: 0,
segments: [{ label: "default" }, { label: "slow", detail: "opus" }],
onChange: index => changes.push(index),
};
const onPick = vi.fn();
const overlay = new PlanReviewOverlay(
"plan",
{ promptTitle: "next", options: APPROVAL_OPTIONS, slider },
{ onPick, onCancel: vi.fn() },
);
overlay.handleInput(RIGHT);
expect(changes).toEqual([1]);
// Clamped at the right edge.
overlay.handleInput(RIGHT);
expect(changes).toEqual([1]);
overlay.handleInput(LEFT);
expect(changes).toEqual([1, 0]);
// The slider must not have moved the option cursor.
overlay.handleInput(ENTER);
expect(onPick).toHaveBeenCalledWith("Approve and execute");
});
it("invokes the external-editor callback on its key", () => {
setKeybindings(KeybindingsManager.inMemory({ "tui.select.cancel": "ctrl+g", "app.editor.external": "ctrl+e" }));
const onExternalEditor = vi.fn();
const overlay = new PlanReviewOverlay(
"plan",
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn(), onExternalEditor },
);
overlay.handleInput("\x05"); // ctrl+e
expect(onExternalEditor).toHaveBeenCalledTimes(1);
});
it("scrolls a long plan to bottom and back to top", () => {
const longPlan = Array.from({ length: 200 }, (_, i) => `para ${i}`).join("\n\n");
const overlay = new PlanReviewOverlay(
longPlan,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
const top = render(overlay);
expect(top).toContain("para 0");
expect(top).not.toContain("para 199");
overlay.handleInput("G");
const bottom = render(overlay);
expect(bottom).toContain("para 199");
expect(bottom).not.toContain("para 0");
overlay.handleInput("g");
const backToTop = render(overlay);
expect(backToTop).toContain("para 0");
expect(backToTop).not.toContain("para 199");
});
it("swaps the displayed plan and resets scroll on setPlanContent", () => {
const longPlan = Array.from({ length: 200 }, (_, i) => `para ${i}`).join("\n\n");
const overlay = new PlanReviewOverlay(
longPlan,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
overlay.handleInput("G"); // scroll away from the top
overlay.setPlanContent("# Fresh plan\n\nbrand new body");
const out = render(overlay);
expect(out).toContain("Fresh plan");
expect(out).toContain("brand new body");
expect(out).not.toContain("para 199");
});
// Plan with ≥2 headings + nesting, wide enough for the sidebar at width 80.
const SECTION_PLAN =
"# Overview\n\nintro body\n\n## Goal\n\ngoal body\n\n## Steps\n\nstep body\n\n# Risks\n\nrisk body\n";
it("renders no per-line ellipsis in the plan body", () => {
const overlay = new PlanReviewOverlay(
"# Plan\n\nshort line one\n\nshort line two",
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
const body = render(overlay)
.split("\n")
.filter(line => line.includes("short line"));
expect(body.length).toBeGreaterThan(0);
for (const line of body) expect(line).not.toContain("…");
});
it("shows a header-less section sidebar and cycles focus regions with Tab", () => {
const overlay = new PlanReviewOverlay(
SECTION_PLAN,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
const out = render(overlay);
// Two-column split chrome (┬ joins the title rule over the divider) and the
// bare section list — no "Contents" label.
expect(out).toContain("┬");
expect(out).not.toContain("Contents");
expect(out).toContain("Overview");
// Tab into the ToC region surfaces its focus-specific help.
overlay.handleInput(TAB);
const tocFocused = render(overlay);
expect(tocFocused).toContain("a annotate");
expect(tocFocused).toContain("d delete");
});
it("omits the single plan-title heading from the ToC", () => {
// One shallow H1 title + two H2 sections: the title is redundant in the ToC.
const overlay = new PlanReviewOverlay(
"# Plan: build the thing\n\nintro\n\n## Design\n\nd\n\n## Rollout\n\nr\n",
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
const sidebar = render(overlay)
.split("\n")
.map(line => line.split("│")[1] ?? "")
.join("\n");
expect(sidebar).toContain("Design");
expect(sidebar).toContain("Rollout");
expect(sidebar).not.toContain("build the thing");
});
it("flows past the end of a region into the actions on Down", () => {
const overlay = new PlanReviewOverlay(
SECTION_PLAN,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
render(overlay);
overlay.handleInput(TAB); // -> toc, first section
// Walk to the last ToC entry, then one more Down drops into the actions.
for (let i = 0; i < 10; i++) overlay.handleInput(DOWN);
const out = render(overlay);
// Actions focus restores the option cursor highlight + actions help.
expect(out).toContain("⏎ confirm");
expect(out).not.toContain("a annotate");
});
it("scrolls the body exactly one line per keystroke in body focus", () => {
// Tall enough to overflow any test viewport, so the body genuinely scrolls.
const rows = Array.from({ length: 400 }, (_, i) => `L${String(i).padStart(3, "0")}`).join("\n");
const overlay = new PlanReviewOverlay(
`# Plan\n\n\`\`\`\n${rows}\n\`\`\`\n`,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
const visibleRows = (): string[] =>
render(overlay)
.split("\n")
.map(line => line.match(/L\d\d\d/)?.[0])
.filter((m): m is string => m !== undefined);
render(overlay); // first render loads the body lines into the ScrollView
overlay.handleInput(TAB); // actions -> body (no sidebar with a single heading)
// Scroll well past the heading/fence so the window is pure code rows.
for (let i = 0; i < 12; i++) overlay.handleInput(DOWN);
const before = visibleRows();
overlay.handleInput(DOWN);
const after = visibleRows();
// One-line scroll: the window advances by exactly one row.
expect(after[0]).toBe(before[1]);
});
it("jumps the body to a section when the ToC cursor moves", () => {
const overlay = new PlanReviewOverlay(
SECTION_PLAN,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
render(overlay);
overlay.handleInput(TAB); // -> toc (Overview)
overlay.handleInput(DOWN); // -> Goal, scrubbing the body to it
const body = render(overlay).split("\n").slice(1, 5).join(" ");
expect(body).toContain("Goal");
});
it("deletes the selected section and restores it with undo", () => {
const onPlanEdited = vi.fn();
const overlay = new PlanReviewOverlay(
SECTION_PLAN,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn(), onPlanEdited },
);
render(overlay);
overlay.handleInput(TAB); // -> toc (Overview)
overlay.handleInput(DOWN); // -> Goal
overlay.handleInput("d");
expect(onPlanEdited).toHaveBeenCalled();
const edited = onPlanEdited.mock.calls.at(-1)?.[0] as string;
expect(edited).not.toContain("## Goal");
expect(edited).toContain("## Steps");
expect(render(overlay)).not.toContain("goal body");
overlay.handleInput("u");
const restored = render(overlay);
expect(restored).toContain("Goal");
expect(restored).toContain("goal body");
});
it("annotates a section and emits feedback for the Refine loop", () => {
const onFeedbackChange = vi.fn();
const overlay = new PlanReviewOverlay(
SECTION_PLAN,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn(), onFeedbackChange },
);
render(overlay);
overlay.handleInput(TAB); // -> toc (Overview)
overlay.handleInput("a"); // annotate Overview
for (const ch of "needs detail") overlay.handleInput(ch);
overlay.handleInput(ENTER); // submit
const out = render(overlay);
expect(out).toContain("needs detail"); // callout in the body
expect(out).toContain("✎"); // marker in the sidebar
expect(onFeedbackChange).toHaveBeenCalled();
const feedback = onFeedbackChange.mock.calls.at(-1)?.[0] as string;
expect(feedback).toContain("Overview");
expect(feedback).toContain("needs detail");
});
// Click a rendered row. The fullscreen overlay paints from screen row 0, so a
// 1-based SGR mouse row equals the rendered-line index + 1.
const clickRow = (overlay: PlanReviewOverlay, needle: string, col = 4): boolean => {
const lines = overlay.render(80);
const row = lines.findIndex(line => stripVTControlCharacters(line).includes(needle));
if (row < 0) return false;
overlay.handleInput(`\x1b[<0;${col};${row + 1}M`);
return true;
};
it("activates an approval option on click", () => {
const onPick = vi.fn();
const overlay = new PlanReviewOverlay(
SECTION_PLAN,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick, onCancel: vi.fn() },
);
render(overlay);
expect(clickRow(overlay, "Refine plan", 10)).toBe(true);
expect(onPick).toHaveBeenCalledWith("Refine plan");
});
it("selects a ToC section on click and scrubs the body to it", () => {
const overlay = new PlanReviewOverlay(
SECTION_PLAN,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
render(overlay);
// Click the "Steps" entry in the sidebar column.
expect(clickRow(overlay, "Steps", 4)).toBe(true);
const out = render(overlay);
expect(out).toContain("a annotate"); // ToC focus
// The body scrubbed to the clicked section.
expect(out.split("\n").slice(1, 5).join(" ")).toContain("Steps");
});
it("includes deleted sections in the refinement feedback", () => {
const onFeedbackChange = vi.fn();
const overlay = new PlanReviewOverlay(
SECTION_PLAN,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn(), onPlanEdited: vi.fn(), onFeedbackChange },
);
render(overlay);
overlay.handleInput(TAB); // -> toc (Overview)
overlay.handleInput(DOWN); // -> Goal
overlay.handleInput("d"); // delete Goal
const feedback = onFeedbackChange.mock.calls.at(-1)?.[0] as string;
expect(feedback).toContain("Remove these sections:");
expect(feedback).toContain("Goal");
});
it("drives the slider with both arrows even when a sidebar is present", () => {
const changes: number[] = [];
const slider: HookSelectorSlider = {
caption: "continue with",
index: 0,
segments: [{ label: "default" }, { label: "slow", detail: "opus" }],
onChange: index => changes.push(index),
};
const overlay = new PlanReviewOverlay(
SECTION_PLAN,
{ promptTitle: "next", options: APPROVAL_OPTIONS, slider },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
render(overlay); // establish the sidebar
// The sidebar sits beside the body, not the slider, so left must still step
// the tier back instead of being stolen to focus the ToC.
overlay.handleInput(RIGHT); // 0 -> 1
overlay.handleInput(LEFT); // 1 -> 0
expect(changes).toEqual([1, 0]);
// Focus stayed on the actions region (left did not jump to the ToC).
expect(render(overlay)).not.toContain("a annotate");
});
it("fast-scrolls the body with Shift+Arrow", () => {
const rows = Array.from({ length: 400 }, (_, i) => `L${String(i).padStart(3, "0")}`).join("\n");
const overlay = new PlanReviewOverlay(
`# Plan\n\n\`\`\`\n${rows}\n\`\`\`\n`,
{ promptTitle: "next", options: APPROVAL_OPTIONS },
{ onPick: vi.fn(), onCancel: vi.fn() },
);
const firstRow = (): number =>
Number(
(
render(overlay)
.split("\n")
.map(line => line.match(/L\d\d\d/)?.[0])
.find((m): m is string => m !== undefined) ?? "L000"
).slice(1),
);
render(overlay); // first render loads the body lines into the ScrollView
overlay.handleInput(TAB); // -> body
// Scroll into the pure-code region so the leading heading rows don't skew
// the absolute row math, then compare a single step to a Shift step.
for (let i = 0; i < 10; i++) overlay.handleInput(DOWN);
const base = firstRow();
overlay.handleInput(SHIFT_DOWN); // Shift+Down — fastScrollLines (5) at once
expect(firstRow() - base).toBe(5);
});
});
@@ -0,0 +1,91 @@
import { describe, expect, it } from "bun:test";
import {
joinPlanSections,
type PlanSection,
parsePlanSections,
sectionDeletionSpan,
stripInlineMarkdown,
} from "@oh-my-pi/pi-coding-agent/modes/components/plan-toc";
const titles = (sections: readonly PlanSection[]): string[] => sections.map(s => s.title);
const levels = (sections: readonly PlanSection[]): number[] => sections.map(s => s.level);
describe("parsePlanSections", () => {
it("splits a preamble and one section per ATX heading, tracking depth", () => {
const sections = parsePlanSections("intro\n\n# Overview\n\nbody\n\n## Goal\n\ngoal\n\n# Risks\n\nrisk\n");
expect(levels(sections)).toEqual([0, 1, 2, 1]);
expect(titles(sections)).toEqual(["", "Overview", "Goal", "Risks"]);
// Preamble is the only level-0 section; it carries no title.
expect(sections[0]!.raw).toBe("intro\n\n");
});
it("emits no preamble section when the document opens with a heading", () => {
const sections = parsePlanSections("# Top\n\nbody\n");
expect(levels(sections)).toEqual([1]);
expect(sections[0]!.title).toBe("Top");
});
it("does not treat '#' inside fenced code blocks as a heading", () => {
const sections = parsePlanSections("# Real\n\n```\n# not a heading\n```\n\n~~~\n## also not\n~~~\n");
expect(levels(sections)).toEqual([1]);
expect(titles(sections)).toEqual(["Real"]);
});
it("requires whitespace after the hashes, so '#tag' is body text", () => {
const sections = parsePlanSections("#tag is not a heading\nmore body\n");
expect(levels(sections)).toEqual([0]);
});
it("strips inline markdown and closing hashes from titles", () => {
const sections = parsePlanSections("## **Goal** & [docs](http://x) ##\n\nbody\n");
expect(sections[0]!.title).toBe("Goal & docs");
});
});
describe("joinPlanSections", () => {
it("round-trips a newline-terminated document", () => {
const text = "intro\n\n# A\n\nbody a\n\n## A1\n\nnested\n\n# B\n\nbody b\n";
expect(joinPlanSections(parsePlanSections(text))).toBe(text);
});
it("guarantees a single trailing newline when the source lacks one", () => {
expect(joinPlanSections(parsePlanSections("# A\n\nbody"))).toBe("# A\n\nbody\n");
});
it("returns an empty string for an empty document", () => {
expect(joinPlanSections(parsePlanSections(""))).toBe("");
});
});
describe("sectionDeletionSpan", () => {
const sections = parsePlanSections("intro\n\n# A\n\na\n\n## A1\n\na1\n\n## A2\n\na2\n\n# B\n\nb\n");
// Indices: 0 preamble, 1 A(L1), 2 A1(L2), 3 A2(L2), 4 B(L1)
it("removes a heading together with its deeper-nested children", () => {
expect(sectionDeletionSpan(sections, 1)).toEqual([1, 2, 3]);
});
it("removes only a leaf section", () => {
expect(sectionDeletionSpan(sections, 2)).toEqual([2]);
});
it("never targets the preamble", () => {
expect(sectionDeletionSpan(sections, 0)).toEqual([]);
});
it("joining the surviving sections drops the deleted subtree", () => {
const span = new Set(sectionDeletionSpan(sections, 1));
const survivors = sections.filter((_, i) => !span.has(i));
const result = joinPlanSections(survivors);
expect(result).toContain("# B");
expect(result).not.toContain("# A");
expect(result).not.toContain("## A1");
});
});
describe("stripInlineMarkdown", () => {
it("collapses emphasis, code, links, and whitespace to readable text", () => {
expect(stripInlineMarkdown("**bold** _it_ `code` [t](u)")).toBe("bold it code t");
expect(stripInlineMarkdown("a b\tc")).toBe("a b c");
});
});
@@ -0,0 +1,52 @@
import { beforeAll, describe, expect, it } from "bun:test";
import { SessionSelectorComponent } from "../../../src/modes/components/session-selector";
import { initTheme } from "../../../src/modes/theme/theme";
import type { SessionInfo } from "../../../src/session/session-manager";
beforeAll(() => {
initTheme();
});
const THUMB = "\u2588"; // ScrollView thumb glyph
function makeSessions(count: number): SessionInfo[] {
return Array.from({ length: count }, (_, i) => ({
path: `/work/TITLE_${i}.jsonl`,
id: `id-${i}`,
cwd: "/work",
title: `TITLE_${i}`,
created: new Date("2024-01-01T00:00:00Z"),
modified: new Date("2024-01-02T00:00:00Z"),
messageCount: 1,
size: 1024,
firstMessage: `body content ${i}`,
allMessagesText: `body content ${i}`,
}));
}
function makeSelector(sessions: SessionInfo[], rows: number): SessionSelectorComponent {
return new SessionSelectorComponent(
sessions,
() => {},
() => {},
() => {},
{ getTerminalRows: () => rows },
);
}
describe("SessionSelectorComponent scrollbar", () => {
it("renders the ScrollView thumb when sessions overflow the viewport", () => {
// 50 titled sessions cannot fit a 30-row viewport, so the picker windows
// them and must surface the shared right-edge scrollbar (the /resume
// overflow path the user reported).
const out = makeSelector(makeSessions(50), 30).render(80).join("\n");
expect(out).toContain(THUMB);
// The old text position indicator must be gone.
expect(out).not.toContain("(1/50)");
});
it("omits the scrollbar when every session fits", () => {
const out = makeSelector(makeSessions(2), 40).render(80).join("\n");
expect(out).not.toContain(THUMB);
});
});
@@ -6,6 +6,7 @@ import { resetSettingsForTest, Settings } from "../../../src/config/settings";
import { AssistantMessageComponent } from "../../../src/modes/components/assistant-message";
import { TranscriptContainer } from "../../../src/modes/components/transcript-container";
import { initTheme } from "../../../src/modes/theme/theme";
import { USER_INTERRUPT_LABEL } from "../../../src/session/messages";
// Models a transcript block that re-lays-out (tool preview collapsing, assistant
// message finalizing, late async result) after it has scrolled past the live
@@ -107,16 +108,16 @@ describe("TranscriptContainer", () => {
// A newer block makes `a` non-live; it now replays its last live render.
const b = new MutableBlock(["b1"]);
container.addChild(b);
expect(container.render(40)).toEqual(["a2", "b1"]);
expect(container.render(40)).toEqual(["a2", "", "b1"]);
// A post-freeze mutation of `a` (its collapse/re-layout) is NOT reflected —
// the committed rows stay stable so no stale duplicate enters scrollback.
a.set(["a3-collapsed"]);
expect(container.render(40)).toEqual(["a2", "b1"]);
expect(container.render(40)).toEqual(["a2", "", "b1"]);
// The live block still updates freely.
b.set(["b2"]);
expect(container.render(40)).toEqual(["a2", "b2"]);
expect(container.render(40)).toEqual(["a2", "", "b2"]);
});
it("reports the live block start for native scrollback pinning (ED3-risk)", () => {
@@ -127,12 +128,12 @@ describe("TranscriptContainer", () => {
container.addChild(a);
container.addChild(b);
expect(container.render(40)).toEqual(["a1", "a2", "b1"]);
expect(container.getNativeScrollbackLiveRegionStart()).toBe(2);
expect(container.render(40)).toEqual(["a1", "a2", "", "b1"]);
expect(container.getNativeScrollbackLiveRegionStart()).toBe(3);
b.set(["b1", "b2"]);
expect(container.render(40)).toEqual(["a1", "a2", "b1", "b2"]);
expect(container.getNativeScrollbackLiveRegionStart()).toBe(2);
expect(container.render(40)).toEqual(["a1", "a2", "", "b1", "b2"]);
expect(container.getNativeScrollbackLiveRegionStart()).toBe(3);
});
it("seals the prior block at its final content when finalize+append coalesce (ED3-risk)", () => {
@@ -151,11 +152,11 @@ describe("TranscriptContainer", () => {
// The transition frame must seal `a` at its final content, not the stale
// mid-stream snapshot ("Nat") it last rendered while live.
expect(container.render(40)).toEqual(["Natives built, now...", "b1"]);
expect(container.render(40)).toEqual(["Natives built, now...", "", "b1"]);
// Once sealed, a later re-layout of `a` stays frozen until the next thaw.
a.set(["a-collapsed"]);
expect(container.render(40)).toEqual(["Natives built, now...", "b1"]);
expect(container.render(40)).toEqual(["Natives built, now...", "", "b1"]);
});
it("thaw() reconciles frozen blocks to their current state", () => {
@@ -167,10 +168,29 @@ describe("TranscriptContainer", () => {
container.addChild(b);
container.render(40);
a.set(["a-final"]);
expect(container.render(40)).toEqual(["a1", "b1"]); // frozen
expect(container.render(40)).toEqual(["a1", "", "b1"]); // frozen
container.thaw();
expect(container.render(40)).toEqual(["a-final", "b1"]); // reconciled
expect(container.render(40)).toEqual(["a-final", "", "b1"]); // reconciled
});
it("invalidate() retires frozen snapshots so resetDisplay reflects current state", () => {
// resetDisplay() (Ctrl+L, and the Ctrl+O expand path) reflows by calling
// TUI.invalidate(), which propagates to this container. That must retire the
// frozen snapshots the same way thaw() does, or a forced full replay would
// still emit the pre-mutation (e.g. collapsed) render.
riskFlag.eagerEraseScrollbackRisk = true;
const container = new TranscriptContainer();
const a = new MutableBlock(["a-collapsed"]);
const b = new MutableBlock(["b1"]);
container.addChild(a);
container.addChild(b);
container.render(40);
a.set(["a-expanded-1", "a-expanded-2"]);
expect(container.render(40)).toEqual(["a-collapsed", "", "b1"]); // frozen
container.invalidate();
expect(container.render(40)).toEqual(["a-expanded-1", "a-expanded-2", "", "b1"]);
});
it("recomputes a frozen block on a width change", () => {
@@ -182,9 +202,9 @@ describe("TranscriptContainer", () => {
container.addChild(b);
container.render(40);
a.set(["a-reflowed"]);
expect(container.render(40)).toEqual(["a1", "b1"]); // frozen at width 40
expect(container.render(40)).toEqual(["a1", "", "b1"]); // frozen at width 40
// A resize is an explicit rebuild that reconciles history, so recompute.
expect(container.render(80)).toEqual(["a-reflowed", "b1"]);
expect(container.render(80)).toEqual(["a-reflowed", "", "b1"]);
});
it("renders every block live on terminals that can rebuild history", () => {
@@ -198,7 +218,7 @@ describe("TranscriptContainer", () => {
// No freezing: a non-live block's mutation is reflected (the renderer can
// rebuild committed history on these terminals).
a.set(["a-updated"]);
expect(container.render(40)).toEqual(["a-updated", "b1"]);
expect(container.render(40)).toEqual(["a-updated", "", "b1"]);
});
it("keeps an unfinalized block live when a finalized block is appended below it (ED3-risk)", () => {
@@ -213,7 +233,7 @@ describe("TranscriptContainer", () => {
// tool while it is still streaming. The tool must NOT freeze here.
const card = new MutableBlock(["rule card"]);
container.addChild(card);
expect(container.render(40)).toEqual(["write (streaming)", "rule card"]);
expect(container.render(40)).toEqual(["write (streaming)", "", "rule card"]);
// The live region begins at the unfinalized tool, not the bottom card.
expect(container.getNativeScrollbackLiveRegionStart()).toBe(0);
@@ -221,11 +241,11 @@ describe("TranscriptContainer", () => {
// tool was kept live, its final content is reflected — the bug was it
// freezing on the streaming preview and never showing the result.
tool.finalize(["✔ write: 4 lines"]);
expect(container.render(40)).toEqual(["✔ write: 4 lines", "rule card"]);
expect(container.render(40)).toEqual(["✔ write: 4 lines", "", "rule card"]);
// Now finalized, it freezes: a later re-layout stays put until the next thaw.
tool.set(["collapsed"]);
expect(container.render(40)).toEqual(["✔ write: 4 lines", "rule card"]);
expect(container.render(40)).toEqual(["✔ write: 4 lines", "", "rule card"]);
});
it("keeps a streaming assistant live so an abort label can land after status rows below it (ED3-risk)", () => {
@@ -251,14 +271,14 @@ describe("TranscriptContainer", () => {
makeAssistantMessage({
content: [{ type: "text", text: "The config file write went through despite the interruption." }],
stopReason: "aborted",
errorMessage: "Operation aborted",
errorMessage: USER_INTERRUPT_LABEL,
}),
);
assistant.markTranscriptBlockFinalized();
const rendered = plain(container.render(80));
expect(rendered).toContain("The config file write went through despite the interruption.");
expect(rendered).toContain("Operation aborted");
expect(rendered).toContain(USER_INTERRUPT_LABEL);
expect(rendered).toContain("Copied raw SSE stream");
expect(container.getNativeScrollbackLiveRegionStart()).not.toBe(0);
});
@@ -272,18 +292,86 @@ describe("TranscriptContainer", () => {
container.addChild(sealed);
container.addChild(pending);
container.addChild(card);
expect(container.render(40)).toEqual(["done", "pending", "card"]);
expect(container.render(40)).toEqual(["done", "", "pending", "", "card"]);
// Live region starts at the pending block (offset 1), so the already-sealed
// leading block can commit while pending + card stay repaintable.
expect(container.getNativeScrollbackLiveRegionStart()).toBe(1);
expect(container.getNativeScrollbackLiveRegionStart()).toBe(2);
// The leading sealed block freezes; its re-layout is not reflected.
sealed.set(["done-collapsed"]);
expect(container.render(40)).toEqual(["done", "pending", "card"]);
expect(container.render(40)).toEqual(["done", "", "pending", "", "card"]);
// The pending block updates freely while live.
pending.finalize(["pending-final"]);
expect(container.render(40)).toEqual(["done", "pending-final", "card"]);
expect(container.getNativeScrollbackLiveRegionStart()).toBe(2);
expect(container.render(40)).toEqual(["done", "", "pending-final", "", "card"]);
expect(container.getNativeScrollbackLiveRegionStart()).toBe(4);
});
});
describe("TranscriptContainer spacing", () => {
it("inserts exactly one blank line between consecutive blocks", () => {
riskFlag.eagerEraseScrollbackRisk = false;
const container = new TranscriptContainer();
container.addChild(new MutableBlock(["a"]));
container.addChild(new MutableBlock(["b"]));
container.addChild(new MutableBlock(["c"]));
// One separator between each block; none above the first.
expect(container.render(40)).toEqual(["a", "", "b", "", "c"]);
});
it("strips a block's plain-blank top/bottom padding", () => {
riskFlag.eagerEraseScrollbackRisk = false;
const container = new TranscriptContainer();
container.addChild(new MutableBlock(["a"]));
// Leading Spacer rows + a trailing paddingY row collapse to just the body.
container.addChild(new MutableBlock(["", " ", "body", ""]));
expect(container.render(40)).toEqual(["a", "", "body"]);
});
it("preserves background-colored padding rows (block-internal design)", () => {
riskFlag.eagerEraseScrollbackRisk = false;
const bgPad = "\x1b[48;2;0;0;0m \x1b[0m";
const container = new TranscriptContainer();
container.addChild(new MutableBlock(["a"]));
// The ANSI-bearing padding row is not "plain blank", so it survives stripping.
container.addChild(new MutableBlock([bgPad, "x", bgPad]));
expect(container.render(40)).toEqual(["a", "", bgPad, "x", bgPad]);
});
it("does not double the gap when a block carries its own trailing blank", () => {
riskFlag.eagerEraseScrollbackRisk = false;
const container = new TranscriptContainer();
// The trailing blank is stripped, so only the container's separator remains.
container.addChild(new MutableBlock(["note", ""]));
container.addChild(new MutableBlock(["b"]));
expect(container.render(40)).toEqual(["note", "", "b"]);
});
it("does not inject separators within a single block's rows", () => {
riskFlag.eagerEraseScrollbackRisk = false;
const container = new TranscriptContainer();
// An IRC card / file-mention list wrapped as one block stays tight inside.
container.addChild(new MutableBlock(["header", " body1", " body2"]));
expect(container.render(40)).toEqual(["header", " body1", " body2"]);
});
it("drops a blank-only block without leaving a stray gap", () => {
riskFlag.eagerEraseScrollbackRisk = false;
const container = new TranscriptContainer();
container.addChild(new MutableBlock(["a"]));
container.addChild(new MutableBlock(["", " "]));
container.addChild(new MutableBlock(["b"]));
expect(container.render(40)).toEqual(["a", "", "b"]);
});
it("counts the separator into the committed prefix below the live region (ED3-risk)", () => {
riskFlag.eagerEraseScrollbackRisk = true;
const container = new TranscriptContainer();
// A finalized block, then a still-live block below it.
container.addChild(new MutableBlock(["a1", "a2"]));
container.addChild(new StreamingBlock(["b"]));
// Separator sits at index 2; the live block's content begins at index 3.
expect(container.render(40)).toEqual(["a1", "a2", "", "b"]);
expect(container.getNativeScrollbackLiveRegionStart()).toBe(3);
});
});
@@ -6,8 +6,9 @@ describe("buildHotkeysMarkdown", () => {
const displayStrings: Record<string, string> = {
"app.clipboard.copyLine": "Alt+Shift+L",
"app.clipboard.copyPrompt": "Ctrl+Shift+P",
"app.plan.toggle": "Alt+M",
"app.plan.toggle": "Alt+Shift+P",
"app.tools.expand": "Ctrl+O",
"app.display.reset": "Ctrl+L",
"app.interrupt": "Esc",
"app.clear": "Ctrl+C",
"app.exit": "Ctrl+D",
@@ -16,7 +17,7 @@ describe("buildHotkeysMarkdown", () => {
"app.model.cycleForward": "Ctrl+P",
"app.model.cycleBackward": "Shift+Ctrl+P",
"app.model.selectTemporary": "Ctrl+Shift+L",
"app.model.select": "Ctrl+L",
"app.model.select": "Alt+M",
"app.history.search": "Ctrl+R",
"app.thinking.toggle": "Ctrl+T",
"app.editor.external": "Ctrl+G",
@@ -35,8 +36,9 @@ describe("buildHotkeysMarkdown", () => {
expect(lines[0]).toBe("**Navigation**");
expect(markdown).toContain("| `Ctrl+Shift+P` | Copy whole prompt |");
expect(markdown).toContain("| `Ctrl+Shift+L` | Select model (temporary) |");
expect(markdown).toContain("| `Ctrl+L` | Select model (set roles) |");
expect(markdown).toContain("| `Alt+M` | Toggle plan mode |");
expect(markdown).toContain("| `Alt+M` | Select model (set roles) |");
expect(markdown).toContain("| `Ctrl+L` | Reset terminal display |");
expect(markdown).toContain("| `Alt+Shift+P` | Toggle plan mode |");
expect(markdown).toContain("| `#` | Open prompt actions |");
for (const line of lines) {
if (line.length === 0) continue;
@@ -53,6 +55,9 @@ describe("buildHotkeysMarkdown", () => {
return "";
}
if (action === "app.model.select") {
return "Alt+M";
}
if (action === "app.display.reset") {
return "Ctrl+L";
}
return "Ctrl+K";
@@ -61,6 +66,6 @@ describe("buildHotkeysMarkdown", () => {
});
expect(markdown).toContain("| `Disabled` | Select model (temporary) |");
expect(markdown).toContain("| `Ctrl+L` | Select model (set roles) |");
expect(markdown).toContain("| `Alt+M` | Select model (set roles) |");
});
});
@@ -1,53 +0,0 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { CommandController } from "@oh-my-pi/pi-coding-agent/modes/controllers/command-controller";
import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types";
import * as native from "@oh-my-pi/pi-natives";
function createController(options: { assistantText?: string; hasAssistantMessage?: boolean; handoffText?: string }) {
const showStatus = vi.fn();
const showError = vi.fn();
const ctx = {
session: {
getLastAssistantText: () => options.assistantText,
hasCopyCandidateAssistantMessage: () => options.hasAssistantMessage ?? options.assistantText !== undefined,
getLastVisibleHandoffText: () => options.handoffText,
},
showStatus,
showError,
} as unknown as InteractiveModeContext;
return { controller: new CommandController(ctx), showStatus, showError };
}
describe("/copy command", () => {
afterEach(() => {
vi.restoreAllMocks();
});
it("falls back to the fresh handoff context when no assistant message exists", () => {
const copySpy = vi.spyOn(native, "copyToClipboard").mockImplementation(() => undefined);
const { controller, showStatus, showError } = createController({
handoffText: "<handoff-context>\n## Goal\nContinue\n</handoff-context>",
});
controller.handleCopyCommand();
expect(copySpy).toHaveBeenCalledWith("<handoff-context>\n## Goal\nContinue\n</handoff-context>");
expect(showStatus).toHaveBeenCalledWith("Copied handoff context to clipboard");
expect(showError).not.toHaveBeenCalled();
});
it("does not fall back to stale handoff context after a textless assistant response", () => {
const copySpy = vi.spyOn(native, "copyToClipboard").mockImplementation(() => undefined);
const { controller, showStatus, showError } = createController({
hasAssistantMessage: true,
handoffText: "<handoff-context>\n## Goal\nContinue\n</handoff-context>",
});
controller.handleCopyCommand();
expect(copySpy).not.toHaveBeenCalled();
expect(showStatus).not.toHaveBeenCalled();
expect(showError).toHaveBeenCalledWith("No agent messages to copy yet.");
});
});
@@ -2,12 +2,12 @@
* Regression test for the abort-guard on `EventController.sendCompletionNotification`.
*
* Bug: a user Ctrl+C on the `ask` tool selector throws `ToolAbortError`,
* the turn ends with `stopReason === "aborted"`, and `handleBackgroundEvent`
* fires `sendCompletionNotification()` unconditionally. The pre-fix code
* then produced a misleading "Task complete" desktop toast for a turn that
* never actually completed. The fix mirrors the `stopReason !== "aborted"`
* pattern already used by `#currentContextTokens`, `#handleMessageEnd`, and
* the retry / TTSR / compaction skip paths in `agent-session.ts`.
* the turn ends with `stopReason === "aborted"`, and `#handleAgentEnd`
* fires `sendCompletionNotification()`. Without a guard this produced a
* misleading "Task complete" desktop toast for a turn that never actually
* completed. The fix mirrors the `stopReason !== "aborted"` pattern already
* used by `#currentContextTokens`, `#handleMessageEnd`, and the
* retry / TTSR / compaction skip paths in `agent-session.ts`.
*/
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
@@ -49,8 +49,6 @@ function makeAssistantMessage(stopReason: StopReason): AssistantMessage {
function makeContext(lastMessage: AssistantMessage | undefined): InteractiveModeContext {
return {
// sendCompletionNotification only fires when backgrounded.
isBackgrounded: true,
sessionManager: {
getSessionName: () => "test-session",
},
@@ -96,16 +94,6 @@ describe("EventController.sendCompletionNotification — abort guard", () => {
expect(spy).toHaveBeenCalledTimes(1);
});
it("honors the existing isBackgrounded gate (no notification when foreground)", () => {
const spy = vi.spyOn(TERMINAL, "sendNotification").mockImplementation(() => {});
settings.override("completion.notify", "on");
const ctx = makeContext(makeAssistantMessage("stop"));
(ctx as unknown as { isBackgrounded: boolean }).isBackgrounded = false;
const controller = new EventController(ctx);
controller.sendCompletionNotification();
expect(spy).toHaveBeenCalledTimes(0);
});
it("honors the existing completion.notify=off gate", () => {
const spy = vi.spyOn(TERMINAL, "sendNotification").mockImplementation(() => {});
settings.override("completion.notify", "off");
@@ -48,7 +48,6 @@ describe("EventController idle compaction teardown", () => {
const runIdleCompaction = vi.fn();
const context = {
isInitialized: true,
isBackgrounded: false,
loadingAnimation: undefined,
streamingComponent: undefined,
streamingMessage: undefined,
@@ -194,11 +194,11 @@ describe("EventController IRC expiry", () => {
await controller.handleEvent({ type: "irc_message", message });
expect(chatContainer.children).toHaveLength(2);
expect(chatContainer.children).toHaveLength(1);
expect(requestRender).toHaveBeenCalledTimes(1);
vi.advanceTimersByTime(9_999);
expect(chatContainer.children).toHaveLength(2);
expect(chatContainer.children).toHaveLength(1);
vi.advanceTimersByTime(1);
expect(chatContainer.children).toHaveLength(0);
@@ -215,7 +215,7 @@ describe("EventController IRC expiry", () => {
await controller.handleEvent({ type: "irc_message", message });
expect(addMessageToChat).toHaveBeenCalledTimes(1);
expect(chatContainer.children).toHaveLength(2);
expect(chatContainer.children).toHaveLength(1);
vi.advanceTimersByTime(10_000);
expect(chatContainer.children).toHaveLength(0);
});
@@ -230,7 +230,7 @@ describe("EventController IRC expiry", () => {
controller.dispose();
vi.advanceTimersByTime(10_000);
expect(chatContainer.children).toHaveLength(2);
expect(chatContainer.children).toHaveLength(1);
expect(requestRender).toHaveBeenCalledTimes(1);
});
});
@@ -0,0 +1,147 @@
/**
* Read-group accretion across assistant completions.
*
* Reasoning models (and codex-style providers) frequently emit one `read` per
* completion as `[thinking?, toolCall]` rather than batching parallel calls.
* The transcript should still collapse an uninterrupted run of those reads into
* a single {@link ReadToolGroupComponent}; a completion that renders visible
* content (non-empty text/thinking) is the only thing that breaks the run, so a
* fresh group starts after it.
*
* Regression: every completion used to reset the active group at `message_start`,
* so consecutive single-read completions never grouped (each rendered as its own
* one-entry block).
*/
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { ReadToolGroupComponent } from "@oh-my-pi/pi-coding-agent/modes/components/read-tool-group";
import { EventController } from "@oh-my-pi/pi-coding-agent/modes/controllers/event-controller";
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types";
import type { AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { Container } from "@oh-my-pi/pi-tui";
beforeAll(async () => {
await initTheme(false, undefined, undefined, "dark", "light");
});
beforeEach(async () => {
resetSettingsForTest();
await Settings.init({ inMemory: true });
});
afterEach(() => {
resetSettingsForTest();
vi.restoreAllMocks();
});
type Block = AssistantMessage["content"][number];
function read(path: string): Block {
return { type: "toolCall", id: `read-${path}`, name: "read", arguments: { path } } as Block;
}
function thinking(text: string): Block {
return { type: "thinking", thinking: text } as Block;
}
function assistantMessage(content: Block[]): AssistantMessage {
return {
role: "assistant",
content,
api: "openai-codex-responses",
provider: "openai-codex",
model: "gpt-5.5",
stopReason: "toolUse",
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
timestamp: Date.now(),
};
}
function createFixture() {
const chatContainer = new Container();
const ctx = {
isInitialized: true,
init: vi.fn(async () => {}),
statusLine: { invalidate: vi.fn() },
updateEditorTopBorder: vi.fn(),
ui: { requestRender: vi.fn(), setEagerNativeScrollbackRebuild: vi.fn(), imageBudget: undefined },
chatContainer,
pendingTools: new Map(),
settings: { get: () => false },
toolOutputExpanded: false,
hideThinkingBlock: false,
setWorkingMessage: vi.fn(),
session: { getToolByName: () => undefined, extensionRunner: undefined },
} as unknown as InteractiveModeContext;
return { controller: new EventController(ctx), chatContainer };
}
/** Drive one assistant completion: message_start then a single full message_update. */
async function streamCompletion(controller: EventController, content: Block[]): Promise<void> {
const message = assistantMessage(content);
await controller.handleEvent({ type: "message_start", message } as AgentSessionEvent);
await controller.handleEvent({ type: "message_update", message } as AgentSessionEvent);
}
function readGroups(chatContainer: Container): ReadToolGroupComponent[] {
return chatContainer.children.filter((c): c is ReadToolGroupComponent => c instanceof ReadToolGroupComponent);
}
function header(group: ReadToolGroupComponent): string {
return Bun.stripANSI(group.render(120).join("\n")).split("\n")[0] ?? "";
}
describe("EventController read-group accretion", () => {
it("collapses a run of single-read completions into one group (mixed/empty thinking)", async () => {
const { controller, chatContainer } = createFixture();
// Mirrors the reported session: first read carries reasoning, the rest have
// empty or absent thinking. None of them should break the run.
await streamCompletion(controller, [thinking("Considering performance optimizations"), read("a.ts:180-250")]);
await streamCompletion(controller, [thinking(""), read("a.ts:1-120")]);
await streamCompletion(controller, [read("b.ts:1-220")]);
await streamCompletion(controller, [read("b.ts:450-535")]);
const groups = readGroups(chatContainer);
expect(groups.length).toBe(1);
expect(header(groups[0]!)).toContain("Read (4)");
});
it("starts a new group after a completion that renders visible reasoning", async () => {
const { controller, chatContainer } = createFixture();
await streamCompletion(controller, [read("a.ts:1-50")]);
await streamCompletion(controller, [read("a.ts:51-100")]);
// Visible reasoning is a separator: the next reads form a distinct group.
await streamCompletion(controller, [thinking("Now let me check the other files"), read("c.ts:1-40")]);
await streamCompletion(controller, [read("c.ts:41-80")]);
const groups = readGroups(chatContainer);
expect(groups.length).toBe(2);
expect(header(groups[0]!)).toContain("Read (2)");
expect(header(groups[1]!)).toContain("Read (2)");
});
it("keeps the active group repaintable until it is finalized", async () => {
const { controller, chatContainer } = createFixture();
await streamCompletion(controller, [read("a.ts:1-50")]);
const [group] = readGroups(chatContainer);
// While it is the active run the block must stay in the live region so its
// header can re-layout from `Read <path>` to `Read (N)` on risk terminals.
expect(group!.isTranscriptBlockFinalized()).toBe(false);
// A visible-reasoning completion breaks the run and finalizes the prior group.
await streamCompletion(controller, [thinking("done exploring"), read("b.ts:1-50")]);
expect(group!.isTranscriptBlockFinalized()).toBe(true);
});
});
@@ -6,11 +6,12 @@ import type { AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-
function createContext() {
const setEagerNativeScrollbackRebuild = vi.fn();
const ensureLoadingAnimation = vi.fn();
const pendingTools = new Map<string, unknown>();
const chatContainer = { addChild: vi.fn(), removeChild: vi.fn() };
const ctx = {
isInitialized: true,
isBackgrounded: false,
settings: { get: () => false },
statusLine: { invalidate: vi.fn() },
updateEditorTopBorder: vi.fn(),
pendingTools,
@@ -18,6 +19,7 @@ function createContext() {
hideThinkingBlock: false,
editor: { getText: vi.fn(() => "") },
flushPendingModelSwitch: vi.fn(),
sessionManager: { getSessionName: () => undefined },
session: {
agent: { state: { messages: [] } },
isCompacting: false,
@@ -25,8 +27,10 @@ function createContext() {
retryAttempt: 0,
},
ui: { setEagerNativeScrollbackRebuild, requestRender: vi.fn() },
clearPinnedError: vi.fn(),
ensureLoadingAnimation,
} as unknown as InteractiveModeContext;
return { ctx, pendingTools, setEagerNativeScrollbackRebuild };
return { ctx, pendingTools, setEagerNativeScrollbackRebuild, ensureLoadingAnimation };
}
// A tool_execution_update for an id that is not pending is a no-op in its handler,
@@ -67,6 +71,17 @@ describe("EventController tool render mode", () => {
vi.restoreAllMocks();
});
it("enables eager native scrollback rebuild before starting the idle Working loader", async () => {
const { ctx, ensureLoadingAnimation, setEagerNativeScrollbackRebuild } = createContext();
const controller = new EventController(ctx);
await controller.handleEvent({ type: "agent_start" } as unknown as AgentSessionEvent);
expect(setEagerNativeScrollbackRebuild).toHaveBeenCalledWith(true);
expect(setEagerNativeScrollbackRebuild.mock.invocationCallOrder[0]!).toBeLessThan(
ensureLoadingAnimation.mock.invocationCallOrder[0]!,
);
});
it("enables eager native scrollback rebuild while a foreground tool is pending", async () => {
const { ctx, pendingTools, setEagerNativeScrollbackRebuild } = createContext();
const controller = new EventController(ctx);
@@ -3,20 +3,25 @@ import { InputController } from "../../../src/modes/controllers/input-controller
import type { InteractiveModeContext } from "../../../src/modes/types";
describe("InputController tool output expansion", () => {
it("allows unknown viewport mutation when toggling tool output expansion", () => {
it("expands children and forces a full display reset to bypass frozen snapshots", () => {
const expandable = { setExpanded: vi.fn() };
const inert = { render: vi.fn(() => []) };
const requestRender = vi.fn();
const resetDisplay = vi.fn();
const ctx = {
toolOutputExpanded: false,
chatContainer: { children: [expandable, inert] },
ui: { requestRender },
ui: { requestRender, resetDisplay },
} as unknown as InteractiveModeContext;
new InputController(ctx).toggleToolOutputExpansion();
expect(ctx.toolOutputExpanded).toBe(true);
expect(expandable.setExpanded).toHaveBeenCalledWith(true);
expect(requestRender).toHaveBeenCalledWith(false, { allowUnknownViewportMutation: true });
// resetDisplay() is the only path that retires the transcript's frozen
// block snapshots and re-emits the whole transcript at its new heights.
// A plain requestRender would replay the stale (collapsed) snapshots.
expect(resetDisplay).toHaveBeenCalledTimes(1);
expect(requestRender).not.toHaveBeenCalled();
});
});
@@ -0,0 +1,219 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai";
import type { AsyncJobRegisterOptions } from "@oh-my-pi/pi-coding-agent/async/job-manager";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { TanCommandController } from "@oh-my-pi/pi-coding-agent/modes/controllers/tan-command-controller";
import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types";
import { MAIN_AGENT_ID } from "@oh-my-pi/pi-coding-agent/registry/agent-registry";
import type { CreateAgentSessionResult } from "@oh-my-pi/pi-coding-agent/sdk";
import * as sdkModule from "@oh-my-pi/pi-coding-agent/sdk";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { TempDir } from "@oh-my-pi/pi-utils";
interface CapturedJobRunContext {
jobId: string;
signal: AbortSignal;
reportProgress: (text: string, details?: Record<string, unknown>) => Promise<void>;
}
type CapturedJobRun = (ctx: CapturedJobRunContext) => Promise<string>;
const model = { provider: "anthropic", id: "claude-sonnet-4-5" } as Model;
function assistantText(text: string): AssistantMessage {
return {
role: "assistant",
content: [{ type: "text", text }],
api: "anthropic-messages",
provider: model.provider,
model: model.id,
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "stop",
timestamp: 0,
};
}
function createContext(overrides?: {
isStreaming?: boolean;
model?: Model;
agentId?: string;
register?: (run: CapturedJobRun, options?: AsyncJobRegisterOptions) => string;
}) {
const tempDir = TempDir.createSync("@omp-tan-controller-");
const parentFile = path.join(tempDir.path(), "parent.jsonl");
// The clone nests inside the parent's artifact directory, like a subagent.
const cloneFile = path.join(parentFile.slice(0, -6), "clone.jsonl");
let capturedRun: CapturedJobRun | undefined;
let capturedOptions: AsyncJobRegisterOptions | undefined;
const sequence: string[] = [];
const register = vi.fn(
(_type: "bash" | "task", _label: string, run: CapturedJobRun, options?: AsyncJobRegisterOptions): string => {
sequence.push("register");
capturedRun = run;
capturedOptions = options;
return overrides?.register ? overrides.register(run, options) : "job-123";
},
);
const session = {
isStreaming: overrides?.isStreaming ?? false,
model: overrides?.model ?? model,
asyncJobManager: { register },
sessionId: "parent-session",
configuredThinkingLevel: vi.fn(() => undefined),
systemPrompt: ["system prompt"],
getActiveToolNames: vi.fn(() => ["read", "bash"]),
modelRegistry: { authStorage: { marker: "auth" } },
getAgentId: vi.fn(() => overrides?.agentId),
sendCustomMessage: vi.fn(async () => {
sequence.push("sendCustomMessage");
}),
} as unknown as InteractiveModeContext["session"];
const sessionManager = {
getSessionFile: vi.fn(() => parentFile),
getCwd: vi.fn(() => tempDir.path()),
getSessionDir: vi.fn(() => tempDir.path()),
ensureOnDisk: vi.fn(async () => {}),
flush: vi.fn(async () => {}),
} as unknown as InteractiveModeContext["sessionManager"];
const cloneManager = {
getSessionFile: vi.fn(() => cloneFile),
} as unknown as SessionManager;
const ctx = {
session,
sessionManager,
settings: Settings.isolated({ "task.enableLsp": true }),
showStatus: vi.fn(),
showWarning: vi.fn(),
showError: vi.fn(),
rebuildChatFromMessages: vi.fn(),
} as unknown as InteractiveModeContext;
return {
tempDir,
parentFile,
cloneFile,
cloneManager,
ctx,
register,
sequence,
get capturedRun() {
return capturedRun;
},
get capturedOptions() {
return capturedOptions;
},
};
}
describe("TanCommandController", () => {
afterEach(() => {
vi.restoreAllMocks();
});
it("rejects empty work before forking", async () => {
const harness = createContext();
const forkSpy = vi.spyOn(SessionManager, "forkFrom").mockResolvedValue(harness.cloneManager);
const controller = new TanCommandController(harness.ctx);
await controller.start(" ");
expect(forkSpy).not.toHaveBeenCalled();
expect(harness.ctx.showStatus).toHaveBeenCalledWith("Usage: /tan <work>");
});
it("rejects while the parent session is streaming", async () => {
const harness = createContext({ isStreaming: true });
const forkSpy = vi.spyOn(SessionManager, "forkFrom").mockResolvedValue(harness.cloneManager);
const controller = new TanCommandController(harness.ctx);
await controller.start("check something");
expect(forkSpy).not.toHaveBeenCalled();
expect(harness.ctx.showWarning).toHaveBeenCalled();
});
it("forks with breadcrumb suppression, registers under Main, and dispatches after receiving the job id", async () => {
const harness = createContext();
const forkSpy = vi.spyOn(SessionManager, "forkFrom").mockResolvedValue(harness.cloneManager);
const controller = new TanCommandController(harness.ctx);
await controller.start("write the release note");
expect(forkSpy).toHaveBeenCalledWith(
harness.parentFile,
harness.tempDir.path(),
harness.parentFile.slice(0, -6),
undefined,
{ suppressBreadcrumb: true },
);
expect(harness.register).toHaveBeenCalledWith("task", "/tan write the release note", expect.any(Function), {
ownerId: MAIN_AGENT_ID,
});
expect(harness.capturedOptions?.ownerId).toBe(MAIN_AGENT_ID);
expect(harness.sequence).toEqual(["register", "sendCustomMessage"]);
expect(harness.ctx.session.sendCustomMessage).toHaveBeenCalledWith(
expect.objectContaining({
customType: "background-tan-dispatch",
details: { jobId: "job-123", work: "write the release note", sessionFile: harness.cloneFile },
}),
{ triggerTurn: false },
);
expect(harness.ctx.rebuildChatFromMessages).toHaveBeenCalled();
expect(harness.ctx.showStatus).toHaveBeenCalledWith("Dispatched background tan job-123");
});
it("aborts the cloned agent when the background job signal aborts", async () => {
const harness = createContext({ agentId: MAIN_AGENT_ID });
vi.spyOn(SessionManager, "forkFrom").mockResolvedValue(harness.cloneManager);
const promptStarted = Promise.withResolvers<void>();
const abortObserved = Promise.withResolvers<void>();
const clone = {
prompt: vi.fn(async () => {
promptStarted.resolve();
await abortObserved.promise;
}),
waitForIdle: vi.fn(async () => {}),
getLastAssistantMessage: vi.fn(() => assistantText("finished")),
abort: vi.fn(() => {
abortObserved.resolve();
}),
dispose: vi.fn(async () => {}),
};
const createAgentSessionSpy = vi
.spyOn(sdkModule, "createAgentSession")
.mockResolvedValue({ session: clone } as unknown as CreateAgentSessionResult);
const controller = new TanCommandController(harness.ctx);
await controller.start("follow the tangent");
const capturedRun = harness.capturedRun;
expect(capturedRun).toBeDefined();
if (!capturedRun) throw new Error("run function was not captured");
const abortController = new AbortController();
const resultPromise = capturedRun({
jobId: "job-123",
signal: abortController.signal,
reportProgress: async () => {},
});
await promptStarted.promise;
abortController.abort();
const result = await resultPromise;
expect(result).toBe("finished");
expect(clone.abort).toHaveBeenCalled();
expect(clone.dispose).toHaveBeenCalled();
expect(createAgentSessionSpy.mock.calls[0]?.[0]).toEqual(
expect.objectContaining({
providerPromptCacheKey: "parent-session",
parentTaskPrefix: expect.stringMatching(/^Tan-/) as unknown as string,
agentDisplayName: "tan",
}),
);
});
});
@@ -0,0 +1,207 @@
import { describe, expect, it } from "bun:test";
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
import {
buildCopyTargets,
type CopySource,
type CopyTarget,
extractCodeBlocks,
extractLastCommand,
extractQuoteBlocks,
} from "@oh-my-pi/pi-coding-agent/modes/utils/copy-targets";
function source(overrides: Partial<CopySource>): CopySource {
return {
messages: [],
getLastVisibleHandoffText: () => undefined,
...overrides,
};
}
function byId(targets: CopyTarget[], id: string): CopyTarget | undefined {
return targets.find(t => t.id === id);
}
function assistantText(text: string): AgentMessage {
return { role: "assistant", content: [{ type: "text", text }] } as unknown as AgentMessage;
}
function assistantCalls(toolCalls: Array<{ name: string; arguments: Record<string, unknown> }>): AgentMessage {
return {
role: "assistant",
content: toolCalls.map((tc, i) => ({ type: "toolCall", id: `tc-${i}`, name: tc.name, arguments: tc.arguments })),
} as unknown as AgentMessage;
}
describe("extractCodeBlocks", () => {
it("captures the language id and strips the trailing newline", () => {
expect(extractCodeBlocks("intro\n```ts\nconst x = 1;\n```\ntail")).toEqual([
{ lang: "ts", code: "const x = 1;" },
]);
});
it("returns blocks in document order with empty lang for bare fences", () => {
const blocks = extractCodeBlocks("```\nplain\n```\n\n```py\nprint(1)\n```");
expect(blocks.map(b => b.lang)).toEqual(["", "py"]);
expect(blocks.map(b => b.code)).toEqual(["plain", "print(1)"]);
});
});
describe("extractQuoteBlocks", () => {
it("collects a `>`-prefixed run and strips the marker plus one space", () => {
const text = "intro\n> line one\n> line two\ntail";
expect(extractQuoteBlocks(text)).toEqual([{ text: "line one\nline two" }]);
});
it("keeps bare `>` separator lines as blank lines and splits on plain text", () => {
const text = "> first\n>\n> second\n\nbreak\n> later";
expect(extractQuoteBlocks(text).map(b => b.text)).toEqual(["first\n\nsecond", "later"]);
});
it("does not treat `>` lines inside a fenced code block as a quote", () => {
const text = "> real quote\n```\n> not a quote\n```";
expect(extractQuoteBlocks(text)).toEqual([{ text: "real quote" }]);
});
});
describe("extractLastCommand", () => {
it("returns the most recent bash command, walking backwards", () => {
const messages = [
assistantCalls([{ name: "bash", arguments: { command: "echo old" } }]),
assistantCalls([{ name: "read", arguments: { path: "x" } }]),
assistantCalls([
{ name: "bash", arguments: { command: "echo a" } },
{ name: "bash", arguments: { command: "echo b" } },
]),
] as unknown as AgentMessage[];
expect(extractLastCommand(messages)).toEqual({ kind: "bash", code: "echo b", language: "bash" });
});
it("joins eval cell code and reports the cell language", () => {
const py = [
assistantCalls([
{ name: "eval", arguments: { cells: [{ language: "py", code: "print(1)" }, { code: "print(2)" }] } },
]),
] as unknown as AgentMessage[];
expect(extractLastCommand(py)).toEqual({ kind: "eval", code: "print(1)\n\nprint(2)", language: "python" });
const js = [
assistantCalls([{ name: "eval", arguments: { cells: [{ language: "js", code: "log(1)" }] } }]),
] as unknown as AgentMessage[];
expect(extractLastCommand(js)?.language).toBe("javascript");
});
});
describe("buildCopyTargets", () => {
it("lists assistant messages most-recent-first, drilling code-bearing ones", () => {
const newer = "Newer message\n```ts\nconst a = 1;\n```\nand\n```py\nprint(2)\n```";
const targets = buildCopyTargets(
source({
messages: [assistantText("Older message"), assistantText(newer)] as unknown as AgentMessage[],
}),
);
// Newest first.
expect(targets[0]?.id).toBe("msg:1");
expect(targets[0]?.label).toBe("Newer message");
expect(targets[1]?.id).toBe("msg:2");
// The newer message is itself a copy target (full text) AND a tree node
// exposing each code block as a child copy target.
const group = targets[0]!;
expect(group.content).toBe(newer);
expect(group.children?.map(c => c.label)).toEqual(["Block 1", "Block 2", "All 2 blocks"]);
expect(group.children?.[0]?.content).toBe("const a = 1;");
expect(group.children?.[0]?.language).toBe("ts"); // drives preview syntax highlighting
expect(group.children?.at(-1)?.content).toBe("const a = 1;\n\nprint(2)");
// The older, code-free message is a leaf that copies its full text.
expect(targets[1]?.children).toBeUndefined();
expect(targets[1]?.content).toBe("Older message");
});
it("exposes a single-block message as content plus one block child (no 'all')", () => {
const targets = buildCopyTargets(
source({ messages: [assistantText("Just one\n```js\nfoo();\n```")] as unknown as AgentMessage[] }),
);
const msg = byId(targets, "msg:1");
expect(msg?.content).toBe("Just one\n```js\nfoo();\n```");
expect(msg?.children?.map(c => c.label)).toEqual(["Block 1"]);
});
it("drills a quoted message into a de-prefixed quote child", () => {
const text = "Copy-paste to the other agent:\n\n> relay this\n> across agents";
const targets = buildCopyTargets(source({ messages: [assistantText(text)] as unknown as AgentMessage[] }));
const msg = byId(targets, "msg:1");
// The message node still copies the full markdown (with markers).
expect(msg?.content).toBe(text);
expect(msg?.hint).toBe("4 lines · 1 quote");
const quote = msg?.children?.find(c => c.id === "msg:1:quote:0");
expect(quote?.label).toBe("Quote 1");
// The drilled child copies the un-prefixed quote, ready to paste onward.
expect(quote?.content).toBe("relay this\nacross agents");
expect(quote?.language).toBeUndefined();
expect(quote?.copyMessage).toBe("Copied quote block 1 to clipboard");
});
it("interleaves code and quote children in document order with combined nodes", () => {
const text = "intro\n```ts\na;\n```\n> q one\n```py\nb\n```\n> q two";
const targets = buildCopyTargets(source({ messages: [assistantText(text)] as unknown as AgentMessage[] }));
const msg = byId(targets, "msg:1");
expect(msg?.children?.map(c => c.id)).toEqual([
"msg:1:code:0",
"msg:1:quote:0",
"msg:1:code:1",
"msg:1:quote:1",
"msg:1:all",
"msg:1:all-quotes",
]);
expect(msg?.hint).toBe("9 lines · 2 code · 2 quote");
expect(msg?.children?.find(c => c.id === "msg:1:all-quotes")?.content).toBe("q one\n\nq two");
});
it("skips tool-only assistant turns and non-assistant messages", () => {
const messages = [
{ role: "user", content: [{ type: "text", text: "hi" }] },
assistantCalls([{ name: "read", arguments: { path: "x" } }]),
assistantText("real answer"),
] as unknown as AgentMessage[];
const targets = buildCopyTargets(source({ messages }));
expect(targets.filter(t => t.id.startsWith("msg:")).map(t => t.label)).toEqual(["real answer"]);
});
it("falls back to handoff context only when there are no assistant messages", () => {
const withMessages = buildCopyTargets(
source({
messages: [assistantText("answer")] as unknown as AgentMessage[],
getLastVisibleHandoffText: () => "<handoff>",
}),
);
expect(byId(withMessages, "handoff")).toBeUndefined();
const fresh = buildCopyTargets(source({ getLastVisibleHandoffText: () => "<handoff>\nGoal" }));
expect(byId(fresh, "handoff")?.content).toBe("<handoff>\nGoal");
expect(byId(fresh, "handoff")?.copyMessage).toBe("Copied handoff context to clipboard");
});
it("interleaves runnable commands after the assistant message that issued them", () => {
const targets = buildCopyTargets(
source({
messages: [
assistantText("older answer"),
assistantCalls([{ name: "bash", arguments: { command: "echo old" } }]),
assistantText("newer answer"),
assistantCalls([{ name: "bash", arguments: { command: "bun check" } }]),
] as unknown as AgentMessage[],
}),
);
expect(targets.map(t => t.id)).toEqual(["msg:1", "cmd:1", "msg:2", "cmd:2"]);
const cmd = byId(targets, "cmd:1");
expect(cmd?.label).toBe("bun check");
expect(cmd?.hint).toBe("bash · 1 line");
expect(cmd?.content).toBe("bun check");
expect(cmd?.language).toBe("bash");
expect(byId(targets, "cmd:2")?.content).toBe("echo old");
});
});
@@ -65,6 +65,7 @@ function makeCtx(sessionManager?: Pick<SessionManager, "buildSessionContext" | "
renderSessionContext: renderSessionContextSpy,
showStatus: vi.fn(),
ui: { requestRender: vi.fn() },
resetTranscript: () => ctx.chatContainer.clear(),
} as unknown as InteractiveModeContext;
return { ctx, buildSessionContextSpy, renderSessionContextSpy };
@@ -35,7 +35,7 @@ const server = Bun.serve({
process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT = `http://localhost:${server.port}/v1/traces`;
process.env.OTEL_SERVICE_NAME = "oh-my-pi-export-probe";
initTelemetryExport();
await initTelemetryExport();
if (!isTelemetryExportEnabled()) {
console.error("PROBE: provider did not register");
await server.stop(true);
@@ -6,7 +6,7 @@
* calls resolveModelRoleValue() but only returns .model, dropping the thinking level.
* #applyPlanModeModel() therefore has no thinking level to apply.
*/
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test";
import * as path from "node:path";
import { Agent, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
@@ -22,7 +22,7 @@ describe("plan mode thinking level", () => {
let modelRegistry: ModelRegistry;
let authStorage: AuthStorage;
beforeEach(async () => {
beforeAll(async () => {
tempDir = TempDir.createSync("@pi-plan-thinking-");
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
authStorage.setRuntimeApiKey("anthropic", "test-key");
@@ -33,6 +33,9 @@ describe("plan mode thinking level", () => {
if (session) {
await session.dispose();
}
});
afterAll(() => {
authStorage.close();
tempDir.removeSync();
});
@@ -1,53 +1,71 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { describe, expect, it } from "bun:test";
import {
humanizePlanTitle,
normalizePlanTitle,
renameApprovedPlanFile,
planFileUrlForSlug,
resolveApprovedPlan,
resolvePlanTitle,
} from "@oh-my-pi/pi-coding-agent/plan-mode/approved-plan";
describe("renameApprovedPlanFile", () => {
let tmpDir: string;
let artifactsDir: string;
beforeEach(async () => {
tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "approved-plan-"));
artifactsDir = path.join(tmpDir, "artifacts");
await fs.mkdir(path.join(artifactsDir, "local"), { recursive: true });
describe("planFileUrlForSlug", () => {
it("maps a slug to its local plan URL", () => {
expect(planFileUrlForSlug("auth-refactor")).toBe("local://auth-refactor-plan.md");
});
});
afterEach(async () => {
await fs.rm(tmpDir, { recursive: true, force: true });
});
function options(planFilePath: string, finalPlanFilePath: string) {
return {
planFilePath,
finalPlanFilePath,
getArtifactsDir: () => artifactsDir,
getSessionId: () => "session-z",
};
describe("resolveApprovedPlan", () => {
/** A `readPlan` backed by an in-memory map of `local://` URL → content. */
function reader(files: Record<string, string>) {
return async (url: string) => (url in files ? files[url] : null);
}
it("fails with actionable error when destination already exists", async () => {
await Bun.write(path.join(artifactsDir, "local", "PLAN.md"), "draft");
await Bun.write(path.join(artifactsDir, "local", "WP_MIGRATION_PLAN.md"), "existing");
await expect(renameApprovedPlanFile(options("local://PLAN.md", "local://WP_MIGRATION_PLAN.md"))).rejects.toThrow(
"Plan destination already exists at local://WP_MIGRATION_PLAN.md",
);
it("locates the plan from the supplied title's slug — no rename", async () => {
const result = await resolveApprovedPlan({
suppliedTitle: "auth-refactor",
statePlanFilePath: "local://PLAN.md",
readPlan: reader({ "local://auth-refactor-plan.md": "# Auth refactor\n\nbody" }),
});
expect(result.planFilePath).toBe("local://auth-refactor-plan.md");
expect(result.planContent).toContain("body");
expect(result.title).toBe("auth-refactor");
});
it("renames PLAN.md to titled artifact path", async () => {
await Bun.write(path.join(artifactsDir, "local", "PLAN.md"), "draft body");
it("strips a trailing -plan from the supplied title before reconstructing the file", async () => {
const result = await resolveApprovedPlan({
suppliedTitle: "auth-plan",
statePlanFilePath: "local://PLAN.md",
readPlan: reader({ "local://auth-plan.md": "# Auth\n\nbody" }),
});
expect(result.planFilePath).toBe("local://auth-plan.md");
});
await renameApprovedPlanFile(options("local://PLAN.md", "local://WP_MIGRATION_PLAN.md"));
it("falls back to the plan-mode state path when the slug file is absent", async () => {
const result = await resolveApprovedPlan({
suppliedTitle: "mismatch",
statePlanFilePath: "local://existing-plan.md",
readPlan: reader({ "local://existing-plan.md": "# Existing\n\nbody" }),
});
expect(result.planFilePath).toBe("local://existing-plan.md");
});
expect(await Bun.file(path.join(artifactsDir, "local", "WP_MIGRATION_PLAN.md")).text()).toBe("draft body");
await expect(fs.stat(path.join(artifactsDir, "local", "PLAN.md"))).rejects.toThrow();
it("scans listed plan files when the title was dropped and state path is empty", async () => {
const result = await resolveApprovedPlan({
suppliedTitle: undefined,
statePlanFilePath: "local://PLAN.md",
readPlan: reader({ "local://discovered-plan.md": "# Discovered\n\nbody" }),
listPlanFiles: async () => ["local://discovered-plan.md"],
});
expect(result.planFilePath).toBe("local://discovered-plan.md");
});
it("throws an actionable error when no plan file exists", async () => {
await expect(
resolveApprovedPlan({
suppliedTitle: "ghost",
statePlanFilePath: "local://PLAN.md",
readPlan: reader({}),
}),
).rejects.toThrow("Plan file not found at local://ghost-plan.md");
});
});
@@ -23,7 +23,7 @@ import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
import type { ReadToolDetails } from "@oh-my-pi/pi-coding-agent/tools/read";
import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read";
const HASHLINE_HEADER_LINE = /^¶(\S+)#([0-9A-F]{4})$/m;
const HASHLINE_HEADER_LINE = /^\[([^#\r\n]+)#([0-9A-F]{4})\]$/m;
const COLUMN_CAP = 64;
const LONG_LINE_LEN = COLUMN_CAP * 3;
@@ -32,6 +32,42 @@ describe("ReadToolGroupComponent", () => {
expect(rendered.toLowerCase()).not.toContain("ctrl+o");
});
it("uses the read-specific success mark for completed reads", () => {
const component = new ReadToolGroupComponent();
component.updateArgs({ path: "/tmp/example.ts" }, "read-success");
component.updateResult(
{
content: [{ type: "text", text: "line 1" }],
},
false,
"read-success",
);
const rendered = component.render(120).join("\n");
const plain = Bun.stripANSI(rendered);
expect(plain).toContain(themeModule.theme.status.enabled);
expect(plain).not.toContain(themeModule.theme.status.success);
expect(rendered).toContain(themeModule.theme.fg("text", themeModule.theme.status.enabled));
expect(rendered).not.toContain(themeModule.theme.fg("success", themeModule.theme.status.enabled));
});
it("omits duplicate success marks from multi-read child rows", () => {
const component = new ReadToolGroupComponent();
component.updateArgs({ path: "/tmp/one.ts" }, "read-one");
component.updateArgs({ path: "/tmp/two.ts" }, "read-two");
component.updateResult({ content: [{ type: "text", text: "one" }] }, false, "read-one");
component.updateResult({ content: [{ type: "text", text: "two" }] }, false, "read-two");
const plain = Bun.stripANSI(component.render(120).join("\n"));
expect(plain).toContain("Read (2)");
expect(plain).toContain(`${themeModule.theme.tree.branch} /tmp/one.ts`);
expect(plain).toContain(`${themeModule.theme.tree.last} /tmp/two.ts`);
expect(plain).not.toContain(`${themeModule.theme.tree.branch} ${themeModule.theme.status.enabled}`);
expect(plain).not.toContain(`${themeModule.theme.tree.last} ${themeModule.theme.status.enabled}`);
});
it("renders warning previews with warning styling instead of success styling", () => {
const component = new ReadToolGroupComponent({ showContentPreview: true });
component.updateArgs({ path: "/tmp/example.ts" }, "read-1");
@@ -43,6 +43,7 @@ describe("role thinking helper propagation", () => {
const registry = {
getAvailable: () => [model],
getApiKey: async () => "test-key",
resolver: vi.fn(() => async () => "test-key"),
};
const completeSimpleMock = vi.spyOn(ai, "completeSimple").mockResolvedValue({
stopReason: "end_turn",
@@ -66,6 +67,7 @@ describe("role thinking helper propagation", () => {
const registry = {
getAvailable: () => [model],
getApiKey: async () => "test-key",
resolver: vi.fn(() => async () => "test-key"),
};
const completeSimpleMock = vi.spyOn(ai, "completeSimple").mockResolvedValue({
stopReason: "end_turn",
@@ -1,14 +1,35 @@
import { afterEach, describe, expect, it } from "bun:test";
import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { AsyncJobManager } from "@oh-my-pi/pi-coding-agent/async/job-manager";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { Snowflake } from "@oh-my-pi/pi-utils";
describe("AsyncJobManager singleton across concurrent top-level sessions", () => {
const tempDirs: string[] = [];
// Building a ModelRegistry per session is the dominant cost here: createAgentSession
// otherwise runs discoverAuthStorage (a fresh AuthStorage DB create+reload) and a
// background online model refresh for every spawn (~450ms each). The singleton
// ownership behavior under test is independent of model resolution, so we hand every
// session one shared, network-free registry built once (~10ms/session instead).
let sharedTempDir: string;
let sharedAuthStorage: AuthStorage;
let sharedModelRegistry: ModelRegistry;
beforeAll(async () => {
sharedTempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-sdk-async-singleton-shared-"));
sharedAuthStorage = await AuthStorage.create(path.join(sharedTempDir, "auth.db"));
sharedModelRegistry = new ModelRegistry(sharedAuthStorage, path.join(sharedTempDir, "models.yml"));
});
afterAll(() => {
sharedAuthStorage.close();
fs.rmSync(sharedTempDir, { recursive: true, force: true });
});
afterEach(async () => {
for (const tempDir of tempDirs.splice(0)) {
@@ -34,6 +55,7 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () =>
slashCommands: [],
enableMCP: false,
enableLsp: false,
modelRegistry: sharedModelRegistry,
});
return session;
}
@@ -65,7 +87,7 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () =>
// Once the owning primary session disposes the singleton clears, matching
// the documented single-owner invariant.
expect(AsyncJobManager.instance()).toBeUndefined();
});
}, 60000);
it("does not cancel the primary session's running jobs when a secondary session disposes", async () => {
const primary = await spawnTopLevelSession();
@@ -106,7 +128,7 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () =>
} finally {
await primary.dispose();
}
});
}, 60000);
it("refuses async bash from a secondary session instead of routing it to the primary's manager", async () => {
const primary = await spawnTopLevelSession({ "async.enabled": true });
@@ -132,7 +154,7 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () =>
} finally {
await primary.dispose();
}
});
}, 60000);
it("clears a manager installed before a top-level session startup failure takes ownership", async () => {
const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-sdk-async-startup-failure-${Snowflake.next()}-`));
@@ -153,6 +175,7 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () =>
slashCommands: [],
enableMCP: false,
enableLsp: false,
modelRegistry: sharedModelRegistry,
systemPrompt: () => {
throw new Error("forced startup failure");
},
@@ -168,5 +191,5 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () =>
} finally {
await replacement.dispose();
}
});
}, 60000);
});
@@ -93,10 +93,17 @@ describe("createAgentSession credential_disabled subscription", () => {
cwd: dirs.cwd,
agentDir: dirs.agentDir,
authStorage,
// Pin the model registry at a temp models.json. Without an explicit path, ModelRegistry
// loads the developer's real ~/.omp models config on every construction (~100ms each,
// and non-isolated). Pointing it at the (absent) temp file keeps construction at ~2ms and
// avoids leaking host config into the test. Providing the registry also skips the
// fire-and-forget background model discovery, which is irrelevant to credential_disabled.
modelRegistry: new ModelRegistry(authStorage, path.join(dirs.agentDir, "models.json")),
settings: Settings.isolated(),
disableExtensionDiscovery: true,
extensions,
skills: [],
rules: [],
contextFiles: [],
promptTemplates: [],
workspaceTree: emptyWorkspaceTree(dirs.cwd),
@@ -406,7 +413,7 @@ describe("createAgentSession credential_disabled subscription", () => {
embedderEvents.push(event);
},
});
const modelRegistry = new ModelRegistry(authStorage);
const modelRegistry = new ModelRegistry(authStorage, path.join(dirs.agentDir, "models.json"));
const ext = makeRecordingExtension();
const { session } = await createAgentSession({
@@ -448,7 +455,7 @@ describe("createAgentSession credential_disabled subscription", () => {
const dirs = makeDirs("mismatch");
const registryStorage = await AuthStorage.create(path.join(dirs.agentDir, "agent-registry.db"));
const otherStorage = await AuthStorage.create(path.join(dirs.agentDir, "agent-other.db"));
const modelRegistry = new ModelRegistry(registryStorage);
const modelRegistry = new ModelRegistry(registryStorage, path.join(dirs.agentDir, "models-registry.json"));
await expect(
createAgentSession({
@@ -477,7 +484,7 @@ describe("createAgentSession credential_disabled subscription", () => {
// by one microtask so a sync onError() registration lands in time.
const dirs = makeDirs("error-routing");
const authStorage = await AuthStorage.create(path.join(dirs.agentDir, "agent.db"));
const modelRegistry = new ModelRegistry(authStorage);
const modelRegistry = new ModelRegistry(authStorage, path.join(dirs.agentDir, "models.json"));
try {
const throwingExtension: Extension = {
path: "test://throwing-credential-disabled",
@@ -1,4 +1,4 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
@@ -11,6 +11,7 @@ import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { Snowflake } from "@oh-my-pi/pi-utils";
import * as z from "zod/v4";
import { TOOL_DISCOVERY_AUTO_THRESHOLD } from "../src/tool-discovery/mode";
function createMcpCustomTool(name: string, serverName: string, mcpToolName: string): CustomTool {
return {
@@ -46,18 +47,34 @@ const oldSessionMtime = new Date("2000-01-01T00:00:00.000Z");
describe("createAgentSession MCP discovery prompt gating", () => {
let tempDir: string;
let registryDir: string;
let authStorage: AuthStorage;
let modelRegistry: ModelRegistry;
beforeEach(async () => {
tempDir = path.join(os.tmpdir(), `pi-sdk-mcp-discovery-${Snowflake.next()}`);
fs.mkdirSync(tempDir, { recursive: true });
authStorage = await AuthStorage.create(path.join(tempDir, "auth.db"));
// Immutable across tests: ModelRegistry's constructor eagerly loads the bundled
// model catalog (~120ms). The tests pass models explicitly and never mutate the
// registry (refreshInBackground is skipped when modelRegistry is supplied, and
// extension source sync is empty under disableExtensionDiscovery), so build it once.
beforeAll(async () => {
registryDir = path.join(os.tmpdir(), `pi-sdk-mcp-discovery-registry-${Snowflake.next()}`);
fs.mkdirSync(registryDir, { recursive: true });
authStorage = await AuthStorage.create(path.join(registryDir, "auth.db"));
modelRegistry = new ModelRegistry(authStorage);
});
afterEach(() => {
afterAll(() => {
authStorage.close();
if (registryDir && fs.existsSync(registryDir)) {
fs.rmSync(registryDir, { recursive: true, force: true });
}
});
beforeEach(() => {
tempDir = path.join(os.tmpdir(), `pi-sdk-mcp-discovery-${Snowflake.next()}`);
fs.mkdirSync(tempDir, { recursive: true });
});
afterEach(() => {
if (tempDir && fs.existsSync(tempDir)) {
fs.rmSync(tempDir, { recursive: true, force: true });
}
@@ -88,6 +105,34 @@ describe("createAgentSession MCP discovery prompt gating", () => {
);
});
it("default auto discovery hides MCP tools once the total tool set is too large", async () => {
const mcpTools = Array.from({ length: TOOL_DISCOVERY_AUTO_THRESHOLD + 1 }, (_, index) =>
createMcpCustomTool(`mcp__auto_tool_${index}`, "auto", `tool_${index}`),
);
const { session } = await createAgentSession({
cwd: tempDir,
agentDir: tempDir,
modelRegistry,
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated({}),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
customTools: mcpTools,
});
const activeNames = session.getActiveToolNames();
expect(session.isToolDiscoveryEnabled()).toBe(true);
expect(activeNames).toContain("search_tool_bm25");
expect(activeNames).not.toContain("mcp__auto_tool_0");
expect(session.getDiscoverableTools({ source: "mcp" })).toHaveLength(TOOL_DISCOVERY_AUTO_THRESHOLD + 1);
});
it("advertises discovery guidance for builtin-only tools.discoveryMode all sessions", async () => {
const { session } = await createAgentSession({
cwd: tempDir,
@@ -12,6 +12,7 @@ import { Snowflake } from "@oh-my-pi/pi-utils";
describe("createAgentSession deferred model pattern resolution", () => {
let tempDir: string;
const authStoragesToClose: AuthStorage[] = [];
beforeEach(() => {
tempDir = path.join(os.tmpdir(), `pi-sdk-model-selection-${Snowflake.next()}`);
@@ -19,6 +20,10 @@ describe("createAgentSession deferred model pattern resolution", () => {
});
afterEach(() => {
for (const authStorage of authStoragesToClose) {
authStorage.close();
}
authStoragesToClose.length = 0;
if (tempDir && fs.existsSync(tempDir)) {
fs.rmSync(tempDir, { recursive: true, force: true });
}
@@ -52,10 +57,20 @@ describe("createAgentSession deferred model pattern resolution", () => {
});
};
function buildSessionOptions(modelPattern: string) {
async function buildSessionOptions(modelPattern: string) {
// Pass an explicit ModelRegistry so createAgentSession skips its implicit
// ModelRegistry.refreshInBackground() — a network model-discovery pass
// (~250ms/session) that contributes nothing here: the model resolves from
// the inline extension provider, never from network catalogs. Mirrors the
// explicit-registry pattern the resume tests below already rely on.
const authStorage = await AuthStorage.create(path.join(tempDir, "auth.db"));
authStoragesToClose.push(authStorage);
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml"));
return {
cwd: tempDir,
agentDir: tempDir,
authStorage,
modelRegistry,
sessionManager: SessionManager.inMemory(),
disableExtensionDiscovery: true,
extensions: [providerExtension],
@@ -71,7 +86,7 @@ describe("createAgentSession deferred model pattern resolution", () => {
test("resolves explicit modelPattern after extension providers register", async () => {
const { session, modelFallbackMessage } = await createAgentSession(
buildSessionOptions("runtime-provider/runtime-model"),
await buildSessionOptions("runtime-provider/runtime-model"),
);
expect(session.model).toBeDefined();
@@ -82,7 +97,7 @@ describe("createAgentSession deferred model pattern resolution", () => {
test("does not silently fallback when explicit modelPattern is unresolved", async () => {
const { session, modelFallbackMessage } = await createAgentSession(
buildSessionOptions("missing-provider/missing-model"),
await buildSessionOptions("missing-provider/missing-model"),
);
expect(session.model).toBeUndefined();
@@ -95,7 +110,7 @@ describe("createAgentSession deferred model pattern resolution", () => {
settings.setModelRole("default", "pi/smol:high");
const { session } = await createAgentSession({
...buildSessionOptions("runtime-provider/runtime-reasoning-model"),
...(await buildSessionOptions("runtime-provider/runtime-reasoning-model")),
settings,
});
@@ -1,12 +1,14 @@
import { afterEach, describe, expect, it } from "bun:test";
import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { type AssistantMessage, getBundledModel } from "@oh-my-pi/pi-ai";
import type { Rule } from "@oh-my-pi/pi-coding-agent/capability/rule";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk";
import { SecretObfuscator } from "@oh-my-pi/pi-coding-agent/secrets";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { getSessionsDir, Snowflake } from "@oh-my-pi/pi-utils";
@@ -55,6 +57,23 @@ function getAssistantText(message: AssistantMessage | undefined): string {
describe("createAgentSession session storage isolation", () => {
const tempDirs: string[] = [];
// One shared, fully-populated (bundled models load synchronously in the
// constructor) registry for every case. Passing it via options skips the
// per-call discoverAuthStorage() SQLite open and the refreshInBackground()
// network model probe inside createAgentSession — the two real wall-clock
// sinks here. None of these cases assert on model discovery, so an
// ambient-credential-free in-memory auth store keeps them deterministic.
let sharedAuthStorage: AuthStorage;
let sharedModelRegistry: ModelRegistry;
beforeAll(async () => {
sharedAuthStorage = await AuthStorage.create(":memory:");
sharedModelRegistry = new ModelRegistry(sharedAuthStorage);
});
afterAll(() => {
sharedAuthStorage.close();
});
afterEach(async () => {
for (const tempDir of tempDirs.splice(0)) {
@@ -72,6 +91,7 @@ describe("createAgentSession session storage isolation", () => {
const { session } = await createAgentSession({
cwd,
agentDir,
modelRegistry: sharedModelRegistry,
settings: Settings.isolated(),
disableExtensionDiscovery: true,
skills: [],
@@ -105,6 +125,7 @@ describe("createAgentSession session storage isolation", () => {
const { session } = await createAgentSession({
cwd,
agentDir,
modelRegistry: sharedModelRegistry,
settings: Settings.isolated(),
rules: [rule],
disableExtensionDiscovery: true,
@@ -137,6 +158,7 @@ describe("createAgentSession session storage isolation", () => {
const commonOptions = {
cwd,
agentDir,
modelRegistry: sharedModelRegistry,
settings: Settings.isolated({ "secrets.enabled": true }),
disableExtensionDiscovery: true,
skills: [],
@@ -206,6 +228,7 @@ describe("createAgentSession session storage isolation", () => {
const { session } = await createAgentSession({
cwd,
agentDir,
modelRegistry: sharedModelRegistry,
sessionManager: resumedManager,
model,
settings: Settings.isolated({ "secrets.enabled": true }),
+27 -1
View File
@@ -1,10 +1,12 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import type { Skill } from "@oh-my-pi/pi-coding-agent/sdk";
import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { cleanupTempHome } from "./helpers/temp-home-cleanup";
@@ -24,6 +26,25 @@ describe("createAgentSession skills option", () => {
let skillsDir: string;
let tempHomeDir = "";
let originalHome: string | undefined;
// Auth storage (SQLite DB) and the model registry are immutable across these tests: skill
// discovery never touches models, and building them per test would make createAgentSession call
// modelRegistry.refreshInBackground(), whose online model discovery saturates the event loop and
// serializes the otherwise-parallel capability scans (~340ms/call). Supplying a prebuilt registry
// skips that refresh entirely (~24ms/call).
let sharedDir: string;
let sharedAuthStorage: AuthStorage;
let sharedModelRegistry: ModelRegistry;
beforeAll(async () => {
sharedDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-sdk-skills-shared-"));
sharedAuthStorage = await AuthStorage.create(path.join(sharedDir, "auth.db"));
sharedModelRegistry = new ModelRegistry(sharedAuthStorage, path.join(sharedDir, "models.yml"));
});
afterAll(() => {
sharedAuthStorage.close();
fs.rmSync(sharedDir, { recursive: true, force: true });
});
beforeEach(() => {
tempDir = path.join(os.tmpdir(), `pi-sdk-test-${Date.now()}-${Math.random().toString(36).slice(2)}`);
@@ -74,6 +95,7 @@ Loaded via symbolic link.
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
modelRegistry: sharedModelRegistry,
settings: createIsolatedSkillsSettings(),
});
@@ -87,6 +109,7 @@ Loaded via symbolic link.
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
modelRegistry: sharedModelRegistry,
settings: createIsolatedSkillsSettings(),
});
@@ -102,6 +125,7 @@ Loaded via symbolic link.
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
modelRegistry: sharedModelRegistry,
settings: createIsolatedSkillsSettings(),
});
@@ -112,6 +136,7 @@ Loaded via symbolic link.
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
modelRegistry: sharedModelRegistry,
skills: [], // Explicitly empty - like --no-skills
settings: createIsolatedSkillsSettings(),
});
@@ -135,6 +160,7 @@ Loaded via symbolic link.
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
modelRegistry: sharedModelRegistry,
skills: [customSkill],
settings: createIsolatedSkillsSettings(),
});
@@ -4,7 +4,11 @@ import * as os from "node:os";
import * as path from "node:path";
import { getBundledModel } from "@oh-my-pi/pi-ai";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { createAgentSession, type ExtensionFactory } from "@oh-my-pi/pi-coding-agent/sdk";
import {
type CreateAgentSessionOptions,
createAgentSession,
type ExtensionFactory,
} from "@oh-my-pi/pi-coding-agent/sdk";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { Snowflake } from "@oh-my-pi/pi-utils";
import * as z from "zod/v4";
@@ -34,6 +38,35 @@ const toolActivationExtension: ExtensionFactory = pi => {
describe("createAgentSession defaultInactive tool activation", () => {
const tempDirs: string[] = [];
const makeTempDir = (): string => {
const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`);
tempDirs.push(tempDir);
fs.mkdirSync(tempDir, { recursive: true });
return tempDir;
};
// Shared options for every session. `rules: []` and `workspaceTree` short-circuit
// the two slow startup scans (rule discovery + native workspace walk, ~100ms each)
// that are irrelevant to tool activation: these tests assert only which tools are
// registered/active and that tool names appear in the system prompt. Each call
// returns fresh `settings`/`sessionManager` instances to keep tests isolated.
const baseOptions = (tempDir: string): CreateAgentSessionOptions => ({
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
rules: [],
workspaceTree: { rootPath: tempDir, rendered: "", truncated: false, totalLines: 0, agentsMdFiles: [] },
});
afterEach(() => {
for (const tempDir of tempDirs.splice(0)) {
fs.rmSync(tempDir, { recursive: true, force: true });
@@ -43,24 +76,11 @@ describe("createAgentSession defaultInactive tool activation", () => {
});
it("excludes defaultInactive extension tools from the initial active set unless explicitly requested", async () => {
const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`);
tempDirs.push(tempDir);
fs.mkdirSync(tempDir, { recursive: true });
const tempDir = makeTempDir();
const { session } = await createAgentSession({
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
...baseOptions(tempDir),
extensions: [toolActivationExtension],
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
});
try {
@@ -77,24 +97,11 @@ describe("createAgentSession defaultInactive tool activation", () => {
});
it("allows explicitly requested defaultInactive extension tools into the initial active set", async () => {
const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`);
tempDirs.push(tempDir);
fs.mkdirSync(tempDir, { recursive: true });
const tempDir = makeTempDir();
const { session } = await createAgentSession({
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
...baseOptions(tempDir),
extensions: [toolActivationExtension],
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
toolNames: ["read", "default_inactive_tool"],
});
@@ -113,23 +120,10 @@ describe("createAgentSession defaultInactive tool activation", () => {
// (e.g. `["read", "search", "find", "lsp", "web_search"]`). Without this
// invariant, `yield` ended up registered but not active, and the model
// could not satisfy the idle-reminder contract that demands a `yield` call.
const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`);
tempDirs.push(tempDir);
fs.mkdirSync(tempDir, { recursive: true });
const tempDir = makeTempDir();
const { session } = await createAgentSession({
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
...baseOptions(tempDir),
requireYieldTool: true,
toolNames: ["read", "search", "find", "web_search"],
});
@@ -149,23 +143,10 @@ describe("createAgentSession defaultInactive tool activation", () => {
// the registry has no `deferrable` tool, so the previous gate dropped
// `resolve` from the registry and plan mode silently activated without
// it — leaving the agent stuck after drafting the plan.
const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`);
tempDirs.push(tempDir);
fs.mkdirSync(tempDir, { recursive: true });
const tempDir = makeTempDir();
const { session } = await createAgentSession({
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
...baseOptions(tempDir),
toolNames: ["read", "search", "find", "web_search"],
});
@@ -177,26 +158,14 @@ describe("createAgentSession defaultInactive tool activation", () => {
});
it("drops the hidden resolve tool when neither a deferrable tool nor plan mode can use it", async () => {
const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`);
tempDirs.push(tempDir);
fs.mkdirSync(tempDir, { recursive: true });
const tempDir = makeTempDir();
const settings = Settings.isolated();
settings.set("plan.enabled", false);
const { session } = await createAgentSession({
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
...baseOptions(tempDir),
settings,
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
toolNames: ["read", "search", "find", "web_search"],
});
@@ -208,23 +177,10 @@ describe("createAgentSession defaultInactive tool activation", () => {
});
it("does not register the xAI TTS tool unless enabled", async () => {
const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`);
tempDirs.push(tempDir);
fs.mkdirSync(tempDir, { recursive: true });
const tempDir = makeTempDir();
const { session } = await createAgentSession({
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
...baseOptions(tempDir),
});
try {
@@ -237,23 +193,11 @@ describe("createAgentSession defaultInactive tool activation", () => {
});
it("registers the xAI TTS tool when enabled", async () => {
const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`);
tempDirs.push(tempDir);
fs.mkdirSync(tempDir, { recursive: true });
const tempDir = makeTempDir();
const { session } = await createAgentSession({
cwd: tempDir,
agentDir: tempDir,
sessionManager: SessionManager.inMemory(),
...baseOptions(tempDir),
settings: Settings.isolated({ "tts.enabled": true }),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
});
try {
@@ -0,0 +1,87 @@
import { describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as path from "node:path";
import {
CURRENT_SESSION_VERSION,
type SessionHeader,
SessionManager,
} from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { getTerminalId } from "@oh-my-pi/pi-tui";
import { getAgentDir, getTerminalSessionsDir, setAgentDir, TempDir } from "@oh-my-pi/pi-utils";
interface JsonlMessageEntry {
type: "message";
id: string;
parentId: string | null;
timestamp: string;
message: {
role: "user";
content: string;
timestamp: number;
};
}
describe("SessionManager.forkFrom", () => {
it("suppresses terminal breadcrumbs while preserving source history under a new parented session", async () => {
using tempDir = TempDir.createSync("@omp-session-fork-");
const previousAgentDir = getAgentDir();
const previousTermSessionId = process.env.TERM_SESSION_ID;
setAgentDir(path.join(tempDir.path(), "agent"));
process.env.TERM_SESSION_ID = "omp-fork-test";
try {
const cwd = path.join(tempDir.path(), "project");
const sessionDir = path.join(tempDir.path(), "sessions");
await fs.mkdir(sessionDir, { recursive: true });
const sourceFile = path.join(sessionDir, "source.jsonl");
const timestamp = new Date().toISOString();
const sourceHeader: SessionHeader = {
type: "session",
version: CURRENT_SESSION_VERSION,
id: "source-session",
timestamp,
cwd,
};
const sourceMessage: JsonlMessageEntry = {
type: "message",
id: "message-1",
parentId: null,
timestamp,
message: { role: "user", content: "hello", timestamp: Date.now() },
};
const sourceText = `${JSON.stringify(sourceHeader)}\n${JSON.stringify(sourceMessage)}\n`;
await Bun.write(sourceFile, sourceText);
const terminalId = getTerminalId();
expect(terminalId).toBeString();
const breadcrumbFile = path.join(getTerminalSessionsDir(), terminalId ?? "missing");
await fs.rm(breadcrumbFile, { force: true });
const forked = await SessionManager.forkFrom(sourceFile, cwd, sessionDir, undefined, {
suppressBreadcrumb: true,
});
await Bun.sleep(10);
const cloneFile = forked.getSessionFile();
expect(cloneFile).toBeString();
if (!cloneFile) throw new Error("expected forked session file");
expect(await Bun.file(sourceFile).text()).toBe(sourceText);
expect(await Bun.file(breadcrumbFile).exists()).toBe(false);
expect(cloneFile).not.toBe(sourceFile);
const lines = (await Bun.file(cloneFile).text()).trim().split("\n");
const cloneHeader = JSON.parse(lines[0] ?? "{}") as SessionHeader;
const cloneMessage = JSON.parse(lines[1] ?? "{}") as JsonlMessageEntry;
expect(cloneHeader.id).not.toBe(sourceHeader.id);
expect(cloneHeader.parentSession).toBe(sourceHeader.id);
expect(cloneHeader.cwd).toBe(cwd);
expect(cloneMessage.message.content).toBe("hello");
} finally {
if (previousTermSessionId === undefined) {
delete process.env.TERM_SESSION_ID;
} else {
process.env.TERM_SESSION_ID = previousTermSessionId;
}
setAgentDir(previousAgentDir);
}
});
});
@@ -49,6 +49,15 @@ describe("setup wizard scene selection", () => {
expect(scenes.map(scene => scene.id)).toEqual(ALL_SCENES.map(scene => scene.id));
});
it("keeps CURRENT_SETUP_VERSION in sync with the highest scene minVersion", () => {
// main.ts's cold-launch gate sources CURRENT_SETUP_VERSION from the tiny
// `setup-version` module to decide whether to load the wizard at all. If a
// new scene raises the bar but the constant is not bumped, stale installs
// would never see the scene. Guard the invariant the gate relies on.
const highestMinVersion = Math.max(...ALL_SCENES.map(scene => scene.minVersion));
expect(CURRENT_SETUP_VERSION).toBe(highestMinVersion);
});
it("runs only scenes newer than the stored setup version", async () => {
const scenes = [testScene("v1-a", 1), testScene("v1-b", 1), testScene("v2", 2)];
const selected = await selectSetupScenes(1, scenes, fakeContextWithConfiguredModel(), { isTTY: true });
@@ -13,7 +13,6 @@ function createRuntime() {
editor: { setText } as unknown as InteractiveModeContext["editor"],
handleBtwCommand,
} as unknown as InteractiveModeContext,
handleBackgroundCommand: () => {},
},
};
}
@@ -20,7 +20,6 @@ function createRuntimeHarness(overrides?: { setForcedToolChoice?: (toolName: str
return {
runtime: {
ctx,
handleBackgroundCommand: () => {},
},
setForcedToolChoice,
setText,
@@ -0,0 +1,42 @@
import { describe, expect, it, vi } from "bun:test";
import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types";
import { executeBuiltinSlashCommand } from "@oh-my-pi/pi-coding-agent/slash-commands/builtin-registry";
function createRuntimeHarness(handleFreshCommand: InteractiveModeContext["handleFreshCommand"]) {
const setText = vi.fn();
return {
setText,
handleFreshCommand,
runtime: {
ctx: {
editor: { setText } as unknown as InteractiveModeContext["editor"],
handleFreshCommand,
} as InteractiveModeContext,
},
};
}
describe("/fresh slash command", () => {
it("awaits provider-state refresh before resolving", async () => {
const deferred = Promise.withResolvers<void>();
const handleFreshCommand = vi.fn(() => deferred.promise);
const harness = createRuntimeHarness(handleFreshCommand);
let settled = false;
const execution = executeBuiltinSlashCommand("/fresh", harness.runtime).then(result => {
settled = true;
return result;
});
await Promise.resolve();
expect(harness.setText).toHaveBeenCalledWith("");
expect(handleFreshCommand).toHaveBeenCalledTimes(1);
expect(settled).toBe(false);
deferred.resolve();
expect(await execution).toBe(true);
expect(settled).toBe(true);
});
});
@@ -4,7 +4,7 @@ import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/typ
import { executeBuiltinSlashCommand } from "@oh-my-pi/pi-coding-agent/slash-commands/builtin-registry";
type RuntimeHarness = {
runtime: { ctx: InteractiveModeContext; handleBackgroundCommand: () => void };
runtime: { ctx: InteractiveModeContext };
getStatus: () => string | undefined;
getWarning: () => string | undefined;
getSelectorMode: () => "login" | "logout" | undefined;
@@ -36,7 +36,6 @@ const createRuntimeHarness = (manualInput: OAuthManualInputManager): RuntimeHarn
return {
runtime: {
ctx,
handleBackgroundCommand: () => {},
},
getStatus: () => statusMessage,
getWarning: () => warningMessage,
@@ -13,7 +13,6 @@ function createRuntime() {
editor: { setText } as unknown as InteractiveModeContext["editor"],
handleOmfgCommand,
} as unknown as InteractiveModeContext,
handleBackgroundCommand: () => {},
},
};
}
@@ -30,7 +30,7 @@ function createPlanHarness(opts: { planModeEnabled: boolean; confirmExit: boolea
} as unknown as InteractiveModeContext;
return {
runtime: { ctx, handleBackgroundCommand: () => {} },
runtime: { ctx },
state,
addToHistory,
setText,
@@ -53,7 +53,7 @@ function createGoalHarness(opts: { goalModeEnabled: boolean; dropOnCall: boolean
} as unknown as InteractiveModeContext;
return {
runtime: { ctx, handleBackgroundCommand: () => {} },
runtime: { ctx },
state,
addToHistory,
setText,
@@ -16,7 +16,6 @@ function createRuntime(didRetry: boolean) {
editor: { setText } as unknown as InteractiveModeContext["editor"],
showStatus,
} as unknown as InteractiveModeContext,
handleBackgroundCommand: () => {},
},
};
}
@@ -28,7 +28,6 @@ function createRuntimeHarness(options?: {
handleSessionCommand,
handleSessionDeleteCommand,
} as InteractiveModeContext,
handleBackgroundCommand: () => {},
},
};
}
@@ -31,7 +31,6 @@ function tuiRuntime() {
handleShakeCommand,
showWarning,
} as unknown as InteractiveModeContext,
handleBackgroundCommand: vi.fn(),
};
return { handleShakeCommand, setText, showWarning, runtime };
}
@@ -5,7 +5,6 @@ import { executeBuiltinSlashCommand } from "@oh-my-pi/pi-coding-agent/slash-comm
function createRuntime() {
const showModelSelector = vi.fn();
const setText = vi.fn();
const handleBackgroundCommand = vi.fn();
return {
showModelSelector,
setText,
@@ -13,9 +12,7 @@ function createRuntime() {
ctx: {
editor: { setText } as unknown as InteractiveModeContext["editor"],
showModelSelector,
handleBackgroundCommand,
} as unknown as InteractiveModeContext,
handleBackgroundCommand,
},
};
}
@@ -0,0 +1,42 @@
import { describe, expect, it, vi } from "bun:test";
import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types";
import { executeBuiltinSlashCommand } from "@oh-my-pi/pi-coding-agent/slash-commands/builtin-registry";
function createRuntime() {
const handleTanCommand = vi.fn(async () => {});
const setText = vi.fn();
return {
handleTanCommand,
setText,
runtime: {
ctx: {
editor: { setText } as unknown as InteractiveModeContext["editor"],
handleTanCommand,
} as unknown as InteractiveModeContext,
},
};
}
describe("/tan slash command", () => {
it("routes the full work item through the tan handler", async () => {
const harness = createRuntime();
const handled = await executeBuiltinSlashCommand("/tan add a changelog note", harness.runtime);
expect(handled).toBe(true);
expect(harness.setText).toHaveBeenCalledWith("");
expect(harness.handleTanCommand).toHaveBeenCalledWith("add a changelog note");
});
it("preserves the raw multi-word suffix after /tan", async () => {
const harness = createRuntime();
const handled = await executeBuiltinSlashCommand(
"/tan investigate why prompt cache reuse matters here",
harness.runtime,
);
expect(handled).toBe(true);
expect(harness.handleTanCommand).toHaveBeenCalledWith("investigate why prompt cache reuse matters here");
});
});
@@ -0,0 +1,34 @@
import { describe, expect, it } from "bun:test";
import * as path from "node:path";
const sourceRoot = path.join(import.meta.dir, "..", "src");
describe("startup import graph", () => {
it("keeps normal startup off the aggregate modes barrel", async () => {
const mainSource = await Bun.file(path.join(sourceRoot, "main.ts")).text();
expect(mainSource).toContain('import { InteractiveMode } from "./modes/interactive-mode";');
expect(mainSource).not.toContain('from "./modes"');
});
it("keeps branch-only mode runners out of the modes barrel", async () => {
const modesBarrelSource = await Bun.file(path.join(sourceRoot, "modes/index.ts")).text();
expect(modesBarrelSource).toContain('from "./interactive-mode"');
expect(modesBarrelSource).not.toContain("runAcpMode");
expect(modesBarrelSource).not.toContain("runPrintMode");
expect(modesBarrelSource).not.toContain("runRpcMode");
expect(modesBarrelSource).not.toContain("./rpc/rpc-mode");
});
it("keeps marketplace implementation behind the lightweight auto-update starter", async () => {
const mainSource = await Bun.file(path.join(sourceRoot, "main.ts")).text();
const starterSource = await Bun.file(
path.join(sourceRoot, "extensibility/plugins/marketplace-auto-update.ts"),
).text();
expect(mainSource).toContain('from "./extensibility/plugins/marketplace-auto-update"');
expect(mainSource).not.toContain('from "./extensibility/plugins/marketplace"');
expect(starterSource).toContain('await import("./marketplace")');
});
});
@@ -15,9 +15,11 @@
* (messages.length shrinks) resets the cache.
*/
import { afterAll, beforeAll, describe, expect, it } from "bun:test";
import { countTokens } from "@oh-my-pi/pi-natives";
import { resetSettingsForTest, Settings } from "../src/config/settings";
import { StatusLineComponent } from "../src/modes/components/status-line";
import { initTheme } from "../src/modes/theme/theme";
import { computeNonMessageTokens, estimateToolSchemaTokens } from "../src/modes/utils/context-usage";
import type { AgentSession } from "../src/session/agent-session";
beforeAll(async () => {
@@ -122,6 +124,38 @@ describe("StatusLineComponent incremental context breakdown cache", () => {
expect(v3.usedTokens).toBeGreaterThan(v2.usedTokens);
});
it("non-message token shortcut matches previous category sum semantics", () => {
const session = makeSession({
messages: [],
systemPrompt: [
"You are an assistant.\n\n<skills>\n- code: Write code\n- review: Review code\n</skills>",
"Loaded context file",
"Runtime note",
],
tools: [
{
name: "bash",
description: "Run shell commands",
parameters: { type: "object", properties: { command: { type: "string" } } },
},
],
skills: [
{ name: "code", description: "Write code" },
{ name: "review", description: "Review code" },
],
});
const skillsTokens = countTokens(["code", "Write code", "review", "Review code"]);
const previousCategorySum =
Math.max(0, countTokens(session.systemPrompt?.[0] ?? "") - skillsTokens) +
countTokens((session.systemPrompt ?? []).slice(1)) +
estimateToolSchemaTokens(session.agent?.state?.tools ?? []) +
skillsTokens;
expect(new StatusLineComponent(session).getCachedContextBreakdown().usedTokens).toBe(previousCategorySum);
expect(computeNonMessageTokens(session)).toBe(previousCategorySum);
});
it("zero messages: produces only non-message tokens, no crash", () => {
const session = makeSession({ messages: [] });
const comp = new StatusLineComponent(session);

Some files were not shown because too many files have changed in this diff Show More