test(coding-agent): replaced Bun.sleep and wall-clock timing

- Replaced Bun.sleep and wall-clock timing with fake timers (vi.useFakeTimers), release gates, and deterministic polling across 15+ test files to eliminate flakiness and improve speed.
- Consolidated per-test fixture setup into beforeAll/afterAll lifecycle hooks across 20+ test files, reducing redundant initialization and improving test performance by reusing shared immutable fixtures.
- Stubbed network calls in ModelRegistry and test discovery to prevent unintended outbound requests during test execution.
- Replaced subprocess-based test coordination (file markers, Bun.sleep polling) with in-memory fakes (FakeWebSocket, FakeLspServer, VirtualClock) for deterministic, fast test execution.
This commit is contained in:
can1357
2026-06-15 11:48:44 +02:00
parent 9e65499e15
commit 6385afdfb7
28 changed files with 2901 additions and 2446 deletions
+4
View File
@@ -11,6 +11,10 @@
- Capped unexpected-stop auto-continuation to three retry attempts before giving up on repeated stops
- Updated the `edit` tool's hashline prompt, grammar, and docs to recommend the `.=` inclusive range separator (`SWAP 1.=3:`); the legacy `..` form still parses.
### Fixed
- Fixed ModelRegistry tests making outbound network calls by automatically stubbing fetch during test execution.
## [15.13.2] - 2026-06-15
### Added
@@ -59,7 +59,7 @@ import {
resolveCanonicalVariant,
resolveModelReference,
} from "@oh-my-pi/pi-catalog/identity";
import { isRecord, logger } from "@oh-my-pi/pi-utils";
import { isBunTestRuntime, isRecord, logger } from "@oh-my-pi/pi-utils";
import { parseModelString, resolveProviderModelReference } from "../config/model-resolver";
import type { AuthStorage, OAuthCredential } from "../session/auth-storage";
import { type ApiKeyResolverModel, type ApiKeyResolverOptions, createApiKeyResolver } from "./api-key-resolver";
@@ -690,7 +690,11 @@ export class ModelRegistry {
modelsPath?: string,
options?: { fetch?: FetchImpl },
) {
this.#fetch = options?.fetch ?? fetch;
this.#fetch =
options?.fetch ??
(isBunTestRuntime()
? () => Promise.reject(new Error("network disabled in model-registry runtime test"))
: fetch);
this.#modelsConfigFile = ModelsConfigFile.relocate(modelsPath);
this.#cacheDbPath = modelsPath ? path.join(path.dirname(modelsPath), "models.db") : undefined;
// Set up fallback resolver for custom provider API keys
@@ -121,6 +121,34 @@ function makeEvalSession(
return { session, sessionFile, sessionId: `${prefix}:${crypto.randomUUID()}` };
}
/**
* Spy `runSubprocess` so a `parallel()` fan-out overlaps deterministically: every
* bridge call parks until the pool saturates at `limit` concurrent calls in flight,
* then all proceed. Proves the pool reaches its ceiling without a wall-clock sleep —
* the pool itself caps how many run at once, so an unbounded pool would drive
* `maxInFlight` past `limit` and fail the bound.
*/
function spyConcurrencyBarrier(limit: number): { maxInFlight: () => number } {
let inFlight = 0;
let max = 0;
let saturate: (() => void) | undefined;
const saturated = new Promise<void>(resolve => {
saturate = resolve;
});
vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => {
inFlight++;
max = Math.max(max, inFlight);
if (inFlight >= limit) saturate?.();
try {
await saturated;
return singleResult(options, { output: options.assignment ?? "" });
} finally {
inFlight--;
}
});
return { maxInFlight: () => max };
}
describe("runEvalAgent", () => {
afterEach(() => {
vi.restoreAllMocks();
@@ -298,8 +326,17 @@ describe("runEvalAgent", () => {
});
describe("agent() through eval runtimes", () => {
// One shared JS worker backs every agent() JavaScript test below. Spawning a
// worker (thread + module-graph import) is fixed infrastructure cost, not
// behavior under test; reusing it keeps the suite fast. Each run still threads
// its own ToolSession (settings/mock are read live through the bridge per call)
// and top-level `const`/`let` are demoted to `var`, so reuse never leaks state
// these tests observe. Torn down in afterAll via disposeAllVmContexts().
const sharedJsSessionId = "agent-bridge-shared-js";
afterEach(() => {
vi.restoreAllMocks();
vi.useRealTimers();
});
afterAll(async () => {
@@ -309,7 +346,7 @@ describe("agent() through eval runtimes", () => {
it("exposes agent() in JavaScript and parses structured output", async () => {
using tempDir = TempDir.createSync("@omp-eval-agent-js-");
const { session, sessionFile, sessionId } = makeEvalSession(tempDir, "js-agent");
const { session, sessionFile } = makeEvalSession(tempDir, "js-agent");
mockAgents();
vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options =>
singleResult(options, {
@@ -319,7 +356,7 @@ describe("agent() through eval runtimes", () => {
const result = await executeJs(
'const text = await agent("hi"); const data = await agent("json", { schema: { type: "object" } }); return JSON.stringify([text, data]);',
{ cwd: tempDir.path(), sessionId, session, sessionFile },
{ cwd: tempDir.path(), sessionId: sharedJsSessionId, session, sessionFile },
);
expect(result.exitCode).toBe(0);
@@ -334,35 +371,24 @@ describe("agent() through eval runtimes", () => {
"task.enableLsp": true,
"task.maxConcurrency": 2,
});
const { session, sessionFile, sessionId } = makeEvalSession(tempDir, "js-agent-parallel", settings);
const { session, sessionFile } = makeEvalSession(tempDir, "js-agent-parallel", settings);
mockAgents();
let inFlight = 0;
let maxInFlight = 0;
vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => {
inFlight++;
maxInFlight = Math.max(maxInFlight, inFlight);
try {
await Bun.sleep(options.assignment === "a" ? 30 : 10);
return singleResult(options, { output: options.assignment ?? "" });
} finally {
inFlight--;
}
});
const barrier = spyConcurrencyBarrier(2);
const result = await executeJs(
'const values = await parallel(["a", "b", "c", "d"].map(name => () => agent(name))); return JSON.stringify(values);',
{ cwd: tempDir.path(), sessionId, session, sessionFile },
{ cwd: tempDir.path(), sessionId: sharedJsSessionId, session, sessionFile },
);
expect(result.exitCode).toBe(0);
expect(JSON.parse(result.output.trim())).toEqual(["a", "b", "c", "d"]);
expect(maxInFlight).toBeGreaterThan(1);
expect(maxInFlight).toBeLessThanOrEqual(2);
expect(barrier.maxInFlight()).toBeGreaterThan(1);
expect(barrier.maxInFlight()).toBeLessThanOrEqual(2);
});
it("propagates JavaScript parallel() rejections", async () => {
using tempDir = TempDir.createSync("@omp-eval-agent-js-reject-");
const { session, sessionFile, sessionId } = makeEvalSession(tempDir, "js-agent-reject");
const { session, sessionFile } = makeEvalSession(tempDir, "js-agent-reject");
mockAgents();
vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => {
if (options.assignment === "bad") {
@@ -373,7 +399,7 @@ describe("agent() through eval runtimes", () => {
const result = await executeJs('await parallel([() => agent("ok"), () => agent("bad")]);', {
cwd: tempDir.path(),
sessionId,
sessionId: sharedJsSessionId,
session,
sessionFile,
});
@@ -416,18 +442,7 @@ describe("agent() through eval runtimes", () => {
});
const { session, sessionFile, sessionId } = makeEvalSession(tempDir, "py-agent-parallel", settings);
mockAgents();
let inFlight = 0;
let maxInFlight = 0;
vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => {
inFlight++;
maxInFlight = Math.max(maxInFlight, inFlight);
try {
await Bun.sleep(options.assignment === "a" ? 30 : 10);
return singleResult(options, { output: options.assignment ?? "" });
} finally {
inFlight--;
}
});
const barrier = spyConcurrencyBarrier(2);
const result = await executePython(
'import json\nprint(json.dumps(parallel([lambda n=n: agent(n) for n in ["a", "b", "c", "d"]])))',
@@ -440,8 +455,8 @@ describe("agent() through eval runtimes", () => {
expect(result.exitCode).toBe(0);
expect(JSON.parse(result.output.trim())).toEqual(["a", "b", "c", "d"]);
expect(maxInFlight).toBeGreaterThan(1);
expect(maxInFlight).toBeLessThanOrEqual(2);
expect(barrier.maxInFlight()).toBeGreaterThan(1);
expect(barrier.maxInFlight()).toBeLessThanOrEqual(2);
});
it("interrupting a Python parallel() fan-out settles the kernel cleanly and preserves session state", async () => {
@@ -526,7 +541,7 @@ describe("agent() through eval runtimes", () => {
it("streams enriched agent progress through onStatus before the cell finishes", async () => {
using tempDir = TempDir.createSync("@omp-eval-agent-progress-");
const { session, sessionFile, sessionId } = makeEvalSession(tempDir, "js-agent-progress");
const { session, sessionFile } = makeEvalSession(tempDir, "js-agent-progress");
mockAgents();
const makeProgress = (options: ExecutorOptions, overrides: Partial<AgentProgress>): AgentProgress => ({
@@ -580,7 +595,7 @@ describe("agent() through eval runtimes", () => {
const events: Array<{ op: string; [key: string]: unknown }> = [];
const result = await executeJs('await agent("investigate", { label: "Scout" });', {
cwd: tempDir.path(),
sessionId,
sessionId: sharedJsSessionId,
session,
sessionFile,
onStatus: event => events.push(event),
@@ -622,16 +637,28 @@ describe("agent() through eval runtimes", () => {
mockAgents();
// runSubprocess runs far past the eval timeout budget and emits NO progress
// of its own. The bridge pause must make that delegated time invisible to
// the watchdog.
// of its own; the bridge pause must make that delegated time invisible to
// the watchdog. Fake timers replace the real wait: the subprocess parks on
// `released` so the test can advance the clock past the budget while the
// bridge call is provably in flight, then release it deterministically.
let release: (() => void) | undefined;
const released = new Promise<void>(resolve => {
release = resolve;
});
let markInFlight: (() => void) | undefined;
const inFlight = new Promise<void>(resolve => {
markInFlight = resolve;
});
vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => {
await Bun.sleep(40);
markInFlight?.();
await released;
return singleResult(options, { output: "done" });
});
const ops: string[] = [];
vi.useFakeTimers();
using idle = new IdleTimeout(20);
const result = await runEvalAgent(
const resultPromise = runEvalAgent(
{ prompt: "investigate" },
{
session,
@@ -644,11 +671,22 @@ describe("agent() through eval runtimes", () => {
},
);
// The bridge paused the watchdog; the subprocess is now blocked in flight.
await inFlight;
// Burn far more than the 20ms budget while paused: the watchdog stays armed-off.
vi.advanceTimersByTime(1_000);
expect(idle.signal.aborted).toBe(false);
release?.();
const result = await resultPromise;
expect(result.text).toBe("done");
expect(ops).toEqual([EVAL_TIMEOUT_PAUSE_OP, EVAL_TIMEOUT_RESUME_OP]);
expect(idle.signal.aborted).toBe(false);
await Bun.sleep(60);
// RESUME re-armed a fresh window; once the runtime stays idle past it the
// watchdog finally fires.
vi.advanceTimersByTime(idle.idleMs + 5);
expect(idle.signal.aborted).toBe(true);
});
@@ -657,9 +695,20 @@ describe("agent() through eval runtimes", () => {
const { session } = makeEvalSession(tempDir, "js-agent-progress-timeout-pause");
mockAgents();
// Stream frequent progress snapshots (op:"agent") for well past the budget.
// Stream frequent progress snapshots (op:"agent") well past the budget.
// They render as status, but timeout accounting is controlled only by the
// bridge pause/resume events.
// bridge pause/resume events — so even a flood of snapshots must not re-arm
// the watchdog. Fake timers make "past the budget" deterministic: the
// subprocess emits its snapshots, parks on `released`, and the test advances
// the clock far past the window before releasing it.
let release: (() => void) | undefined;
const released = new Promise<void>(resolve => {
release = resolve;
});
let markInFlight: (() => void) | undefined;
const inFlight = new Promise<void>(resolve => {
markInFlight = resolve;
});
vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => {
for (let i = 0; i < 20; i++) {
options.onProgress?.({
@@ -679,15 +728,16 @@ describe("agent() through eval runtimes", () => {
cost: 0,
durationMs: i * 10,
});
await Bun.sleep(40);
}
markInFlight?.();
await released;
return singleResult(options, { output: "done" });
});
const ops: string[] = [];
// Timing invariant (keep, do not re-tighten): total mock work (20*40ms = 800ms) > idle window (250ms) > scheduling jitter (~tens of ms).
vi.useFakeTimers();
using idle = new IdleTimeout(250);
const result = await runEvalAgent(
const resultPromise = runEvalAgent(
{ prompt: "investigate" },
{
session,
@@ -700,6 +750,16 @@ describe("agent() through eval runtimes", () => {
},
);
// All snapshots have streamed and the subprocess is blocked in flight.
await inFlight;
// Far exceed the 250ms budget while paused: the snapshots already delivered
// must not have re-armed the watchdog.
vi.advanceTimersByTime(10_000);
expect(idle.signal.aborted).toBe(false);
release?.();
const result = await resultPromise;
expect(result.text).toBe("done");
expect(ops[0]).toBe(EVAL_TIMEOUT_PAUSE_OP);
expect(ops).toContain("agent");
@@ -1,8 +1,8 @@
import { afterEach, describe, expect, it } from "bun:test";
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import { TempDir } from "@oh-my-pi/pi-utils";
import { Settings } from "../../config/settings";
import type { ToolSession } from "../../tools";
import { disposeAllVmContexts } from "../js/context-manager";
import { disposeAllVmContexts, setWorkerCloseTimeoutMsForTests } from "../js/context-manager";
import { executeJs } from "../js/executor";
const originalWorker = globalThis.Worker;
@@ -180,8 +180,18 @@ function installFakeWorker(stats: FakeWorkerStats, behavior: FakeWorkerBehavior)
}
describe("JavaScript eval worker lifecycle", () => {
let restoreCloseTimeoutMs = 0;
beforeEach(() => {
// Shrink the graceful-close grace period so the "close acked but the worker
// never exits -> force terminate" contract is proven without a real 1s wait.
restoreCloseTimeoutMs = setWorkerCloseTimeoutMsForTests(1);
});
afterEach(async () => {
// Dispose while the shrunk timeout is still active so a hung worker's afterEach
// close also force-terminates instantly, then restore the production default.
await disposeAllVmContexts();
setWorkerCloseTimeoutMsForTests(restoreCloseTimeoutMs);
Object.defineProperty(globalThis, "Worker", {
configurable: true,
writable: true,
@@ -60,6 +60,22 @@ const resettingSessions = new Map<string, Promise<void>>();
// SIGILL/SIGSEGV. Callers that pass a larger per-cell budget still dominate.
const WORKER_INIT_TIMEOUT_MS = 15_000;
const WORKER_CLOSE_TIMEOUT_MS = 1_000;
// Active graceful-close grace period before a worker that ack'd `close` but never
// emitted its `close` event is force-terminated. Defaults to the production floor;
// tests override it (and restore it) to exercise the close-timeout -> terminate
// path without a real wall-clock wait.
let workerCloseTimeoutMs: number = WORKER_CLOSE_TIMEOUT_MS;
/**
* Test-only seam: override the graceful-close grace period (ms). Returns the
* previous value so callers can restore it. Production always uses
* {@link WORKER_CLOSE_TIMEOUT_MS}; never call this outside tests.
*/
export function setWorkerCloseTimeoutMsForTests(ms: number): number {
const previous = workerCloseTimeoutMs;
workerCloseTimeoutMs = ms;
return previous;
}
export async function executeInVmContext(options: {
sessionKey: string;
@@ -492,7 +508,7 @@ function wrapBunWorker(worker: Worker): WorkerHandle {
finishIfClosed();
});
worker.addEventListener("close", onClose);
timeout = setTimeout(() => finish(false), WORKER_CLOSE_TIMEOUT_MS);
timeout = setTimeout(() => finish(false), workerCloseTimeoutMs);
worker.postMessage({ type: "close" } satisfies WorkerInbound);
return await closed;
},
@@ -557,7 +573,7 @@ function spawnInlineWorker(): WorkerHandle {
if (msg.type === "closed") finish(true);
});
this.send({ type: "close" });
timeout = setTimeout(() => finish(false), WORKER_CLOSE_TIMEOUT_MS);
timeout = setTimeout(() => finish(false), workerCloseTimeoutMs);
return await closed;
},
async terminate() {
@@ -1,16 +1,21 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai";
import { createSessionRuntime } from "@oh-my-pi/pi-coding-agent/autoresearch/state";
import { openAutoresearchStorage } from "@oh-my-pi/pi-coding-agent/autoresearch/storage";
import {
type AutoresearchStorage,
openAutoresearchStorage,
type SessionRow,
} from "@oh-my-pi/pi-coding-agent/autoresearch/storage";
import { createInitExperimentTool } from "@oh-my-pi/pi-coding-agent/autoresearch/tools/init-experiment";
import { createLogExperimentTool } from "@oh-my-pi/pi-coding-agent/autoresearch/tools/log-experiment";
import { createRunExperimentTool } from "@oh-my-pi/pi-coding-agent/autoresearch/tools/run-experiment";
import { createUpdateNotesTool } from "@oh-my-pi/pi-coding-agent/autoresearch/tools/update-notes";
import type { LogDetails, RunDetails } from "@oh-my-pi/pi-coding-agent/autoresearch/types";
import type { ASIData, LogDetails, NumericMetricMap, RunDetails } from "@oh-my-pi/pi-coding-agent/autoresearch/types";
import type { ExtensionAPI, ExtensionContext } from "@oh-my-pi/pi-coding-agent/extensibility/extensions";
import * as git from "@oh-my-pi/pi-coding-agent/utils/git";
import { Snowflake } from "@oh-my-pi/pi-utils";
import { $ } from "bun";
@@ -68,16 +73,49 @@ function createPiHarness(initialTools: string[] = []): PiHarness {
return { api, activeTools, appendEntries, setActiveToolsCalls };
}
async function initGitRepo(dir: string): Promise<{ baselineCommit: string }> {
await Bun.write(path.join(dir, "README.md"), "# baseline\n");
// One shell invocation instead of five: the git processes are unavoidable
// (config identity is read by the production tool's own commits), but
// chaining collapses the per-call Node↔shell spawn overhead.
// `git init` + identity + a baseline commit costs ~75ms; doing it once and
// filesystem-copying the resulting repo per test is ~1ms. Every test then runs
// inside a real repo, so the production tools resolve HEAD/branch from `.git` on
// disk (sub-millisecond) instead of spawning fallback git subprocesses for every
// `repo.root` / `branch.current` / `head.sha` lookup against a bare temp dir.
let templateRepo: string;
let templateBranchRepo: string;
let templateBaselineCommit: string;
beforeAll(async () => {
templateRepo = makeTempDir("pi-autoresearch-template");
await Bun.write(path.join(templateRepo, "README.md"), "# baseline\n");
await $`git init --initial-branch=main && git config user.email tester@example.com && git config user.name Tester && git add -A && git commit -m baseline`
.cwd(dir)
.cwd(templateRepo)
.quiet();
const sha = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
return { baselineCommit: sha };
templateBaselineCommit = (await $`git rev-parse HEAD`.cwd(templateRepo).text()).trim();
// Second fixture: harness committed and already on an `autoresearch/*` branch,
// the baseline for log_experiment's on-branch keep/discard scenarios.
templateBranchRepo = makeTempDir("pi-autoresearch-template-branch");
fs.cpSync(templateRepo, templateBranchRepo, { recursive: true });
await Bun.write(path.join(templateBranchRepo, "autoresearch.sh"), "#!/usr/bin/env bash\necho METRIC m=1\n");
await $`git add -A && git commit -m harness && git checkout -b autoresearch/base`.cwd(templateBranchRepo).quiet();
});
afterAll(() => {
fs.rmSync(templateRepo, { recursive: true, force: true });
fs.rmSync(templateBranchRepo, { recursive: true, force: true });
});
// Independent working copy of the template repo: baseline commit on `main`,
// committer identity configured, ready for per-test branch/commit scenarios.
function freshRepo(): { dir: string; baselineCommit: string } {
const dir = makeTempDir();
fs.cpSync(templateRepo, dir, { recursive: true });
return { dir, baselineCommit: templateBaselineCommit };
}
// Like freshRepo, but already on an `autoresearch/*` branch with the harness
// committed — the baseline for log_experiment's on-branch keep/discard paths.
function freshBranchRepo(): { dir: string } {
const dir = makeTempDir();
fs.cpSync(templateBranchRepo, dir, { recursive: true });
return { dir };
}
async function checkoutBranch(dir: string, name: string): Promise<void> {
@@ -88,6 +126,41 @@ async function writeHarnessStub(dir: string, body = "echo METRIC m=1"): Promise<
await Bun.write(path.join(dir, "autoresearch.sh"), `#!/usr/bin/env bash\n${body}\n`);
}
// Insert a completed-but-unlogged run straight into storage, mirroring what
// run_experiment persists. Tests that exercise log_experiment use this instead
// of spawning the benchmark subprocess (and run_experiment's own git status
// calls), which are incidental to the log contract under test.
function seedCompletedRun(
storage: AutoresearchStorage,
session: SessionRow,
opts: {
preRunDirtyPaths?: string[];
parsedPrimary?: number | null;
parsedMetrics?: NumericMetricMap | null;
parsedAsi?: ASIData | null;
} = {},
): void {
const now = Date.now();
const run = storage.insertRun({
sessionId: session.id,
segment: session.currentSegment,
command: "bash autoresearch.sh",
startedAt: now,
logPath: "",
preRunDirtyPaths: opts.preRunDirtyPaths ?? [],
});
storage.markRunCompleted({
runId: run.id,
completedAt: now + 1,
durationMs: 1,
exitCode: 0,
timedOut: false,
parsedPrimary: opts.parsedPrimary ?? null,
parsedMetrics: opts.parsedMetrics ?? null,
parsedAsi: opts.parsedAsi ?? null,
});
}
describe("init_experiment", () => {
let dbOverride: string;
@@ -102,7 +175,7 @@ describe("init_experiment", () => {
});
it("opens a new session and persists scope and metric metadata", async () => {
const dir = makeTempDir();
const dir = freshRepo().dir;
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const tool = createInitExperimentTool({
@@ -144,7 +217,7 @@ describe("init_experiment", () => {
});
it("updates fields without bumping segment when no new_segment flag is passed", async () => {
const dir = makeTempDir();
const dir = freshRepo().dir;
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const tool = createInitExperimentTool({
@@ -175,7 +248,7 @@ describe("init_experiment", () => {
});
it("bumps segment when new_segment is true on a re-init", async () => {
const dir = makeTempDir();
const dir = freshRepo().dir;
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const tool = createInitExperimentTool({
@@ -196,7 +269,7 @@ describe("init_experiment", () => {
});
it("rejects when autoresearch.sh is missing on first init", async () => {
const dir = makeTempDir();
const dir = freshRepo().dir;
const runtime = createSessionRuntime();
const tool = createInitExperimentTool({
dashboard: dashboardStub(),
@@ -216,8 +289,7 @@ describe("init_experiment", () => {
});
it("auto-commits pending harness changes on an autoresearch branch", async () => {
const dir = makeTempDir();
const { baselineCommit: initialBaseline } = await initGitRepo(dir);
const { dir, baselineCommit: initialBaseline } = freshRepo();
await checkoutBranch(dir, "autoresearch/setup-test");
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
@@ -234,7 +306,7 @@ describe("init_experiment", () => {
createCtx(dir),
);
expect(result.details?.harnessCommitted).toBe(true);
const newHead = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
const newHead = await git.head.sha(dir);
expect(newHead).not.toBe(initialBaseline);
expect(result.details?.baselineCommit).toBe(newHead);
const status = (await $`git status --porcelain`.cwd(dir).text()).trim();
@@ -244,8 +316,7 @@ describe("init_experiment", () => {
});
it("does not auto-commit when not on an autoresearch branch", async () => {
const dir = makeTempDir();
const { baselineCommit: initialBaseline } = await initGitRepo(dir);
const { dir, baselineCommit: initialBaseline } = freshRepo();
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const tool = createInitExperimentTool({
@@ -261,7 +332,7 @@ describe("init_experiment", () => {
createCtx(dir),
);
expect(result.details?.harnessCommitted).toBe(false);
const newHead = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
const newHead = await git.head.sha(dir);
expect(newHead).toBe(initialBaseline);
// Harness file is still in the worktree, untracked.
expect(fs.existsSync(path.join(dir, "autoresearch.sh"))).toBe(true);
@@ -282,7 +353,7 @@ describe("run_experiment", () => {
});
it("rejects when no session is active", async () => {
const dir = makeTempDir();
const dir = freshRepo().dir;
const runtime = createSessionRuntime();
const run = createRunExperimentTool({
dashboard: dashboardStub(),
@@ -294,7 +365,7 @@ describe("run_experiment", () => {
});
it("accepts arbitrary commands, parses METRIC/ASI, and stores a run", async () => {
const dir = makeTempDir();
const dir = freshRepo().dir;
await writeHarnessStub(dir, "echo METRIC runtime_ms=42; echo METRIC memory_mb=12; echo ASI hypothesis=baseline");
const runtime = createSessionRuntime();
const init = createInitExperimentTool({
@@ -320,6 +391,7 @@ describe("run_experiment", () => {
expect(details.parsedMetrics).toMatchObject({ runtime_ms: 42, memory_mb: 12 });
expect(details.parsedAsi).toMatchObject({ hypothesis: "baseline" });
expect(details.passed).toBe(true);
expect(details.command).toBe("bash autoresearch.sh");
expect(fs.existsSync(details.benchmarkLogPath)).toBe(true);
const storage = await openAutoresearchStorage(dir);
@@ -331,7 +403,7 @@ describe("run_experiment", () => {
});
it("abandons a prior pending run instead of blocking", async () => {
const dir = makeTempDir();
const dir = freshRepo().dir;
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const initTool = createInitExperimentTool({
@@ -345,33 +417,15 @@ describe("run_experiment", () => {
getRuntime: () => runtime,
pi: createPiHarness().api,
});
await run.execute("r1", {}, undefined, undefined, createCtx(dir));
// A seeded pending run stands in for the first run_experiment; the contract
// under test is that the second run abandons it rather than blocking.
const storage = await openAutoresearchStorage(dir);
seedCompletedRun(storage, storage.getActiveSession()!, { parsedPrimary: 1, parsedMetrics: { m: 1 } });
const result = await run.execute("r2", {}, undefined, undefined, createCtx(dir));
const details = result.details as RunDetails;
expect(details.abandonedPriorRun).not.toBeNull();
expect(details.runNumber).not.toBe(details.abandonedPriorRun);
});
it("runs ./autoresearch.sh and parses METRIC/ASI from its output", async () => {
const dir = makeTempDir();
await writeHarnessStub(dir, "echo METRIC m=99");
const runtime = createSessionRuntime();
const init = createInitExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: createPiHarness().api,
});
await init.execute("i", { name: "x", primary_metric: "m" }, undefined, undefined, createCtx(dir));
const run = createRunExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: createPiHarness().api,
});
const result = await run.execute("r", {}, undefined, undefined, createCtx(dir));
const details = result.details as RunDetails;
expect(details.command).toBe("bash autoresearch.sh");
expect(details.parsedPrimary).toBe(99);
});
});
describe("log_experiment", () => {
@@ -408,12 +462,12 @@ describe("log_experiment", () => {
undefined,
createCtx(dir),
);
const run = createRunExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: harness.api,
const storage = await openAutoresearchStorage(dir);
seedCompletedRun(storage, storage.getActiveSession()!, {
preRunDirtyPaths: ["autoresearch.sh"],
parsedPrimary: 10,
parsedMetrics: { runtime_ms: 10 },
});
await run.execute("r", {}, undefined, undefined, createCtx(dir));
const log = createLogExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
@@ -423,7 +477,7 @@ describe("log_experiment", () => {
}
it("rejects when no pending run exists", async () => {
const dir = makeTempDir();
const dir = freshRepo().dir;
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const harness = createPiHarness();
@@ -449,7 +503,7 @@ describe("log_experiment", () => {
});
it("stores keep with metric and updates baseline", async () => {
const dir = makeTempDir();
const dir = freshRepo().dir;
const { log, runtime } = await setupRun(dir);
const result = await log.execute(
"l",
@@ -467,8 +521,7 @@ describe("log_experiment", () => {
});
it("flags scope deviations and warns when justification is missing", async () => {
const dir = makeTempDir();
await initGitRepo(dir);
const dir = freshRepo().dir;
const { log } = await setupRun(dir);
fs.mkdirSync(path.join(dir, "forbidden"), { recursive: true });
await Bun.write(path.join(dir, "forbidden", "x.ts"), "export const v = 1;\n");
@@ -486,8 +539,7 @@ describe("log_experiment", () => {
});
it("records the justification when provided", async () => {
const dir = makeTempDir();
await initGitRepo(dir);
const dir = freshRepo().dir;
const { log } = await setupRun(dir);
fs.mkdirSync(path.join(dir, "forbidden"), { recursive: true });
await Bun.write(path.join(dir, "forbidden", "x.ts"), "export const v = 1;\n");
@@ -509,6 +561,8 @@ describe("log_experiment", () => {
});
it("flags previously logged runs via flag_runs", async () => {
// Bare temp dir (no repo): the session is created with `branch: null`, so the
// tool's branch lookup must also resolve to null to match it.
const dir = makeTempDir();
const storage = await openAutoresearchStorage(dir);
const session = storage.openSession({
@@ -595,9 +649,8 @@ describe("log_experiment", () => {
});
it("on a non-autoresearch branch, discard reverts only run-modified files", async () => {
const dir = makeTempDir();
const dir = freshRepo().dir;
await writeHarnessStub(dir);
await initGitRepo(dir);
// Commit `src/edit-me.ts` to baseline so it is tracked, not in pre-run dirty paths.
fs.mkdirSync(path.join(dir, "src"), { recursive: true });
await Bun.write(path.join(dir, "src", "edit-me.ts"), "export const v = 1;\n");
@@ -649,12 +702,7 @@ describe("log_experiment", () => {
});
it("on an autoresearch branch, discard reverts uncommitted changes but preserves prior commits", async () => {
const dir = makeTempDir();
await initGitRepo(dir);
// Commit the harness on main so it is part of the autoresearch branch's baseline.
await writeHarnessStub(dir);
await $`git add -A && git commit -m harness`.cwd(dir).quiet();
await checkoutBranch(dir, "autoresearch/test-20260501");
const dir = freshBranchRepo().dir;
const runtime = createSessionRuntime();
const harness = createPiHarness();
const init = createInitExperimentTool({
@@ -666,14 +714,12 @@ describe("log_experiment", () => {
// Simulate a previously kept iteration by committing it directly on the branch.
await Bun.write(path.join(dir, "src", "kept.ts"), "export const v = 1;\n");
await $`git add -A && git commit -m "kept iteration"`.cwd(dir).quiet();
const headBeforeDiscard = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
const headBeforeDiscard = await git.head.sha(dir);
const run = createRunExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: harness.api,
});
await run.execute("r", {}, undefined, undefined, createCtx(dir));
const storage = await openAutoresearchStorage(dir);
// On-branch discard resets to HEAD and ignores preRunDirtyPaths, so a
// seeded pending run drives log_experiment without the run subprocess.
seedCompletedRun(storage, storage.getActiveSession()!, { parsedPrimary: 1, parsedMetrics: { m: 1 } });
// Current iteration's uncommitted edits.
await Bun.write(path.join(dir, "src", "kept.ts"), "export const v = 999;\n");
await Bun.write(path.join(dir, "scratch.ts"), "// junk\n");
@@ -690,26 +736,21 @@ describe("log_experiment", () => {
undefined,
createCtx(dir),
);
const headAfter = (await $`git rev-parse HEAD`.cwd(dir).text()).trim();
const headAfter = await git.head.sha(dir);
// Prior commits survive — discard does not rewind history.
expect(headAfter).toBe(headBeforeDiscard);
// Uncommitted iteration changes are gone.
expect(fs.readFileSync(path.join(dir, "src", "kept.ts"), "utf8")).toBe("export const v = 1;\n");
expect(fs.existsSync(path.join(dir, "scratch.ts"))).toBe(false);
const status = (await $`git status --porcelain`.cwd(dir).text()).trim();
const status = (await git.status(dir, { porcelainV1: true })).trim();
expect(status).toBe("");
});
it("on an autoresearch branch, keep commits files that were dirty before run_experiment", async () => {
const dir = makeTempDir();
await initGitRepo(dir);
await writeHarnessStub(dir);
await $`git add -A && git commit -m harness`.cwd(dir).quiet();
const dir = freshBranchRepo().dir;
// Seed a tracked file that the agent will edit during the iteration.
fs.mkdirSync(path.join(dir, "src"), { recursive: true });
await Bun.write(path.join(dir, "src", "store.ts"), "export const v = 1;\n");
await $`git add -A && git commit -m seed`.cwd(dir).quiet();
await checkoutBranch(dir, "autoresearch/keep-test");
const runtime = createSessionRuntime();
const harness = createPiHarness();
const init = createInitExperimentTool({
@@ -727,12 +768,8 @@ describe("log_experiment", () => {
// Agent edits BEFORE running the benchmark — the iteration's diff is dirty
// at run_experiment time.
await Bun.write(path.join(dir, "src", "store.ts"), "export const v = 2;\n");
const run = createRunExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: harness.api,
});
await run.execute("r", {}, undefined, undefined, createCtx(dir));
const storage = await openAutoresearchStorage(dir);
seedCompletedRun(storage, storage.getActiveSession()!, { parsedPrimary: 1, parsedMetrics: { m: 1 } });
const log = createLogExperimentTool({
dashboard: dashboardStub(),
@@ -748,18 +785,14 @@ describe("log_experiment", () => {
);
const details = result.details as LogDetails;
expect(details.experiment.modifiedPaths).toContain("src/store.ts");
const status = (await $`git status --porcelain`.cwd(dir).text()).trim();
const status = (await git.status(dir, { porcelainV1: true })).trim();
expect(status).toBe("");
const lastMsg = (await $`git log -1 --pretty=%B`.cwd(dir).text()).trim();
expect(lastMsg).toContain("improvement");
});
it("flags off-scope dirty files even when they were dirty before run_experiment", async () => {
const dir = makeTempDir();
await initGitRepo(dir);
await writeHarnessStub(dir);
await $`git add -A && git commit -m harness`.cwd(dir).quiet();
await checkoutBranch(dir, "autoresearch/scope-test");
const dir = freshBranchRepo().dir;
const runtime = createSessionRuntime();
const harness = createPiHarness();
const init = createInitExperimentTool({
@@ -777,12 +810,8 @@ describe("log_experiment", () => {
// Off-scope edit BEFORE run_experiment.
fs.mkdirSync(path.join(dir, "forbidden"), { recursive: true });
await Bun.write(path.join(dir, "forbidden", "x.ts"), "export const v = 1;\n");
const run = createRunExperimentTool({
dashboard: dashboardStub(),
getRuntime: () => runtime,
pi: harness.api,
});
await run.execute("r", {}, undefined, undefined, createCtx(dir));
const storage = await openAutoresearchStorage(dir);
seedCompletedRun(storage, storage.getActiveSession()!, { parsedPrimary: 1, parsedMetrics: { m: 1 } });
const log = createLogExperimentTool({
dashboard: dashboardStub(),
@@ -815,7 +844,7 @@ describe("update_notes", () => {
});
it("replaces session notes and refreshes runtime state", async () => {
const dir = makeTempDir();
const dir = freshRepo().dir;
await writeHarnessStub(dir);
const runtime = createSessionRuntime();
const harness = createPiHarness();
+127 -109
View File
@@ -13,28 +13,52 @@ import * as piNatives from "@oh-my-pi/pi-natives";
// OutputSink when bash-executor pulls settings via resolveOutputSinkHeadBytes.
const ARTIFACT_HEAD_BYTES_DEFAULT = 20 * 1024;
const BACKGROUND_COMPLETION_RACE_MS = 750;
const KILL_MARKER_DELAY_SECONDS = "0.4";
const KILL_MARKER_DELAY_MS = 400;
// We prove a killed process never wrote its marker by observing until the
// wall-clock instant the marker WOULD have appeared (spawn + delay) plus a
// margin. Anchoring the deadline to a pre-spawn timestamp — instead of blindly
// sleeping a fixed amount after executeBash returns — keeps the wait bounded
// without shrinking the kill-propagation margin: the timeout/abort fires at
// ~100ms, well before the 400ms marker write, so the margin between kill and
// write is unchanged; only the redundant observation tail goes away.
const KILL_MARKER_OBSERVE_MARGIN_MS = 300;
// Killed-vs-orphaned proof: the command's marker write is gated on a `release`
// file the test creates only AFTER the cancel has landed. A truly killed process
// never reaches the write; an orphan still polling reacts within one poll
// interval. This replaces the old "sleep a fixed marker delay, then look"
// approach, which paid that delay in wall-clock time on every run.
const KILL_POLL_SECONDS = "0.01"; // a survivor re-checks `release` every ~10ms
const KILL_SETTLE_MS = 25; // let the kill signal land before we touch `release`
const KILL_REACT_MS = 50; // > one poll interval: a survivor would write its marker
function makeTempDir(): string {
return fs.mkdtempSync(path.join(os.tmpdir(), "omp-bash-exec-"));
}
/** Spin-wait until the wall-clock deadline, polling rather than blind-sleeping. */
async function waitUntil(deadlineMs: number): Promise<void> {
while (Date.now() < deadlineMs) {
await Bun.sleep(20);
function shellQuote(value: string): string {
return `'${value.replace(/'/g, "'\\''")}'`;
}
/** Resolve once `predicate()` holds or `deadlineMs` passes, polling every 2ms. */
async function pollUntil(predicate: () => boolean, deadlineMs: number): Promise<void> {
while (!predicate() && Date.now() < deadlineMs) {
await Bun.sleep(2);
}
}
/**
* Shell that blocks until `release` exists, then writes `marker`. Optionally
* touches `started` first so a test can wait until the command is actually
* running before cancelling it.
*/
function releaseGuardedWrite(marker: string, release: string, started?: string): string {
const touch = started ? `touch ${shellQuote(started)}; ` : "";
return `${touch}while [ ! -f ${shellQuote(release)} ]; do sleep ${KILL_POLL_SECONDS}; done; echo done > ${shellQuote(marker)}`;
}
/**
* After a cancel has been observed, prove the command was killed (not orphaned):
* settle so the kill lands, unblock a would-be survivor via `release`, then give
* it more than one poll interval to write. A killed command never writes `marker`.
*/
async function expectMarkerNeverWritten(marker: string, release: string): Promise<void> {
await Bun.sleep(KILL_SETTLE_MS);
fs.writeFileSync(release, "");
await Bun.sleep(KILL_REACT_MS);
expect(fs.existsSync(marker)).toBe(false);
}
describe("executeBash", () => {
let tempDir: string;
@@ -323,7 +347,10 @@ exit 64
return;
}
const result = await executeBash('python3 -c "import time; time.sleep(10)" & echo $!', {
// Redirect the backgrounded job's stdout so it doesn't hold the executor's
// output pipe open (which would add the ~250ms background-drain grace);
// `$!` still reports the real external PID, which is all this test checks.
const result = await executeBash('python3 -c "import time; time.sleep(10)" >/dev/null 2>&1 & echo $!', {
cwd: tempDir,
timeout: 5000,
});
@@ -467,27 +494,32 @@ exit 64
return;
}
// Compress the JS-side fallback timer (floored at 1000ms in the source) so
// the safety-net fires deterministically without a real 1s wait. Only long
// timers are shrunk — fs/subprocess setup keeps real scheduling — and the
// reported "1 seconds" derives from the configured timeout, not the timer.
const realSetTimeout = globalThis.setTimeout;
vi.spyOn(globalThis, "setTimeout").mockImplementation(((handler: () => void, ms?: number, ...rest: unknown[]) =>
realSetTimeout(
handler,
typeof ms === "number" && ms >= 1000 ? 5 : ms,
...rest,
)) as typeof globalThis.setTimeout);
vi.spyOn(piNatives.Shell.prototype, "run").mockImplementation((_options, onChunk) => {
onChunk?.(null, "started\n");
return new Promise(() => {});
return Promise.withResolvers<never>().promise;
});
const abortSpy = vi.spyOn(piNatives.Shell.prototype, "abort").mockResolvedValue();
const promise = executeBash("sleep 10", {
const result = await executeBash("sleep 10", {
cwd: tempDir,
timeout: 1000,
sessionKey: "hung-native-timeout",
});
const raced = await Promise.race([
promise.then(result => ({ type: "result" as const, result })),
Bun.sleep(1500).then(() => ({ type: "timeout" as const })),
]);
expect(raced.type).toBe("result");
if (raced.type === "result") {
expect(raced.result.cancelled).toBe(true);
expect(raced.result.output).toContain("Command timed out after 1 seconds");
}
expect(result.cancelled).toBe(true);
expect(result.output).toContain("Command timed out after 1 seconds");
expect(abortSpy).toHaveBeenCalled();
});
@@ -501,7 +533,7 @@ exit 64
timeout: 5000,
signal: controller.signal,
});
await Bun.sleep(100);
await Bun.sleep(50);
controller.abort();
const result = await promise;
expect(result.cancelled).toBe(true);
@@ -554,7 +586,7 @@ exit 64
const sessionKey = "parallel-overlap";
const order: string[] = [];
const slow = executeBash('sleep 0.6 && echo "A-done"', { cwd: tempDir, timeout: 5000, sessionKey }).then(
const slow = executeBash('sleep 0.15 && echo "A-done"', { cwd: tempDir, timeout: 5000, sessionKey }).then(
result => {
order.push("slow");
return result;
@@ -571,7 +603,7 @@ exit 64
expect(fastResult.exitCode).toBe(0);
expect(fastResult.output).toContain("B-done");
// If the second call had queued behind the persistent session it could
// not finish before the 600ms sleep of the first.
// not finish before the 150ms sleep of the first.
expect(order).toEqual(["fast", "slow"]);
});
@@ -579,21 +611,30 @@ exit 64
if (process.platform === "win32") return;
const sessionKey = "parallel-timeout-isolation";
const owner = executeBash('sleep 1.3 && echo "owner-done"', {
cwd: tempDir,
timeout: 5000,
sessionKey,
});
// Overlaps with the owner for its whole lifetime; times out at the 1s floor.
const overlapping = await executeBash("sleep 5", { cwd: tempDir, timeout: 1000, sessionKey });
const started = path.join(tempDir, "owner.started");
const release = path.join(tempDir, "owner.release");
// The owner holds the persistent session open (blocked on `release`) so the
// overlapping call is guaranteed to find the session busy and degrade to an
// isolated one-shot shell — the path whose timeout cleanup must NOT touch
// the owner's persistent session.
const owner = executeBash(
`touch ${shellQuote(started)}; while [ ! -f ${shellQuote(release)} ]; do sleep 0.02; done; echo "owner-done"`,
{ cwd: tempDir, timeout: 5000, sessionKey },
);
await pollUntil(() => fs.existsSync(started), Date.now() + 4000);
expect(fs.existsSync(started)).toBe(true);
// Overlaps the owner; degrades to an isolated shell and times out there.
const overlapping = await executeBash("sleep 5", { cwd: tempDir, timeout: 100, sessionKey });
expect(overlapping.cancelled).toBe(true);
// The overlapping timeout must not quarantine or delete the persistent
// session owned by the first call: release it and confirm it completes.
fs.writeFileSync(release, "");
const ownerResult = await owner;
expect(ownerResult.exitCode).toBe(0);
expect(ownerResult.output).toContain("owner-done");
// The overlapping timeout must not quarantine or delete the persistent
// session owned by the first call.
const after = await executeBash('echo "still-ok"', { cwd: tempDir, timeout: 5000, sessionKey });
expect(after.exitCode).toBe(0);
expect(after.output).toContain("still-ok");
@@ -635,17 +676,20 @@ exit 64
expect(result.output).toContain("a");
});
it("handles multi-million line output without freeze or OOM", async () => {
it("handles large output without freeze or OOM", async () => {
if (process.platform === "win32") return;
// 5 million lines ~= 40MB of output. Before the 64KB read buffer and
// direct-push fixes, this would freeze or OOM the process.
const lineCount = 5_000_000;
// Once raw output exceeds the truncation cap, the streaming + middle-elision
// path is volume-independent, so a few hundred KB exercises the same
// no-freeze / no-OOM contract the original 40MB did without paying several
// seconds to generate it. 100k lines of `seq` is ~690KB — an order of
// magnitude past the ~71KB head+tail cap asserted below.
const lineCount = 100_000;
let chunkCount = 0;
const start = Date.now();
const result = await executeBash(`seq 1 ${lineCount}`, {
cwd: tempDir,
timeout: 30_000,
timeout: 10_000,
onChunk: () => {
chunkCount++;
},
@@ -656,11 +700,12 @@ exit 64
expect(result.exitCode).toBe(0);
expect(result.cancelled).toBe(false);
// Output summary should reflect all lines
// Output summary reflects every line even though the visible text is capped.
expect(result.totalLines).toBeGreaterThanOrEqual(lineCount);
// Truncated output should be bounded by head + tail + marker overhead
// (middle-elision keeps the head budget plus the tail spill window).
// Truncated output stays bounded by head + tail + marker overhead
// (middle-elision keeps the head budget plus the tail spill window) — proof
// the full ~690KB stream was never accumulated in the visible buffer.
expect(result.outputBytes).toBeLessThanOrEqual(DEFAULT_MAX_BYTES + ARTIFACT_HEAD_BYTES_DEFAULT + 1024);
// The tail should still contain numeric values near the end of the range.
@@ -673,14 +718,14 @@ exit 64
.filter(Number.isFinite);
expect(tailValues.some(value => value >= lineCount - 500 && value <= lineCount)).toBe(true);
// With 64KB read buffer, ~40MB should produce ~600 chunks, not 5M.
// Allow generous headroom but ensure it's orders of magnitude below lineCount.
// Chunks are coalesced by the read buffer, so onChunk fires orders of
// magnitude less often than once per line — the proof the stream is neither
// delivered line-by-line nor buffered whole.
expect(chunkCount).toBeLessThan(lineCount / 100);
// Should complete in reasonable time (not frozen). On a modern machine
// seq 1 5000000 itself takes ~0.5s; with JS overhead allow 20s.
expect(elapsed).toBeLessThan(20_000);
}, 35_000);
// Should complete promptly (not frozen).
expect(elapsed).toBeLessThan(10_000);
}, 15_000);
it("sources snapshot env vars across session commands", async () => {
if (process.platform === "win32") {
@@ -783,42 +828,35 @@ exit 64
if (process.platform === "win32") return;
const marker = path.join(tempDir, "marker.txt");
const markerEscaped = marker.replace(/'/g, "'\\''");
const release = path.join(tempDir, "marker.release");
// Command creates marker after a short delay, but we timeout before then.
const start = Date.now();
const result = await executeBash(`sleep ${KILL_MARKER_DELAY_SECONDS} && echo done > '${markerEscaped}'`, {
// The foreground command can only write its marker once `release` exists,
// which we never create until after the timeout fires. A killed process
// never reaches the write; an un-killed one would the moment we release it.
const result = await executeBash(releaseGuardedWrite(marker, release), {
cwd: tempDir,
timeout: 100,
timeout: 50,
});
expect(result.cancelled).toBe(true);
// Observe past the instant the marker would have been written had the
// process survived. If it was killed (not orphaned), it never appears.
await waitUntil(start + KILL_MARKER_DELAY_MS + KILL_MARKER_OBSERVE_MARGIN_MS);
expect(fs.existsSync(marker)).toBe(false);
await expectMarkerNeverWritten(marker, release);
});
it("kills background jobs on timeout", async () => {
if (process.platform === "win32") return;
const marker = path.join(tempDir, "marker-bg.txt");
const markerEscaped = marker.replace(/'/g, "'\\''");
const release = path.join(tempDir, "marker-bg.release");
const start = Date.now();
const result = await executeBash(
`{ sleep ${KILL_MARKER_DELAY_SECONDS}; echo done > '${markerEscaped}'; } & sleep 10`,
{
cwd: tempDir,
timeout: 100,
},
);
// The marker writer is a backgrounded subshell that survives the foreground
// `sleep` unless the whole process group is killed on timeout.
const result = await executeBash(`{ ${releaseGuardedWrite(marker, release)}; } & sleep 10`, {
cwd: tempDir,
timeout: 50,
});
expect(result.cancelled).toBe(true);
await waitUntil(start + KILL_MARKER_DELAY_MS + KILL_MARKER_OBSERVE_MARGIN_MS);
expect(fs.existsSync(marker)).toBe(false);
await expectMarkerNeverWritten(marker, release);
});
it("kills background jobs on abort", async () => {
@@ -827,67 +865,47 @@ exit 64
const marker = path.join(tempDir, "marker-bg-abort.txt");
const release = path.join(tempDir, "marker-bg-abort.release");
const started = path.join(tempDir, "marker-bg-abort.started");
const markerEscaped = marker.replace(/'/g, "'\\''");
const releaseEscaped = release.replace(/'/g, "'\\''");
const startedEscaped = started.replace(/'/g, "'\\''");
const controller = new AbortController();
const promise = executeBash(
`{ touch '${startedEscaped}'; while [ ! -f '${releaseEscaped}' ]; do sleep 0.05; done; echo done > '${markerEscaped}'; } & sleep 10`,
{
cwd: tempDir,
timeout: 10000,
signal: controller.signal,
},
);
const promise = executeBash(`{ ${releaseGuardedWrite(marker, release, started)}; } & sleep 10`, {
cwd: tempDir,
timeout: 10000,
signal: controller.signal,
});
const startDeadline = Date.now() + 4000;
while (!fs.existsSync(started) && Date.now() < startDeadline) {
await Bun.sleep(2);
}
// Abort only once the backgrounded subshell is actually running.
await pollUntil(() => fs.existsSync(started), Date.now() + 4000);
expect(fs.existsSync(started)).toBe(true);
controller.abort();
const result = await promise;
expect(result.cancelled).toBe(true);
expect(result.output).toContain("Command cancelled");
// The backgrounded subshell only writes its marker once `release` exists.
// If abort failed to kill the process group, the orphan is still polling
// for `release` every 50ms — touching it makes a survivor react within one
// poll. A short settle first lets the kill signal propagate before we probe.
await Bun.sleep(100);
fs.writeFileSync(release, "");
await Bun.sleep(200);
expect(fs.existsSync(marker)).toBe(false);
await expectMarkerNeverWritten(marker, release);
});
it("kills spawned process on abort (not just orphans it)", async () => {
if (process.platform === "win32") return;
const marker = path.join(tempDir, "marker.txt");
const markerEscaped = marker.replace(/'/g, "'\\''");
const release = path.join(tempDir, "marker.release");
const started = path.join(tempDir, "marker.started");
const controller = new AbortController();
// Command creates marker after a short delay.
const start = Date.now();
const promise = executeBash(`sleep ${KILL_MARKER_DELAY_SECONDS} && echo done > '${markerEscaped}'`, {
const promise = executeBash(releaseGuardedWrite(marker, release, started), {
cwd: tempDir,
timeout: 10000,
signal: controller.signal,
});
// Abort before the command can create the marker.
await Bun.sleep(100);
// Abort only once the foreground command is actually running.
await pollUntil(() => fs.existsSync(started), Date.now() + 4000);
expect(fs.existsSync(started)).toBe(true);
controller.abort();
const result = await promise;
expect(result.cancelled).toBe(true);
expect(result.output).toContain("Command cancelled");
// Observe past the instant the marker would have been written had the
// process survived. If it was killed (not orphaned), it never appears.
await waitUntil(start + KILL_MARKER_DELAY_MS + KILL_MARKER_OBSERVE_MARGIN_MS);
expect(fs.existsSync(marker)).toBe(false);
await expectMarkerNeverWritten(marker, release);
});
});
@@ -1,11 +1,14 @@
/**
* End-to-end contract: a host started with both link variants marks view-link
* guests read-only in `welcome` and refuses their mutating frames, while
* full-link guests keep prompt/abort/agent-cmd capability. Runs over a real
* websocket relay (in-test stand-in speaking the documented relay contract)
* with real sealing — only the TUI context is stubbed.
* full-link guests keep prompt/abort/agent-cmd capability. Runs over an
* in-process relay + fake WebSocket transport (no real sockets, no handshake
* or polling latency) that speaks the documented relay forwarding contract,
* with real AES-GCM sealing — only the TUI context and the network transport
* are stubbed. One host/relay boots once and is reused; guest frames ride the
* in-memory transport, so the suite stays fast and time-independent.
*/
import { afterEach, describe, expect, it } from "bun:test";
import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test";
import { importRoomKey } from "@oh-my-pi/pi-coding-agent/collab/crypto";
import { CollabHost } from "@oh-my-pi/pi-coding-agent/collab/host";
import {
@@ -18,61 +21,114 @@ import {
import { CollabSocket } from "@oh-my-pi/pi-coding-agent/collab/relay-client";
import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types";
interface RelayData {
role: "host" | "guest";
peerId: number;
// ── In-memory transport ────────────────────────────────────────────────────
// FakeWebSocket + InMemoryRelay replace the real Bun.serve relay and loopback
// WebSocket. They mirror the production relay's forwarding contract exactly
// (4-byte peerId envelope routing, peer-joined/peer-left control frames) but
// deliver every frame on a microtask with zero network or timer latency. Real
// CollabSocket / CollabHost run unchanged on top, so sealing, enveloping, the
// hello→welcome handshake, and read-only enforcement are all exercised.
/** Active relay the fake transport routes through; set for the lifetime of this file. */
let activeRelay: InMemoryRelay | null = null;
class FakeWebSocket {
static readonly CONNECTING = 0;
static readonly OPEN = 1;
static readonly CLOSING = 2;
static readonly CLOSED = 3;
binaryType = "blob";
readyState: number = FakeWebSocket.CONNECTING;
readonly role: "host" | "guest";
peerId = 0;
onopen: (() => void) | null = null;
onmessage: ((event: { data: unknown }) => void) | null = null;
onerror: (() => void) | null = null;
onclose: ((event: { code: number; reason: string }) => void) | null = null;
readonly #relay: InMemoryRelay;
constructor(url: string) {
const relay = activeRelay;
if (!relay) throw new Error("FakeWebSocket: no active in-memory relay");
this.#relay = relay;
this.role = new URL(url).searchParams.get("role") === "host" ? "host" : "guest";
queueMicrotask(() => {
if (this.readyState !== FakeWebSocket.CONNECTING) return;
this.readyState = FakeWebSocket.OPEN;
relay.connect(this);
this.onopen?.();
});
}
send(data: Uint8Array): void {
if (this.readyState !== FakeWebSocket.OPEN) return;
// Snapshot: the relay rewrites the peerId in place, and the sender may
// reuse the buffer once send() returns.
const bytes = new Uint8Array(data);
queueMicrotask(() => this.#relay.forward(this, bytes));
}
close(_code?: number): void {
if (this.readyState === FakeWebSocket.CLOSED) return;
this.readyState = FakeWebSocket.CLOSED;
this.#relay.disconnect(this);
queueMicrotask(() => this.onclose?.({ code: 1000, reason: "closed" }));
}
/** Relay → this socket: a binary frame, delivered as ArrayBuffer (binaryType "arraybuffer"). */
deliver(bytes: Uint8Array): void {
if (this.readyState !== FakeWebSocket.OPEN) return;
const copy = new Uint8Array(bytes);
queueMicrotask(() => this.onmessage?.({ data: copy.buffer }));
}
/** Relay → this socket: a JSON control message. */
deliverControl(json: string): void {
if (this.readyState !== FakeWebSocket.OPEN) return;
queueMicrotask(() => this.onmessage?.({ data: json }));
}
}
type RelaySocket = Bun.ServerWebSocket<RelayData>;
/** Single-room in-memory relay mirroring the production forwarding contract. */
class InMemoryRelay {
#host: FakeWebSocket | null = null;
readonly #guests = new Map<number, FakeWebSocket>();
#nextPeerId = 1;
/** Single-room test relay mirroring the production forwarding contract. */
function startTestRelay(): { url: string; stop(): void } {
let host: RelaySocket | null = null;
const guests = new Map<number, RelaySocket>();
let nextPeerId = 1;
const server = Bun.serve({
port: 0,
fetch(req, srv): Response | undefined {
const role = new URL(req.url).searchParams.get("role") === "host" ? "host" : "guest";
const data: RelayData = { role, peerId: 0 };
if (srv.upgrade(req, { data })) return undefined;
return new Response("upgrade failed", { status: 400 });
},
websocket: {
open(ws: RelaySocket): void {
if (ws.data.role === "host") {
host = ws;
return;
}
ws.data.peerId = nextPeerId++;
guests.set(ws.data.peerId, ws);
host?.send(JSON.stringify({ t: "peer-joined", peer: ws.data.peerId }));
},
message(ws: RelaySocket, message: string | Buffer): void {
if (typeof message === "string") return;
const bytes = new Uint8Array(message);
if (ws.data.role === "host") {
const envelope = unpackEnvelope(bytes);
if (!envelope) return;
if (envelope.peerId === 0) {
for (const guest of guests.values()) guest.send(bytes);
} else {
guests.get(envelope.peerId)?.send(bytes);
}
return;
}
rewriteEnvelopePeer(bytes, ws.data.peerId);
host?.send(bytes);
},
close(ws: RelaySocket): void {
if (ws.data.role === "guest") {
guests.delete(ws.data.peerId);
host?.send(JSON.stringify({ t: "peer-left", peer: ws.data.peerId }));
}
},
},
});
return { url: `ws://localhost:${server.port}`, stop: () => server.stop(true) };
connect(ws: FakeWebSocket): void {
if (ws.role === "host") {
this.#host = ws;
return;
}
ws.peerId = this.#nextPeerId++;
this.#guests.set(ws.peerId, ws);
this.#host?.deliverControl(JSON.stringify({ t: "peer-joined", peer: ws.peerId }));
}
forward(from: FakeWebSocket, bytes: Uint8Array): void {
if (from.role === "host") {
const envelope = unpackEnvelope(bytes);
if (!envelope) return;
if (envelope.peerId === 0) {
for (const guest of this.#guests.values()) guest.deliver(bytes);
} else {
this.#guests.get(envelope.peerId)?.deliver(bytes);
}
return;
}
rewriteEnvelopePeer(bytes, from.peerId);
this.#host?.deliver(bytes);
}
disconnect(ws: FakeWebSocket): void {
if (ws.role === "host") {
if (this.#host === ws) this.#host = null;
return;
}
this.#guests.delete(ws.peerId);
this.#host?.deliverControl(JSON.stringify({ t: "peer-left", peer: ws.peerId }));
}
}
interface HostHarness {
@@ -142,7 +198,7 @@ interface TestGuest {
}
/** Frames the host broadcasts on its own schedule (debounced state/agents, entry/event/bus taps). */
const BROADCAST_FRAME_TYPES = new Set<CollabFrame["t"]>(["state", "agents", "entry", "event", "bus"]);
const BROADCAST_FRAME_TYPES: Record<string, true> = { state: true, agents: true, entry: true, event: true, bus: true };
/**
* Raw guest speaking the wire protocol directly. `writeToken` overrides the link's token (e.g. forged).
@@ -160,7 +216,7 @@ async function joinAsGuest(link: string, name: string, writeTokenOverride?: stri
const queue: CollabFrame[] = [];
const waiters: ((frame: CollabFrame) => void)[] = [];
socket.onFrame = frame => {
if (BROADCAST_FRAME_TYPES.has(frame.t)) return;
if (BROADCAST_FRAME_TYPES[frame.t]) return;
const waiter = waiters.shift();
if (waiter) waiter(frame);
else queue.push(frame);
@@ -177,29 +233,46 @@ async function joinAsGuest(link: string, name: string, writeTokenOverride?: stri
return { socket, nextFrame };
}
const cleanups: (() => void | Promise<void>)[] = [];
// ── Shared host/relay, booted once ──────────────────────────────────────────
// Booting the relay + host and connecting the host socket is the only heavy
// step; it is identical across all three tests (none mutate host config), so it
// runs once. Per-test guest state is reset in afterEach.
afterEach(async () => {
for (const cleanup of cleanups.splice(0).reverse()) await cleanup();
const RealWebSocket = globalThis.WebSocket;
const guestCleanups: (() => void)[] = [];
let harness: HostHarness;
let host: CollabHost;
beforeAll(async () => {
globalThis.WebSocket = FakeWebSocket as unknown as typeof WebSocket;
activeRelay = new InMemoryRelay();
harness = makeHostContext();
host = new CollabHost(harness.ctx);
// Port is irrelevant: the fake transport routes by the `role` query param.
await host.start("ws://localhost:8787");
});
async function setup(): Promise<HostHarness & { host: CollabHost }> {
const relay = startTestRelay();
cleanups.push(relay.stop);
const harness = makeHostContext();
const host = new CollabHost(harness.ctx);
await host.start(relay.url);
cleanups.push(() => host.stop("test done"));
return { ...harness, host };
}
afterEach(() => {
for (const cleanup of guestCleanups.splice(0).reverse()) cleanup();
harness.prompts.length = 0;
harness.aborts.count = 0;
});
afterAll(async () => {
// Restore the real transport first so the global is clean even if stop() throws;
// the host's socket holds its own FakeWebSocket/relay refs, so teardown still works.
globalThis.WebSocket = RealWebSocket;
activeRelay = null;
await host.stop("test done");
});
describe("collab read-only links", () => {
it("welcomes view-link guests read-only and refuses their mutating frames", async () => {
const { host, prompts, aborts } = await setup();
const { prompts, aborts } = harness;
expect(host.viewLink).not.toBe(host.link);
const guest = await joinAsGuest(host.viewLink, "viewer");
cleanups.push(() => guest.socket.close());
guestCleanups.push(() => guest.socket.close());
const welcome = await guest.nextFrame();
if (welcome.t !== "welcome") throw new Error(`expected welcome, got ${welcome.t}`);
expect(welcome.readOnly).toBe(true);
@@ -223,10 +296,10 @@ describe("collab read-only links", () => {
});
it("keeps full write capability for guests holding the write token", async () => {
const { host, prompts, nextPrompt } = await setup();
const { prompts, nextPrompt } = harness;
const guest = await joinAsGuest(host.link, "writer");
cleanups.push(() => guest.socket.close());
guestCleanups.push(() => guest.socket.close());
const welcome = await guest.nextFrame();
if (welcome.t !== "welcome") throw new Error(`expected welcome, got ${welcome.t}`);
expect(welcome.readOnly).toBeUndefined();
@@ -239,12 +312,12 @@ describe("collab read-only links", () => {
});
it("treats a forged write token as read-only", async () => {
const { host, prompts } = await setup();
const { prompts } = harness;
// A viewer knows the room key but not the token; garbage must not escalate.
const forged = Buffer.alloc(16, 0xab).toString("base64url");
const guest = await joinAsGuest(host.viewLink, "forger", forged);
cleanups.push(() => guest.socket.close());
guestCleanups.push(() => guest.socket.close());
const welcome = await guest.nextFrame();
if (welcome.t !== "welcome") throw new Error(`expected welcome, got ${welcome.t}`);
+80 -85
View File
@@ -1,4 +1,4 @@
import { afterEach, beforeEach, describe, expect, test } from "bun:test";
import { afterAll, beforeAll, describe, expect, test } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
@@ -9,50 +9,72 @@ const gitInitHelp = await $`git init -h`.quiet().nothrow().text();
const supportsReftable = gitInitHelp.includes("--ref-format");
describe.skipIf(!supportsReftable)("git reftable support", () => {
let testRepoDir: string;
// All git plumbing (init, commits, branch, worktree) is real I/O that exercises
// the reftable backend, but none of it is the contract under test in the bodies
// below — the bodies test our *resolution* code against an already-built reftable
// repo. So the heavy plumbing is built once in beforeAll and shared:
// - `sharedRepoDir`: a reftable repo with two committed branches (main +
// feature-branch, HEAD on feature-branch), reused by the repository and
// worktree tests. Neither test mutates it, so one fixture is safe.
// - `worktreeDir`: a linked worktree (wt-branch) off the shared repo.
// - `configRepoDir`: a separate reftable repo for the config-comment test,
// which rewrites `.git/config` on disk and therefore must not share state.
let sharedRepoDir: string;
let worktreeDir: string;
let configRepoDir: string;
let headSha: string;
beforeEach(async () => {
testRepoDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-reftable-"));
beforeAll(async () => {
// Shared reftable repo: two distinct commits on two branches so ref
// resolution has independent main/feature-branch targets to resolve.
sharedRepoDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-reftable-"));
const initResult = await $`git init --ref-format=reftable --initial-branch=main`.cwd(sharedRepoDir).quiet();
if (initResult.exitCode !== 0) throw new Error(`reftable git init failed (exit ${initResult.exitCode})`);
await $`git config user.name "Test User"`.cwd(sharedRepoDir).quiet();
await $`git config user.email "test@example.com"`.cwd(sharedRepoDir).quiet();
await fs.writeFile(path.join(sharedRepoDir, "file.txt"), "hello world");
await $`git add file.txt`.cwd(sharedRepoDir).quiet();
await $`git commit -m "initial commit"`.cwd(sharedRepoDir).quiet();
await $`git checkout -b feature-branch`.cwd(sharedRepoDir).quiet();
await fs.writeFile(path.join(sharedRepoDir, "file2.txt"), "hello feature");
await $`git add file2.txt`.cwd(sharedRepoDir).quiet();
await $`git commit -m "feature commit"`.cwd(sharedRepoDir).quiet();
// Ground-truth HEAD sha, resolved independently of our utilities.
headSha = (await $`git rev-parse HEAD`.cwd(sharedRepoDir).quiet().text()).trim();
// Linked worktree on its own branch, off the shared repo's HEAD.
worktreeDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-reftable-wt-"));
await $`git worktree add ${worktreeDir} -b wt-branch`.cwd(sharedRepoDir).quiet();
// Independent reftable repo for the config-comment test (no commits needed;
// it only inspects/rewrites the freshly-initialized config).
configRepoDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-reftable-cfg-"));
const cfgInit = await $`git init --ref-format=reftable --initial-branch=main`.cwd(configRepoDir).quiet();
if (cfgInit.exitCode !== 0) throw new Error(`reftable git init (config) failed (exit ${cfgInit.exitCode})`);
});
afterEach(async () => {
await fs.rm(testRepoDir, { recursive: true, force: true });
afterAll(async () => {
await $`git worktree remove ${worktreeDir} -f`.cwd(sharedRepoDir).quiet().nothrow();
await fs.rm(worktreeDir, { recursive: true, force: true }).catch(() => {});
await fs.rm(sharedRepoDir, { recursive: true, force: true }).catch(() => {});
await fs.rm(configRepoDir, { recursive: true, force: true }).catch(() => {});
});
test("resolves references in a reftable repository", async () => {
// Initialize the repository with reftable format
const initResult = await $`git init --ref-format=reftable --initial-branch=main`.cwd(testRepoDir).quiet();
expect(initResult.exitCode).toBe(0);
// Configure basic user details so we can commit
await $`git config user.name "Test User"`.cwd(testRepoDir).quiet();
await $`git config user.email "test@example.com"`.cwd(testRepoDir).quiet();
// Create a file and commit it
await fs.writeFile(path.join(testRepoDir, "file.txt"), "hello world");
await $`git add file.txt`.cwd(testRepoDir).quiet();
await $`git commit -m "initial commit"`.cwd(testRepoDir).quiet();
// Create and checkout a branch
await $`git checkout -b feature-branch`.cwd(testRepoDir).quiet();
await fs.writeFile(path.join(testRepoDir, "file2.txt"), "hello feature");
await $`git add file2.txt`.cwd(testRepoDir).quiet();
await $`git commit -m "feature commit"`.cwd(testRepoDir).quiet();
// Let's test the git utilities on this repo
const repository = await git.repo.resolve(testRepoDir);
const repository = await git.repo.resolve(sharedRepoDir);
expect(repository).not.toBeNull();
const currentBranch = await git.branch.current(testRepoDir);
const currentBranch = await git.branch.current(sharedRepoDir);
expect(currentBranch).toBe("feature-branch");
const headSha = await git.head.sha(testRepoDir);
expect(headSha).not.toBeNull();
expect(headSha).toHaveLength(40);
const resolvedHeadSha = await git.head.sha(sharedRepoDir);
expect(resolvedHeadSha).not.toBeNull();
expect(resolvedHeadSha).toHaveLength(40);
expect(resolvedHeadSha).toBe(headSha);
// Resolve refs/heads/main and refs/heads/feature-branch
const mainSha = await git.ref.resolve(testRepoDir, "refs/heads/main");
const featureSha = await git.ref.resolve(testRepoDir, "refs/heads/feature-branch");
const mainSha = await git.ref.resolve(sharedRepoDir, "refs/heads/main");
const featureSha = await git.ref.resolve(sharedRepoDir, "refs/heads/feature-branch");
expect(mainSha).not.toBeNull();
expect(featureSha).not.toBeNull();
expect(mainSha).toHaveLength(40);
@@ -60,32 +82,28 @@ describe.skipIf(!supportsReftable)("git reftable support", () => {
expect(featureSha).toBe(headSha);
// Test HEAD resolution (object shape)
const headState = await git.head.resolve(testRepoDir);
const headState = await git.head.resolve(sharedRepoDir);
expect(headState).not.toBeNull();
if (headState?.kind !== "ref") throw new Error("expected ref head");
expect(headState.branchName).toBe("feature-branch");
expect(headState.commit).toBe(headSha);
// Test HEAD resolution sync
const headStateSync = git.head.resolveSync(testRepoDir);
const headStateSync = git.head.resolveSync(sharedRepoDir);
expect(headStateSync).not.toBeNull();
if (headStateSync?.kind !== "ref") throw new Error("expected ref head sync");
expect(headStateSync.branchName).toBe("feature-branch");
expect(headStateSync.commit).toBe(headSha);
// Test exists check
const mainExists = await git.ref.exists(testRepoDir, "refs/heads/main");
const nonexistentExists = await git.ref.exists(testRepoDir, "refs/heads/nonexistent");
const mainExists = await git.ref.exists(sharedRepoDir, "refs/heads/main");
const nonexistentExists = await git.ref.exists(sharedRepoDir, "refs/heads/nonexistent");
expect(mainExists).toBe(true);
expect(nonexistentExists).toBe(false);
});
test("handles git config trailing comments correctly", async () => {
// Initialize the repository with reftable format
const initResult = await $`git init --ref-format=reftable --initial-branch=main`.cwd(testRepoDir).quiet();
expect(initResult.exitCode).toBe(0);
const repository = await git.repo.resolve(testRepoDir);
const repository = await git.repo.resolve(configRepoDir);
expect(repository).not.toBeNull();
if (!repository) return;
expect(await git.repo.isReftable(repository)).toBe(true);
@@ -101,7 +119,7 @@ describe.skipIf(!supportsReftable)("git reftable support", () => {
);
await fs.writeFile(configPath, newConfigWithSemicolon);
const repository2 = await git.repo.resolve(testRepoDir);
const repository2 = await git.repo.resolve(configRepoDir);
expect(repository2).not.toBeNull();
if (repository2) {
expect(await git.repo.isReftable(repository2)).toBe(true);
@@ -112,7 +130,7 @@ describe.skipIf(!supportsReftable)("git reftable support", () => {
const newConfigWithHash = baseConfig.replace("refstorage = reftable", "refstorage = reftable # trailing hash");
await fs.writeFile(configPath, newConfigWithHash);
const repository3 = await git.repo.resolve(testRepoDir);
const repository3 = await git.repo.resolve(configRepoDir);
expect(repository3).not.toBeNull();
if (repository3) {
expect(await git.repo.isReftable(repository3)).toBe(true);
@@ -123,7 +141,7 @@ describe.skipIf(!supportsReftable)("git reftable support", () => {
const newConfigWithQuotes = baseConfig.replace("refstorage = reftable", 'refstorage = "reftable ; not comment"');
await fs.writeFile(configPath, newConfigWithQuotes);
const repository4 = await git.repo.resolve(testRepoDir);
const repository4 = await git.repo.resolve(configRepoDir);
expect(repository4).not.toBeNull();
if (repository4) {
// This value would be "reftable ; not comment", which shouldn't match "reftable"
@@ -138,7 +156,7 @@ describe.skipIf(!supportsReftable)("git reftable support", () => {
);
await fs.writeFile(configPath, newConfigWithAdjacentHash);
const repository5 = await git.repo.resolve(testRepoDir);
const repository5 = await git.repo.resolve(configRepoDir);
expect(repository5).not.toBeNull();
if (repository5) {
expect(await git.repo.isReftable(repository5)).toBe(true);
@@ -152,7 +170,7 @@ describe.skipIf(!supportsReftable)("git reftable support", () => {
);
await fs.writeFile(configPath, newConfigWithAdjacentSemicolon);
const repository6 = await git.repo.resolve(testRepoDir);
const repository6 = await git.repo.resolve(configRepoDir);
expect(repository6).not.toBeNull();
if (repository6) {
expect(await git.repo.isReftable(repository6)).toBe(true);
@@ -166,7 +184,7 @@ describe.skipIf(!supportsReftable)("git reftable support", () => {
);
await fs.writeFile(configPath, newConfigWithSectionComment);
const repository7 = await git.repo.resolve(testRepoDir);
const repository7 = await git.repo.resolve(configRepoDir);
expect(repository7).not.toBeNull();
if (repository7) {
expect(await git.repo.isReftable(repository7)).toBe(true);
@@ -175,45 +193,22 @@ describe.skipIf(!supportsReftable)("git reftable support", () => {
});
test("resolves references in a reftable worktree", async () => {
// Initialize the repository with reftable format
const initResult = await $`git init --ref-format=reftable --initial-branch=main`.cwd(testRepoDir).quiet();
expect(initResult.exitCode).toBe(0);
// Resolve the repository for the worktree (built in beforeAll)
const repository = await git.repo.resolve(worktreeDir);
expect(repository).not.toBeNull();
if (!repository) return;
// Configure basic user details so we can commit
await $`git config user.name "Test User"`.cwd(testRepoDir).quiet();
await $`git config user.email "test@example.com"`.cwd(testRepoDir).quiet();
expect(repository.gitDir).not.toBe(repository.commonDir);
expect(await git.repo.isReftable(repository)).toBe(true);
// Create a file and commit it
await fs.writeFile(path.join(testRepoDir, "file.txt"), "hello world");
await $`git add file.txt`.cwd(testRepoDir).quiet();
await $`git commit -m "initial commit"`.cwd(testRepoDir).quiet();
// Check current branch on worktree
const currentBranch = await git.branch.current(worktreeDir);
expect(currentBranch).toBe("wt-branch");
// Create a linked worktree
const worktreeDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-reftable-wt-"));
try {
await $`git worktree add ${worktreeDir} -b wt-branch`.cwd(testRepoDir).quiet();
// Resolve the repository for the worktree
const repository = await git.repo.resolve(worktreeDir);
expect(repository).not.toBeNull();
if (!repository) return;
expect(repository.gitDir).not.toBe(repository.commonDir);
expect(await git.repo.isReftable(repository)).toBe(true);
// Check current branch on worktree
const currentBranch = await git.branch.current(worktreeDir);
expect(currentBranch).toBe("wt-branch");
// Check that HEAD resolves correctly in the worktree
const headState = await git.head.resolve(worktreeDir);
expect(headState).not.toBeNull();
if (headState?.kind !== "ref") throw new Error("expected ref head in worktree");
expect(headState.branchName).toBe("wt-branch");
} finally {
// Clean up the worktree
await $`git worktree remove ${worktreeDir} -f`.cwd(testRepoDir).quiet().nothrow();
await fs.rm(worktreeDir, { recursive: true, force: true }).catch(() => {});
}
// Check that HEAD resolves correctly in the worktree
const headState = await git.head.resolve(worktreeDir);
expect(headState).not.toBeNull();
if (headState?.kind !== "ref") throw new Error("expected ref head in worktree");
expect(headState.branchName).toBe("wt-branch");
});
});
@@ -113,7 +113,12 @@ async function toolNamesFor(harness: GoalHarness): Promise<string[]> {
}
async function waitForMicrotasks(): Promise<void> {
await Bun.sleep(0);
// Pure microtask flush — deterministic and fake-timer-safe (no macrotask /
// real-clock dependency). Lets queued `.then` callbacks settle so a fired
// continuation tick would be observed before we assert it was dropped.
await Promise.resolve();
await Promise.resolve();
await Promise.resolve();
}
async function armInputWaiter(mode: InteractiveMode): Promise<{
@@ -150,6 +155,7 @@ describe("InteractiveMode goal mode integration", () => {
});
afterEach(async () => {
vi.useRealTimers();
vi.restoreAllMocks();
await harness.cleanup();
});
@@ -232,15 +238,19 @@ describe("InteractiveMode goal mode integration", () => {
// taking the streaming branch, or any extension that triggers a turn).
// Without the streaming-aware guard the timer fires onInputCallback
// with a `goal-continuation` and submitInteractiveInput resurfaces
// AgentBusyError via promptCustomMessage.
// AgentBusyError via promptCustomMessage. Driven with fake timers so the
// 800ms window is exercised deterministically without a real wall-clock wait.
await harness.mode.handleGoalModeCommand("Ship the release");
vi.useFakeTimers();
const waiter = await armInputWaiter(harness.mode);
let streaming = true;
Object.defineProperty(harness.session, "isStreaming", { configurable: true, get: () => streaming });
// Let the 800ms timer fire while streaming is true.
await Bun.sleep(900);
// Fire the armed 800ms continuation timer while streaming is true.
vi.advanceTimersByTime(800);
await waitForMicrotasks();
expect(waiter.getResolvedText()).toBeUndefined();
@@ -1,4 +1,4 @@
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs/promises";
import * as path from "node:path";
import { Agent, AgentBusyError } from "@oh-my-pi/pi-agent-core";
@@ -61,22 +61,38 @@ function compactNumber(value: number): string {
}
describe("InteractiveMode plan review rendering", () => {
// Per-test, mutated by tests (planMode flags, spies, model roles, dispose/recreate).
let tempDir: TempDir;
let authStorage: AuthStorage;
let session: AgentSession;
let mode: InteractiveMode;
// Shared across the whole describe: AuthStorage (a SQLite db) and ModelRegistry
// are the expensive pieces (~14ms/test combined) and tests only ever read from
// them — `find()` is a pure lookup over a model list frozen at construction, and
// the lone `setRuntimeApiKey` re-call is idempotent. Hoisting them out of
// `beforeEach` is the dominant body-time win.
let sharedTempDir: TempDir;
let authStorage: AuthStorage;
let modelRegistry: ModelRegistry;
beforeAll(() => {
beforeAll(async () => {
initTheme();
resetSettingsForTest();
sharedTempDir = TempDir.createSync("@pi-plan-review-shared-");
await Settings.init({ inMemory: true, cwd: sharedTempDir.path() });
authStorage = await AuthStorage.create(path.join(sharedTempDir.path(), "testauth.db"));
authStorage.setRuntimeApiKey("anthropic", "test-key");
modelRegistry = new ModelRegistry(authStorage);
});
afterAll(() => {
authStorage?.close();
sharedTempDir?.removeSync();
});
beforeEach(async () => {
resetSettingsForTest();
tempDir = TempDir.createSync("@pi-plan-review-");
await Settings.init({ inMemory: true, cwd: tempDir.path() });
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
authStorage.setRuntimeApiKey("anthropic", "test-key");
const modelRegistry = new ModelRegistry(authStorage);
const model = modelRegistry.find("anthropic", "claude-sonnet-4-5");
if (!model) {
throw new Error("Expected claude-sonnet-4-5 to exist in registry");
@@ -102,15 +118,12 @@ describe("InteractiveMode plan review rendering", () => {
vi.restoreAllMocks();
const currentMode = mode;
const currentSession = session;
const currentAuthStorage = authStorage;
const currentTempDir = tempDir;
mode = undefined as unknown as InteractiveMode;
session = undefined as unknown as AgentSession;
authStorage = undefined as unknown as AuthStorage;
tempDir = undefined as unknown as TempDir;
currentMode?.stop();
await currentSession?.dispose();
currentAuthStorage?.close();
currentTempDir?.removeSync();
setKeybindings(KeybindingsManager.inMemory());
resetSettingsForTest();
@@ -264,6 +277,9 @@ describe("InteractiveMode plan review rendering", () => {
return { hide: vi.fn() } as never;
});
let feedback = "";
// Resolve the instant the real $EDITOR subprocess commits its output back
// through onFeedbackChange — a deterministic signal, not a polled timer.
const { promise: editorApplied, resolve: markEditorApplied } = Promise.withResolvers<void>();
try {
Bun.env.EDITOR = editorPath;
@@ -272,7 +288,12 @@ describe("InteractiveMode plan review rendering", () => {
"# Plan\n\nIntro\n\n## Rollout\n\nSteps\n\n## Verify\n\nChecks\n",
"Plan mode - next step",
["Approve and execute", "Refine plan"],
{ onFeedbackChange: value => (feedback = value) },
{
onFeedbackChange: value => {
feedback = value;
if (value.includes("- include smoke test")) markEditorApplied();
},
},
);
expect(capturedOverlay).toBeDefined();
@@ -282,9 +303,8 @@ describe("InteractiveMode plan review rendering", () => {
overlay.handleInput("a");
for (const ch of "draft") overlay.handleInput(ch);
overlay.handleInput("\x05"); // ctrl+e
for (let i = 0; i < 50 && !feedback.includes("- include smoke test"); i++) {
await Bun.sleep(10);
}
// The subprocess is real; block on its commit signal instead of polling.
await editorApplied;
expect(feedback).toContain("## Rollout\n```md\n- add rollback command\n- include smoke test\n```");
overlay.handleInput("\x1b[B"); // Rollout -> Verify
@@ -407,7 +427,6 @@ describe("InteractiveMode plan review rendering", () => {
mode.stop();
await session.dispose();
const modelRegistry = new ModelRegistry(authStorage);
const executionModel = modelRegistry.find("anthropic", "claude-sonnet-4-5");
const planModel = modelRegistry.find("anthropic", "claude-opus-4-6");
if (!executionModel?.contextWindow || !planModel?.contextWindow) {
@@ -1160,21 +1179,16 @@ describe("InteractiveMode plan review rendering", () => {
});
}
it("B1: Approve and compact context + ok outcome → flag cleared by finally", async () => {
await approveWithCompact("ok");
expect(session.isPlanCompactAbortPending).toBe(false);
});
it("B2: Approve and compact context + cancelled outcome → flag cleared by finally even without aborted message_end", async () => {
await approveWithCompact("cancelled");
// The leak-guard contract: no aborted message_end consumed the flag,
// but `finally` still cleared it so the next real abort cannot be
// silenced.
expect(session.isPlanCompactAbortPending).toBe(false);
});
it("B3: Approve and compact context + failed outcome → flag cleared by finally", async () => {
await approveWithCompact("failed");
// B1-B3: every terminal compaction outcome must leave the flag cleared by
// `#approvePlan`'s `finally`. No aborted message_end is required to consume it,
// so a stranded flag could otherwise silence the next unrelated abort. One
// parametrized case per outcome keeps ok/cancelled/failed each covered.
it.each([
"ok",
"cancelled",
"failed",
] as const)("B1-B3: Approve and compact context + %s outcome → flag cleared by finally", async outcome => {
await approveWithCompact(outcome);
expect(session.isPlanCompactAbortPending).toBe(false);
});
@@ -5,7 +5,8 @@
* never passed `session.promptTemplates` into the autocomplete provider.
*/
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as os from "node:os";
import * as path from "node:path";
import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core";
import { type Api, Effort, type Model } from "@oh-my-pi/pi-ai";
@@ -36,35 +37,57 @@ function makeTool(name: string): AgentTool {
describe("InteractiveMode prompt-template autocomplete (#2462)", () => {
let tempDir: TempDir;
let authStorage: AuthStorage;
let registry: ModelRegistry;
let model: Model<Api>;
let tools: AgentTool[];
let originalHome: string | undefined;
let mode: InteractiveMode | undefined;
let session: AgentSession | undefined;
beforeAll(() => {
beforeAll(async () => {
initTheme();
});
beforeEach(async () => {
Bun.gc(true);
resetSettingsForTest();
// One empty temp dir doubles as the project cwd and the (isolated) home
// directory. Pointing $HOME here keeps `refreshSlashCommandState`'s capability
// scan off the real home dir — that scan was the per-test latency and a source
// of nondeterminism (it picked up whatever slash commands / plugins happened to
// live in the developer's or CI's home).
tempDir = TempDir.createSync("@pi-prompt-template-autocomplete-");
originalHome = process.env.HOME;
process.env.HOME = tempDir.path();
await Settings.init({ inMemory: true, cwd: tempDir.path() });
Settings.instance.set("startup.quiet", true);
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
authStorage.setRuntimeApiKey("anthropic", "test-key");
// ModelRegistry (bundled-model load) and the resolved model are immutable across
// these tests, so build them once rather than per test.
registry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
model = modelOrThrow(registry, "claude-sonnet-4-5");
tools = [makeTool("read")];
});
beforeEach(() => {
// Re-assert the home seam each test (afterEach's restoreAllMocks clears it).
// os.homedir() is what the capability loader reads to locate user-level slash
// commands; aiming it at the empty temp home makes discovery fast and
// deterministic regardless of the real home's contents.
vi.spyOn(os, "homedir").mockReturnValue(tempDir.path());
});
afterEach(async () => {
vi.restoreAllMocks();
mode?.stop();
await session?.dispose();
authStorage?.close();
tempDir?.removeSync();
mode = undefined;
session = undefined;
authStorage = undefined as unknown as AuthStorage;
tempDir = undefined as unknown as TempDir;
});
afterAll(() => {
authStorage?.close();
if (originalHome === undefined) delete process.env.HOME;
else process.env.HOME = originalHome;
tempDir?.removeSync();
resetSettingsForTest();
Bun.gc(true);
});
function modelOrThrow(registry: ModelRegistry, id: string): Model<Api> {
@@ -73,12 +96,10 @@ describe("InteractiveMode prompt-template autocomplete (#2462)", () => {
return model;
}
async function createHarness(
templates: PromptTemplate[],
): Promise<{ mode: InteractiveMode; session: AgentSession }> {
const registry = new ModelRegistry(authStorage, path.join(tempDir.path(), `models-${Bun.nanoseconds()}.yml`));
const model = modelOrThrow(registry, "claude-sonnet-4-5");
const tools = [makeTool("read")];
function createHarness(templates: PromptTemplate[]): { mode: InteractiveMode; session: AgentSession } {
// SessionManager and AgentSession can't be shared: AgentSession.dispose() closes
// its SessionManager, so each test gets a fresh pair. They're cheap (in-memory)
// next to the hoisted ModelRegistry/AuthStorage/temp-dir setup.
const manager = SessionManager.create(tempDir.path(), path.join(tempDir.path(), `active-${Bun.nanoseconds()}`));
const created = new AgentSession({
agent: new Agent({
@@ -117,7 +138,7 @@ describe("InteractiveMode prompt-template autocomplete (#2462)", () => {
}
it("includes discovered prompt templates in slash-command autocomplete", async () => {
const created = await createHarness([
const created = createHarness([
{
name: "review",
description: "Review code for bugs (project)",
@@ -142,7 +163,7 @@ describe("InteractiveMode prompt-template autocomplete (#2462)", () => {
});
it("does not duplicate templates whose names collide with builtin slash commands", async () => {
const created = await createHarness([
const created = createHarness([
{
name: "exit",
description: "Custom exit template (project)",
@@ -163,7 +184,7 @@ describe("InteractiveMode prompt-template autocomplete (#2462)", () => {
});
it("does not duplicate templates whose names collide with builtin slash command aliases", async () => {
const created = await createHarness([
const created = createHarness([
{
name: "models",
description: "Custom models template (project)",
@@ -1,4 +1,4 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
@@ -8,8 +8,39 @@ import {
readInstalledPluginsRegistry,
} from "@oh-my-pi/pi-coding-agent/extensibility/plugins/marketplace";
// Fixture: the valid-marketplace directory used across all tests.
const FIXTURE_DIR = path.join(import.meta.dir, "fixtures", "valid-marketplace");
// Minimal marketplace fixture, built once into a temp dir (see beforeAll). It carries only
// what these tests assert — one plugin entry plus a plugin.json for the version-fallback path —
// so each install's recursive cache copy stays a single file rather than the full shared fixture.
let FIXTURE_DIR: string;
function buildMinimalFixture(): string {
const root = fs.mkdtempSync(path.join(os.tmpdir(), "omp-mgr-fixture-"));
const pluginDir = path.join(root, "plugins", "hello-plugin");
fs.mkdirSync(path.join(pluginDir, ".claude-plugin"), { recursive: true });
fs.mkdirSync(path.join(root, ".claude-plugin"), { recursive: true });
fs.writeFileSync(
path.join(root, ".claude-plugin", "marketplace.json"),
JSON.stringify({
name: "test-marketplace",
owner: { name: "Test Author", email: "test@example.com" },
metadata: { description: "A test marketplace for unit tests", version: "1.0.0" },
plugins: [
{
name: "hello-plugin",
source: "./plugins/hello-plugin",
description: "A test plugin that greets",
version: "1.0.0",
},
],
}),
);
// Consulted only when the catalog version is stripped (the version-fallback test).
fs.writeFileSync(
path.join(pluginDir, ".claude-plugin", "plugin.json"),
JSON.stringify({ name: "hello-plugin", version: "1.0.0" }),
);
return root;
}
// ── Test helper ───────────────────────────────────────────────────────────────
@@ -52,6 +83,14 @@ function createTestContext(): TestContext {
describe("MarketplaceManager", () => {
let ctx: TestContext;
beforeAll(() => {
FIXTURE_DIR = buildMinimalFixture();
});
afterAll(() => {
fs.rmSync(FIXTURE_DIR, { recursive: true, force: true });
});
beforeEach(() => {
ctx = createTestContext();
});
@@ -102,9 +141,6 @@ describe("MarketplaceManager", () => {
it("updateMarketplace re-fetches and updates updatedAt", async () => {
const added = await ctx.manager.addMarketplace(FIXTURE_DIR);
// Small sleep so clock advances
await Bun.sleep(5);
const updated = await ctx.manager.updateMarketplace("test-marketplace");
expect(updated.name).toBe("test-marketplace");
expect(updated.addedAt).toBe(added.addedAt);
@@ -193,7 +229,7 @@ describe("MarketplaceManager", () => {
});
});
it("installPlugin with scope:project → stores project scope", async () => {
it("installPlugin with scope:project → persisted in project registry, isolated from user", async () => {
await ctx.manager.addMarketplace(FIXTURE_DIR);
const instEntry = await ctx.manager.installPlugin("hello-plugin", "test-marketplace", {
scope: "project",
@@ -202,9 +238,11 @@ describe("MarketplaceManager", () => {
expect(instEntry.version).toBe("1.0.0");
expect(fs.existsSync(instEntry.installPath)).toBe(true);
// Verify scope was persisted to the registry, not just returned in-memory.
const installed = await ctx.manager.listInstalledPlugins();
expect(installed[0].entries[0].scope).toBe("project");
// Persisted to the project registry with project scope — and absent from the user registry.
const projectReg = await readInstalledPluginsRegistry(path.join(ctx.tmpDir, "project_installed_plugins.json"));
expect(projectReg.plugins["hello-plugin@test-marketplace"]?.[0].scope).toBe("project");
const userReg = await readInstalledPluginsRegistry(path.join(ctx.tmpDir, "installed_plugins.json"));
expect(userReg.plugins["hello-plugin@test-marketplace"]).toBeUndefined();
});
it("installPlugin already installed → throws without force", async () => {
@@ -305,19 +343,6 @@ describe("MarketplaceManager", () => {
});
// ── Scope feature ────────────────────────────────────────────────────────
it("installPlugin scope:project → writes to project registry, not user registry", async () => {
await ctx.manager.addMarketplace(FIXTURE_DIR);
await ctx.manager.installPlugin("hello-plugin", "test-marketplace", { scope: "project" });
const projectReg = await readInstalledPluginsRegistry(path.join(ctx.tmpDir, "project_installed_plugins.json"));
expect(projectReg.plugins["hello-plugin@test-marketplace"]).toBeDefined();
expect(projectReg.plugins["hello-plugin@test-marketplace"]![0].scope).toBe("project");
// User registry must NOT contain this plugin.
const userReg = await readInstalledPluginsRegistry(path.join(ctx.tmpDir, "installed_plugins.json"));
expect(userReg.plugins["hello-plugin@test-marketplace"]).toBeUndefined();
});
it("installPlugin scope:project when no projectInstalledRegistryPath → throws", async () => {
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), "omp-mgr-noproj-"));
try {
@@ -336,14 +361,22 @@ describe("MarketplaceManager", () => {
}
});
it("uninstallPlugin with plugin in both scopes, no scope arg → throws disambiguation error", async () => {
it("plugin in both scopes, no scope arg → uninstall/setPluginEnabled/upgrade all throw disambiguation", async () => {
await ctx.manager.addMarketplace(FIXTURE_DIR);
await ctx.manager.installPlugin("hello-plugin", "test-marketplace", { scope: "user" });
await ctx.manager.installPlugin("hello-plugin", "test-marketplace", { scope: "project" });
// Each mutating op refuses an ambiguous (both-scope) target before touching any state,
// so the three assertions share one setup without interfering.
await expect(ctx.manager.uninstallPlugin("hello-plugin@test-marketplace")).rejects.toThrow(
/both user and project scope/,
);
await expect(ctx.manager.setPluginEnabled("hello-plugin@test-marketplace", false)).rejects.toThrow(
/both user and project scope/,
);
await expect(ctx.manager.upgradePlugin("hello-plugin@test-marketplace")).rejects.toThrow(
/both user and project scope/,
);
});
it("uninstallPlugin scope:user removes only user entry, keeps project entry", async () => {
@@ -375,26 +408,6 @@ describe("MarketplaceManager", () => {
expect(fs.existsSync(userEntry.installPath)).toBe(true);
});
it("setPluginEnabled with plugin in both scopes, no scope arg → throws disambiguation error", async () => {
await ctx.manager.addMarketplace(FIXTURE_DIR);
await ctx.manager.installPlugin("hello-plugin", "test-marketplace", { scope: "user" });
await ctx.manager.installPlugin("hello-plugin", "test-marketplace", { scope: "project" });
await expect(ctx.manager.setPluginEnabled("hello-plugin@test-marketplace", false)).rejects.toThrow(
/both user and project scope/,
);
});
it("upgradePlugin with plugin in both scopes, no scope arg → throws disambiguation error", async () => {
await ctx.manager.addMarketplace(FIXTURE_DIR);
await ctx.manager.installPlugin("hello-plugin", "test-marketplace", { scope: "user" });
await ctx.manager.installPlugin("hello-plugin", "test-marketplace", { scope: "project" });
await expect(ctx.manager.upgradePlugin("hello-plugin@test-marketplace")).rejects.toThrow(
/both user and project scope/,
);
});
it("listInstalledPlugins marks user entry as shadowed when project entry exists for same ID", async () => {
await ctx.manager.addMarketplace(FIXTURE_DIR);
await ctx.manager.installPlugin("hello-plugin", "test-marketplace", { scope: "user" });
@@ -1,4 +1,4 @@
import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, test, vi } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
@@ -21,12 +21,23 @@ interface SessionFixture {
session: any;
modelRegistry: any;
model: Model;
whenSettled: Promise<void>;
}
const createdDirs = new Set<string>();
let sharedRoot: string | undefined;
function deferred(): { promise: Promise<void>; resolve: () => void } {
let resolve!: () => void;
const promise = new Promise<void>(res => {
resolve = res;
});
return { promise, resolve };
}
async function makeTempDir(prefix: string): Promise<string> {
const dir = path.join(os.tmpdir(), `${prefix}-${Snowflake.next()}`);
const base = sharedRoot ?? os.tmpdir();
const dir = path.join(base, `${prefix}-${Snowflake.next()}`);
await fs.mkdir(dir, { recursive: true });
createdDirs.add(dir);
return dir;
@@ -67,7 +78,10 @@ async function createFixture(overrides?: Partial<Record<string, unknown>>): Prom
});
const model = createModel();
const modelRegistry = createModelRegistry(model);
const refreshBaseSystemPrompt = vi.fn(async () => undefined);
const settled = deferred();
const refreshBaseSystemPrompt = vi.fn(async () => {
settled.resolve();
});
const session = {
sessionManager: {
getSessionFile: () => sessionFile,
@@ -81,24 +95,38 @@ async function createFixture(overrides?: Partial<Record<string, unknown>>): Prom
refreshBaseSystemPrompt,
};
return { agentDir, sessionDir, sessionFile, settings, session, modelRegistry, model };
return { agentDir, sessionDir, sessionFile, settings, session, modelRegistry, model, whenSettled: settled.promise };
}
async function waitFor(assertion: () => Promise<void> | void, timeoutMs = 3000): Promise<void> {
const start = Date.now();
let lastError: unknown;
while (Date.now() - start < timeoutMs) {
try {
await assertion();
return;
} catch (error) {
lastError = error;
}
await Bun.sleep(20);
// Resolve any already-scheduled microtasks/macrotasks without a fixed wall delay.
const flushAsync = (): Promise<void> => new Promise<void>(resolve => setTimeout(resolve, 0));
// Await the pipeline's completion signal (its final `refreshBaseSystemPrompt`)
// instead of polling, racing a generous timeout so a stalled regression fails
// loudly rather than hanging.
async function settle(promise: Promise<void>, label: string, timeoutMs = 3000): Promise<void> {
let timer: ReturnType<typeof setTimeout> | undefined;
const timeout = new Promise<never>((_, reject) => {
timer = setTimeout(() => reject(new Error(`Timed out waiting for ${label}`)), timeoutMs);
});
try {
await Promise.race([promise, timeout]);
} finally {
if (timer) clearTimeout(timer);
}
throw lastError;
}
beforeAll(async () => {
sharedRoot = path.join(os.tmpdir(), `memories-runtime-${Snowflake.next()}`);
await fs.mkdir(sharedRoot, { recursive: true });
});
afterAll(async () => {
if (sharedRoot) await fs.rm(sharedRoot, { recursive: true, force: true });
sharedRoot = undefined;
createdDirs.clear();
});
describe("memories runtime", () => {
let savedXdgData: string | undefined;
let savedXdgState: string | undefined;
@@ -117,10 +145,6 @@ describe("memories runtime", () => {
vi.restoreAllMocks();
process.env.XDG_DATA_HOME = savedXdgData;
process.env.XDG_STATE_HOME = savedXdgState;
for (const dir of createdDirs) {
await fs.rm(dir, { recursive: true, force: true });
}
createdDirs.clear();
});
test("startup gating skips when disabled or subagent depth", async () => {
@@ -161,7 +185,7 @@ describe("memories runtime", () => {
taskDepth: 0,
});
await Bun.sleep(50);
await flushAsync();
expect(stage1Spy).not.toHaveBeenCalled();
});
@@ -213,18 +237,17 @@ describe("memories runtime", () => {
});
const memoryRoot = getMemoryRoot(fx.agentDir, fx.session.sessionManager.getCwd());
await waitFor(async () => {
expect((await fs.readFile(path.join(memoryRoot, "MEMORY.md"), "utf8")).trim()).toBe(
"# Memory\n\nConsolidated body",
);
expect((await fs.readFile(path.join(memoryRoot, "memory_summary.md"), "utf8")).trim()).toBe(
"Consolidated summary",
);
expect(
(await fs.readFile(path.join(memoryRoot, "skills", "deploy-playbook", "SKILL.md"), "utf8")).trim(),
).toBe("# Deploy\nUse blue/green.");
expect(fx.session.refreshBaseSystemPrompt).toHaveBeenCalledTimes(1);
});
await settle(fx.whenSettled, "phase1->phase2 pipeline");
expect((await fs.readFile(path.join(memoryRoot, "MEMORY.md"), "utf8")).trim()).toBe(
"# Memory\n\nConsolidated body",
);
expect((await fs.readFile(path.join(memoryRoot, "memory_summary.md"), "utf8")).trim()).toBe(
"Consolidated summary",
);
expect((await fs.readFile(path.join(memoryRoot, "skills", "deploy-playbook", "SKILL.md"), "utf8")).trim()).toBe(
"# Deploy\nUse blue/green.",
);
expect(fx.session.refreshBaseSystemPrompt).toHaveBeenCalledTimes(1);
expect(ai.completeSimple).toHaveBeenCalled();
expect(ai.completeSimple).toHaveBeenCalledTimes(2);
const phase2Prompt = completeSpy.mock.calls[1]?.[1];
@@ -293,9 +316,8 @@ describe("memories runtime", () => {
});
const memoryRoot = getMemoryRoot(fx.agentDir, fx.session.sessionManager.getCwd());
await waitFor(async () => {
expect((await fs.readFile(path.join(memoryRoot, "MEMORY.md"), "utf8")).trim()).toBe("# Memory\n\nBody");
});
await settle(fx.whenSettled, "effort-clamp pipeline");
expect((await fs.readFile(path.join(memoryRoot, "MEMORY.md"), "utf8")).trim()).toBe("# Memory\n\nBody");
expect(spy).toHaveBeenCalledTimes(2);
// stage1 requested `low`, phase2 requested `medium`; both must clamp up to the
@@ -360,13 +382,12 @@ describe("memories runtime", () => {
taskDepth: 0,
});
await waitFor(async () => {
const files = await fs.readdir(path.join(memoryRoot, "rollout_summaries"));
expect(files.includes("old.md")).toBe(false);
expect(files).toEqual(expect.arrayContaining(["thread-a-alpha.md", "thread-b-beta.md"]));
const raw = await fs.readFile(path.join(memoryRoot, "raw_memories.md"), "utf8");
expect(raw.indexOf("## thread-b")).toBeLessThan(raw.indexOf("## thread-a"));
});
await settle(fx.whenSettled, "phase2 sync/prune");
const files = await fs.readdir(path.join(memoryRoot, "rollout_summaries"));
expect(files.includes("old.md")).toBe(false);
expect(files).toEqual(expect.arrayContaining(["thread-a-alpha.md", "thread-b-beta.md"]));
const raw = await fs.readFile(path.join(memoryRoot, "raw_memories.md"), "utf8");
expect(raw.indexOf("## thread-b")).toBeLessThan(raw.indexOf("## thread-a"));
});
test("phase2 empty-input cleanup removes consolidated files and skills dir", async () => {
@@ -391,14 +412,13 @@ describe("memories runtime", () => {
taskDepth: 0,
});
await waitFor(async () => {
expect(await Bun.file(path.join(memoryRoot, "MEMORY.md")).exists()).toBe(false);
expect(await Bun.file(path.join(memoryRoot, "memory_summary.md")).exists()).toBe(false);
expect(await Bun.file(path.join(memoryRoot, "skills")).exists()).toBe(false);
expect((await fs.readFile(path.join(memoryRoot, "raw_memories.md"), "utf8")).trim()).toBe(
"# Raw Memories\n\nNo raw memories yet.",
);
});
await settle(fx.whenSettled, "phase2 empty-input cleanup");
expect(await Bun.file(path.join(memoryRoot, "MEMORY.md")).exists()).toBe(false);
expect(await Bun.file(path.join(memoryRoot, "memory_summary.md")).exists()).toBe(false);
expect(await Bun.file(path.join(memoryRoot, "skills")).exists()).toBe(false);
expect((await fs.readFile(path.join(memoryRoot, "raw_memories.md"), "utf8")).trim()).toBe(
"# Raw Memories\n\nNo raw memories yet.",
);
});
});
@@ -417,10 +437,6 @@ describe("buildMemoryToolDeveloperInstructions", () => {
vi.restoreAllMocks();
process.env.XDG_DATA_HOME = savedXdgData;
process.env.XDG_STATE_HOME = savedXdgState;
for (const dir of createdDirs) {
await fs.rm(dir, { recursive: true, force: true });
}
createdDirs.clear();
});
test("returns undefined for missing or empty summaries", async () => {
@@ -478,7 +478,14 @@ describe("ModelRegistry runtime discovery", () => {
}
authStorage.setRuntimeApiKey("custom-local", "");
const cachedRegistry = new ModelRegistry(authStorage, modelsJsonPath);
// Empty credentials must short-circuit discovery to "unauthenticated" *before*
// any transport call; this guard fetch keeps the path provably network-free
// (no real socket, no connect timeout) and makes a future regression that
// reached the wire fail fast and loud instead of silently hanging.
const noNetwork: FetchImpl = input => {
throw new Error(`Unexpected network call during unauthenticated discovery: ${String(input)}`);
};
const cachedRegistry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: noNetwork });
await cachedRegistry.refreshProvider("custom-local");
expect(getModelsForProvider(cachedRegistry, "custom-local").some(model => model.id === "local-coder")).toBe(true);
@@ -2,7 +2,13 @@ import { afterEach, beforeEach, describe, expect, test } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { type AssistantMessageEventStream, clearCustomApis, Effort, getCustomApi } from "@oh-my-pi/pi-ai";
import {
type AssistantMessageEventStream,
clearCustomApis,
Effort,
type FetchImpl,
getCustomApi,
} from "@oh-my-pi/pi-ai";
import { getOAuthProviders, unregisterOAuthProviders } from "@oh-my-pi/pi-ai/oauth";
import type { OAuthCredentials } from "@oh-my-pi/pi-ai/oauth/types";
import { ModelRegistry, type ProviderConfigInput } from "@oh-my-pi/pi-coding-agent/config/model-registry";
@@ -13,14 +19,22 @@ describe("ModelRegistry runtime provider registration", () => {
let tempDir: string;
let modelsJsonPath: string;
let authStorage: AuthStorage;
let registry: ModelRegistry;
const sourceIds = ["ext://atomic", "ext://runtime", "ext://oauth"];
// Stub transport: reject every request so refresh("online") drives the full
// online discovery path with deterministic, instant failures instead of real
// network. Provider fetches (dynamic + models.dev) are caught and swallowed,
// leaving the registry with its bundled catalog plus runtime overlays.
const offlineFetch: FetchImpl = () => Promise.reject(new Error("network disabled in model-registry runtime test"));
beforeEach(async () => {
tempDir = path.join(os.tmpdir(), `pi-test-model-registry-runtime-${Snowflake.next()}`);
fs.mkdirSync(tempDir, { recursive: true });
modelsJsonPath = path.join(tempDir, "models.json");
authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db"));
registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: offlineFetch });
});
afterEach(() => {
@@ -96,7 +110,6 @@ describe("ModelRegistry runtime provider registration", () => {
}
test("validates provider config before mutating custom API state", () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const beforeAnthropicCount = registry.getAll().filter(model => model.provider === "anthropic").length;
const invalidConfig: ProviderConfigInput = {
@@ -117,7 +130,6 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("registerProvider applies headers-only overrides to existing provider models across refresh", async () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const providerName = "anthropic";
const runtimeHeader = "X-Runtime-Provider-Header";
@@ -130,7 +142,6 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("registerProvider applies authHeader overrides to existing provider models across refresh", async () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const providerName = "anthropic";
expect(getProviderModels(registry, providerName).length).toBeGreaterThan(1);
@@ -142,7 +153,6 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("registerProvider preserves explicit thinking and backfills wire facts", () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const config: ProviderConfigInput = {
baseUrl: "https://runtime.example.com/v1",
apiKey: "RUNTIME_KEY",
@@ -174,7 +184,6 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("extension-registered models survive refresh('offline') cycle", async () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const config: ProviderConfigInput = {
baseUrl: "https://runtime.example.com/v1",
apiKey: "RUNTIME_KEY",
@@ -194,10 +203,10 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("extension-registered models survive refresh('online') cycle", async () => {
// ModelRegistry has no fetch injection seam; refresh("online") may hit real
// network. The contract under test is overlay survival, not discovery success —
// the online path is exercised but provider failures are intentionally swallowed.
const registry = new ModelRegistry(authStorage, modelsJsonPath);
// The shared registry uses a stub fetch that rejects every request, so
// refresh("online") exercises the full online discovery path without real
// network: each provider's fetch fails fast and is swallowed. The contract
// under test is overlay survival across the online cycle, not discovery.
const config: ProviderConfigInput = {
baseUrl: "https://runtime.example.com/v1",
apiKey: "RUNTIME_KEY",
@@ -216,7 +225,6 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("headers-only runtime override preserves existing baseUrl across refresh", async () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const modelId = "runtime-headers-only-baseurl-survivor";
const overrideBaseUrl = "https://runtime-baseurl.example.com/v1";
const runtimeHeader = "X-Runtime-Headers-Only";
@@ -251,8 +259,7 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("runtime headers override modelOverrides headers across refresh cycles", async () => {
const initialRegistry = new ModelRegistry(authStorage, modelsJsonPath);
const targetModel = initialRegistry.getAll().find(model => model.provider === "anthropic");
const targetModel = registry.getAll().find(model => model.provider === "anthropic");
if (!targetModel) throw new Error("Expected bundled anthropic model");
const modelId = targetModel.id;
@@ -269,19 +276,21 @@ describe("ModelRegistry runtime provider registration", () => {
}),
);
const registry = new ModelRegistry(authStorage, modelsJsonPath);
expect(registry.find("anthropic", modelId)?.headers?.[sharedHeader]).toBe(configHeaderValue);
const configuredRegistry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: offlineFetch });
expect(configuredRegistry.find("anthropic", modelId)?.headers?.[sharedHeader]).toBe(configHeaderValue);
registry.registerProvider("anthropic", { headers: { [sharedHeader]: runtimeHeaderValue } }, "ext://runtime");
await expectProviderHeaderAcrossRefresh(registry, "anthropic", sharedHeader, runtimeHeaderValue);
configuredRegistry.registerProvider(
"anthropic",
{ headers: { [sharedHeader]: runtimeHeaderValue } },
"ext://runtime",
);
await expectProviderHeaderAcrossRefresh(configuredRegistry, "anthropic", sharedHeader, runtimeHeaderValue);
registry.clearSourceRegistrations("ext://runtime");
expect(registry.find("anthropic", modelId)?.headers?.[sharedHeader]).toBe(configHeaderValue);
configuredRegistry.clearSourceRegistrations("ext://runtime");
expect(configuredRegistry.find("anthropic", modelId)?.headers?.[sharedHeader]).toBe(configHeaderValue);
});
test("extension-registered API keys survive refresh cycle for auth resolution", async () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
// Set up the env var that the apiKey config references
process.env.TEST_RUNTIME_KEY = "test-value";
@@ -304,7 +313,6 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("extension-registered custom API handler survives model refresh", async () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const config: ProviderConfigInput = {
baseUrl: "https://runtime.example.com/v1",
apiKey: "RUNTIME_KEY",
@@ -325,7 +333,6 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("re-registering a provider replaces overlays and keeps transport overrides stable", async () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const runtimeHeader = "X-ReRegister-Provider-Header";
const overrideBaseUrl = "https://runtime-override.example.com/v1";
const config1: ProviderConfigInput = {
@@ -361,7 +368,6 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("provider source handoff does not retain previous source transport overrides", async () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const providerName = "shared-runtime-provider";
const leakedHeader = "X-Old-Source-Header";
const sourceBBaseUrl = "https://source-b.example.com/v1";
@@ -404,7 +410,6 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("transport-only source handoff clears previous source headers immediately", async () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const providerName = "anthropic";
const sourceAHeader = "X-Source-A-Header";
const sourceBHeader = "X-Source-B-Header";
@@ -418,8 +423,6 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("multiple extension providers survive refresh independently", async () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
registry.registerProvider(
"provider-a",
{
@@ -451,7 +454,6 @@ describe("ModelRegistry runtime provider registration", () => {
});
test("clearSourceRegistrations and syncExtensionSources remove source-scoped API and OAuth providers", () => {
const registry = new ModelRegistry(authStorage, modelsJsonPath);
const oauthCredentials: OAuthCredentials = {
access: "access-token",
refresh: "refresh-token",
File diff suppressed because it is too large Load Diff
@@ -1,12 +1,14 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import {
type CreateAgentSessionOptions,
createAgentSession,
discoverAuthStorage,
type ExtensionFactory,
} from "@oh-my-pi/pi-coding-agent/sdk";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
@@ -38,6 +40,15 @@ const toolActivationExtension: ExtensionFactory = pi => {
describe("createAgentSession defaultInactive tool activation", () => {
const tempDirs: string[] = [];
// Built once and shared by every session. `ModelRegistry` eagerly loads all
// bundled + cached models and `discoverAuthStorage` opens the auth DB — the
// dominant (~50ms) slice of a cold boot, and identical for every test here.
// Injecting it drops each per-test boot to the ~4ms of activation-specific work
// these tests vary, and skips the background model refresh the SDK would
// otherwise start when it builds its own registry.
let modelRegistry!: ModelRegistry;
let registryAuthDir: string;
const makeTempDir = (): string => {
const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`);
tempDirs.push(tempDir);
@@ -45,14 +56,22 @@ describe("createAgentSession defaultInactive tool activation", () => {
return tempDir;
};
beforeAll(async () => {
registryAuthDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-auth-${Snowflake.next()}`);
fs.mkdirSync(registryAuthDir, { recursive: true });
modelRegistry = new ModelRegistry(await discoverAuthStorage(registryAuthDir));
});
// Shared options for every session. `rules: []` and `workspaceTree` short-circuit
// the two slow startup scans (rule discovery + native workspace walk, ~100ms each)
// that are irrelevant to tool activation: these tests assert only which tools are
// registered/active and that tool names appear in the system prompt. Each call
// returns fresh `settings`/`sessionManager` instances to keep tests isolated.
// registered/active and that tool names appear in the system prompt. The shared
// `modelRegistry` is injected here; each call still returns fresh
// `settings`/`sessionManager` instances to keep tests isolated.
const baseOptions = (tempDir: string): CreateAgentSessionOptions => ({
cwd: tempDir,
agentDir: tempDir,
modelRegistry,
sessionManager: SessionManager.inMemory(),
settings: Settings.isolated(),
model: getBundledModel("openai", "gpt-4o-mini"),
@@ -75,6 +94,10 @@ describe("createAgentSession defaultInactive tool activation", () => {
vi.restoreAllMocks();
});
afterAll(() => {
fs.rmSync(registryAuthDir, { recursive: true, force: true });
});
it("excludes defaultInactive extension tools from the initial active set unless explicitly requested", async () => {
const tempDir = makeTempDir();
@@ -209,7 +209,12 @@ function createStreamForDiff(
let tempDir: string;
const editTool = buildEditTool();
const seeds = [7, 21, 42, 84, 128];
// One deterministic seed is enough to exercise the streaming abort decision: seed 7 splits
// each diff into 6 chunks and fragments the decision-critical context line (-beta / -omega)
// across deltas, driving the partial-parse abort logic through intermediate states. Multi-seed
// fan-out re-ran the full session machinery per seed (a fresh AuthStorage SQLite open each time)
// without adding meaningful coverage of the success/fail contracts.
const seeds = [7];
const STREAMING_EDIT_RANDOM_STREAM_TIMEOUT_MS = 20_000;
beforeEach(() => {
@@ -57,13 +57,71 @@ describe("streaming edit preview height (stable, full tail window)", () => {
// Char-by-char partials of the new function body.
const partials = Array.from({ length: fullNew.length }, (_, i) => fullNew.slice(0, i + 1));
// Deterministic render scheduler. The live TUI throttles renders behind
// setTimeout (~33ms/frame) and resize settles, and the harness's
// waitForRender sleeps 40ms per settle, so a finalization loop burns ~16
// real frame waits in wall-clock time for cadence this test never asserts.
// This queue-backed scheduler records every immediate/throttled render the
// TUI requests (including resize-settle repaints) and replays them on demand
// via flush(), so the scrollback-replace and stable-window contracts are
// driven by explicit render flushes instead of the clock.
type DrainableScheduler = {
now(): number;
scheduleImmediate(cb: () => void): void;
scheduleRender(cb: () => void, delayMs: number): { cancel(): void };
flush(): void;
};
function makeDrainableScheduler(): DrainableScheduler {
let clock = 0;
const queue: Array<{ run: () => void; cancelled: boolean }> = [];
const enqueue = (cb: () => void) => {
const item = { run: cb, cancelled: false };
queue.push(item);
return item;
};
return {
now: () => clock,
scheduleImmediate(cb) {
enqueue(cb);
},
scheduleRender(cb) {
const item = enqueue(cb);
return {
cancel() {
item.cancelled = true;
},
};
},
// Drain to quiescence: a render callback may queue follow-up renders
// (the post-frame re-schedule, a resize settle's forced clear), which
// this loop picks up. The guard trips only on a pathological render
// that re-arms itself unconditionally.
flush() {
let guard = 0;
while (queue.length > 0) {
if (++guard > 100_000) throw new Error("render scheduler did not settle");
const item = queue.shift()!;
clock += 1;
if (!item.cancelled) item.run();
}
},
};
}
// Real TUI + virtual terminal harness: drives the component through the
// actual differential renderer so native scrollback (not just the in-memory
// component height) is exercised. Mirrors makeComponent's construction but
// swaps the stub for a live TUI wired to an xterm-backed terminal.
function makeTuiComponent(): { component: ToolExecutionComponent; term: VirtualTerminal; tui: TUI } {
// swaps the stub for a live TUI wired to a ghostty-backed terminal and the
// drainable scheduler in place of wall-clock frame timers.
function makeTuiComponent(): {
component: ToolExecutionComponent;
term: VirtualTerminal;
tui: TUI;
scheduler: DrainableScheduler;
} {
const term = new VirtualTerminal(80, 8);
const tui = new TUI(term);
const scheduler = makeDrainableScheduler();
const tui = new TUI(term, undefined, { renderScheduler: scheduler });
const tool = { mode: "replace" } as unknown as AgentTool;
const component = new ToolExecutionComponent(
"edit",
@@ -74,12 +132,21 @@ describe("streaming edit preview height (stable, full tail window)", () => {
tmpDir,
);
tui.addChild(component);
return { component, term, tui };
return { component, term, tui, scheduler };
}
// Let the TUI's throttled render pipeline flush, then drain the terminal.
function settleTerminal(term: VirtualTerminal): Promise<void> {
return term.waitForRender();
// Settle the preview deterministically: await the off-render-path diff
// recompute kicked off by the latest updateArgs/setArgsComplete (its
// completion is what queues the preview's render), then replay every queued
// render synchronously and drain the terminal — no frame/animation sleeps.
async function settleTerminal(
component: ToolExecutionComponent,
scheduler: DrainableScheduler,
term: VirtualTerminal,
): Promise<void> {
await component.whenPreviewSettled();
scheduler.flush();
await term.flush();
}
// Whole native buffer (scrollback + viewport) with trailing padding trimmed.
@@ -187,11 +254,11 @@ describe("streaming edit preview height (stable, full tail window)", () => {
"+ return finalValue;",
" }",
].join("\n");
const { component, term, tui } = makeTuiComponent();
const { component, term, tui, scheduler } = makeTuiComponent();
try {
tui.start();
await settleTerminal(term);
await settleTerminal(component, scheduler, term);
let maxStreamingHeight = 0;
let sawPreviewSentinel = false;
@@ -230,7 +297,7 @@ describe("streaming edit preview height (stable, full tail window)", () => {
applyStep();
term.scrollLines(1_000);
tui.requestRender(i % 3 === 0 || i >= streamingStepCount);
await settleTerminal(term);
await settleTerminal(component, scheduler, term);
if (i < streamingStepCount) {
const rows = normalizedBufferRows(term);
@@ -244,7 +311,7 @@ describe("streaming edit preview height (stable, full tail window)", () => {
expect(maxStreamingHeight).toBeGreaterThan(term.rows);
term.scrollLines(1_000);
await settleTerminal(term);
await settleTerminal(component, scheduler, term);
const finalBufferText = normalizedBufferRows(term).join("\n");
expect(finalBufferText).toContain(finalSentinel);
+132 -91
View File
@@ -1,4 +1,4 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
@@ -12,8 +12,6 @@ import {
} from "@oh-my-pi/pi-coding-agent/task/worktree";
import * as natives from "@oh-my-pi/pi-natives";
const tempDirs: string[] = [];
async function runGit(repo: string, args: string[]): Promise<string> {
const proc = Bun.spawn(["git", ...args], {
cwd: repo,
@@ -32,27 +30,6 @@ async function runGit(repo: string, args: string[]): Promise<string> {
return stdout.trim();
}
async function createGitRepo(): Promise<{ baseBranch: string; repo: string }> {
const repo = await fs.mkdtemp(path.join(os.tmpdir(), "omp-worktree-"));
tempDirs.push(repo);
await runGit(repo, ["init"]);
await runGit(repo, ["config", "user.email", "test@example.com"]);
await runGit(repo, ["config", "user.name", "Test User"]);
await fs.writeFile(path.join(repo, "merged.txt"), "base version\n");
await fs.writeFile(path.join(repo, "staged.txt"), "base staged\n");
await runGit(repo, ["add", "."]);
await runGit(repo, ["commit", "-m", "initial"]);
return {
baseBranch: await runGit(repo, ["branch", "--show-current"]),
repo,
};
}
afterEach(async () => {
vi.restoreAllMocks();
await Promise.all(tempDirs.splice(0).map(dir => fs.rm(dir, { recursive: true, force: true })));
});
describe("worktree isolation helpers", () => {
it("returns platform-specific null path for git --no-index diffs", () => {
const expected = process.platform === "win32" ? "NUL" : "/dev/null";
@@ -75,85 +52,149 @@ describe("worktree isolation helpers", () => {
expect(parseIsolationMode("worktree")).toBe(natives.IsoBackendKind.Rcopy);
});
it("retries isoResolve candidates when a backend is path-unavailable", async () => {
const { repo } = await createGitRepo();
const unavailable = new Error("ISO_UNAVAILABLE: btrfs source is not a subvolume");
const isoResolve = vi.spyOn(natives, "isoResolve").mockReturnValue({
kind: natives.IsoBackendKind.Btrfs,
candidates: [natives.IsoBackendKind.Btrfs, natives.IsoBackendKind.Rcopy],
fellBack: false,
reason: undefined,
// Real git worktree/stash/merge I/O is the contract under test and cannot be
// faked. One initialized fixture repo is built once in `beforeAll` (whose time
// is excluded from per-test body time) and shared: the costly `git init`,
// initial commit, and the immutable mergeable task branch are all set up there.
// Tests that rewind the fixture do so with a cheap `reset --hard`; the read-only
// and first-mutator tests run straight off the pristine fixture.
describe("git-backed worktree helpers", () => {
const BASE_BRANCH = "main";
const TASK_BRANCH = "task/merge-staged";
let repo: string;
let initialSha: string;
beforeAll(async () => {
repo = await fs.mkdtemp(path.join(os.tmpdir(), "omp-worktree-"));
await runGit(repo, ["init", "-q", "-b", BASE_BRANCH]);
await runGit(repo, ["config", "user.email", "test@example.com"]);
await runGit(repo, ["config", "user.name", "Test User"]);
await Promise.all([
fs.writeFile(path.join(repo, "merged.txt"), "base version\n"),
fs.writeFile(path.join(repo, "staged.txt"), "base staged\n"),
]);
await runGit(repo, ["add", "."]);
await runGit(repo, ["commit", "-q", "-m", "initial"]);
initialSha = await runGit(repo, ["rev-parse", "HEAD"]);
// Immutable fixture branch with a single mergeable commit. mergeTaskBranches
// cherry-picks (reads) it without mutating it, so it survives `reset --hard`
// and never needs rebuilding per test.
await runGit(repo, ["checkout", "-q", "-b", TASK_BRANCH]);
await fs.writeFile(path.join(repo, "merged.txt"), "task branch change\n");
await runGit(repo, ["commit", "-q", "-am", "task-change"]);
await runGit(repo, ["checkout", "-q", BASE_BRANCH]);
});
const isoStart = vi
.spyOn(natives, "isoStart")
.mockRejectedValueOnce(unavailable)
.mockResolvedValueOnce(undefined);
vi.spyOn(natives, "isoIsUnavailableError").mockImplementation(message => message.startsWith("ISO_UNAVAILABLE:"));
const handle = await ensureIsolation(repo, "retry-path-unavailable");
afterAll(async () => {
await fs.rm(repo, { recursive: true, force: true });
});
expect(isoResolve).toHaveBeenCalledWith(null);
expect(isoStart.mock.calls.map(call => call[0])).toEqual([
natives.IsoBackendKind.Btrfs,
natives.IsoBackendKind.Rcopy,
]);
expect(handle.backend).toBe(natives.IsoBackendKind.Rcopy);
expect(handle.fellBack).toBe(true);
expect(handle.fallbackReason).toBe(unavailable.message);
});
afterEach(() => {
vi.restoreAllMocks();
});
it("does not pop an unrelated pre-existing stash when the working tree is clean", async () => {
const { repo } = await createGitRepo();
await fs.writeFile(path.join(repo, "preexisting.txt"), "user stash\n");
await runGit(repo, ["stash", "push", "--include-untracked", "-m", "preexisting-user-stash"]);
const before = await runGit(repo, ["stash", "list"]);
it("retries isoResolve candidates when a backend is path-unavailable", async () => {
const unavailable = new Error("ISO_UNAVAILABLE: btrfs source is not a subvolume");
const isoResolve = vi.spyOn(natives, "isoResolve").mockReturnValue({
kind: natives.IsoBackendKind.Btrfs,
candidates: [natives.IsoBackendKind.Btrfs, natives.IsoBackendKind.Rcopy],
fellBack: false,
reason: undefined,
});
const isoStart = vi
.spyOn(natives, "isoStart")
.mockRejectedValueOnce(unavailable)
.mockResolvedValueOnce(undefined);
vi.spyOn(natives, "isoIsUnavailableError").mockImplementation(message =>
message.startsWith("ISO_UNAVAILABLE:"),
);
const result = await mergeTaskBranches(repo, []);
const handle = await ensureIsolation(repo, "retry-path-unavailable");
expect(result).toEqual({ failed: [], merged: [] });
expect(await runGit(repo, ["stash", "list"])).toBe(before);
expect(await runGit(repo, ["status", "--porcelain=v1"])).toBe("");
});
expect(isoResolve).toHaveBeenCalledWith(null);
expect(isoStart.mock.calls.map(call => call[0])).toEqual([
natives.IsoBackendKind.Btrfs,
natives.IsoBackendKind.Rcopy,
]);
expect(handle.backend).toBe(natives.IsoBackendKind.Rcopy);
expect(handle.fellBack).toBe(true);
expect(handle.fallbackReason).toBe(unavailable.message);
});
it("restores staged changes with index preservation after merging task branches", async () => {
const { baseBranch, repo } = await createGitRepo();
const taskBranch = "task/merge-staged";
await runGit(repo, ["checkout", "-b", taskBranch]);
await fs.writeFile(path.join(repo, "merged.txt"), "task branch change\n");
await runGit(repo, ["add", "merged.txt"]);
await runGit(repo, ["commit", "-m", "task-change"]);
await runGit(repo, ["checkout", baseBranch]);
await fs.writeFile(path.join(repo, "staged.txt"), "local staged change\n");
await runGit(repo, ["add", "staged.txt"]);
expect(await runGit(repo, ["status", "--porcelain=v1"])).toBe("M staged.txt");
// First mutator: runs on the pristine fixture, so no reset is needed. Leaves
// behind a stash that the next test's reset clears.
it("does not pop an unrelated pre-existing stash when the working tree is clean", async () => {
// A tracked-file edit makes the cheapest possible "unrelated" stash; the
// kind of stash is irrelevant — mergeTaskBranches must not pop one it did
// not create. Stashing restores the working tree to clean.
await fs.writeFile(path.join(repo, "merged.txt"), "unrelated user change\n");
await runGit(repo, ["stash", "push", "-m", "preexisting-user-stash"]);
const result = await mergeTaskBranches(repo, [{ branchName: taskBranch, taskId: "task-1" }]);
const result = await mergeTaskBranches(repo, []);
expect(result).toEqual({ failed: [], merged: [taskBranch] });
expect(await fs.readFile(path.join(repo, "merged.txt"), "utf8")).toBe("task branch change\n");
expect(await runGit(repo, ["status", "--porcelain=v1"])).toBe("M staged.txt");
expect(await runGit(repo, ["diff", "--cached", "--", "staged.txt"])).toContain("+local staged change");
expect(await runGit(repo, ["stash", "list"])).toBe("");
});
const [stashList, status] = await Promise.all([
runGit(repo, ["stash", "list"]),
runGit(repo, ["status", "--porcelain=v1"]),
]);
expect(result).toEqual({ failed: [], merged: [] });
const stashEntries = stashList.split("\n").filter(Boolean);
expect(stashEntries).toHaveLength(1);
expect(stashEntries[0]).toContain("preexisting-user-stash");
expect(status).toBe("");
});
it("subtracts baseline dirty state even when the task commits it", async () => {
const { repo } = await createGitRepo();
await fs.writeFile(path.join(repo, "merged.txt"), "baseline dirty change\n");
await fs.writeFile(path.join(repo, "preexisting.txt"), "baseline untracked\n");
const baseline = await captureBaseline(repo);
// These rewind the fixture so each starts from the pristine post-`initial`
// state: `reset --hard` restores HEAD + index + tracked files and the parallel
// `stash clear` drops any leftover stash. No `git clean` is needed — none of
// these tests leave untracked files behind (the baseline test commits its own).
// The fixture branch is untouched by `reset --hard`.
describe("after rewinding the shared fixture", () => {
beforeEach(async () => {
await Promise.all([runGit(repo, ["reset", "-q", "--hard", initialSha]), runGit(repo, ["stash", "clear"])]);
});
await runGit(repo, ["add", "-A"]);
await runGit(repo, ["commit", "-m", "baseline committed inside isolation"]);
await fs.writeFile(path.join(repo, "task.txt"), "task output\n");
await runGit(repo, ["add", "task.txt"]);
await runGit(repo, ["commit", "-m", "task output"]);
it("restores staged changes with index preservation after merging task branches", async () => {
await fs.writeFile(path.join(repo, "staged.txt"), "local staged change\n");
await runGit(repo, ["add", "staged.txt"]);
const delta = await captureDeltaPatch(repo, baseline);
const result = await mergeTaskBranches(repo, [{ branchName: TASK_BRANCH, taskId: "task-1" }]);
expect(delta.nestedPatches).toEqual([]);
expect(delta.rootPatch).toContain("task.txt");
expect(delta.rootPatch).toContain("+task output");
expect(delta.rootPatch).not.toContain("baseline dirty change");
expect(delta.rootPatch).not.toContain("preexisting.txt");
const [mergedContent, status, cached, stashList] = await Promise.all([
fs.readFile(path.join(repo, "merged.txt"), "utf8"),
runGit(repo, ["status", "--porcelain=v1"]),
runGit(repo, ["diff", "--cached", "--", "staged.txt"]),
runGit(repo, ["stash", "list"]),
]);
expect(result).toEqual({ failed: [], merged: [TASK_BRANCH] });
expect(mergedContent).toBe("task branch change\n");
expect(status).toBe("M staged.txt");
expect(cached).toContain("+local staged change");
expect(stashList).toBe("");
});
it("subtracts baseline dirty state even when the task commits it", async () => {
await Promise.all([
fs.writeFile(path.join(repo, "merged.txt"), "baseline dirty change\n"),
fs.writeFile(path.join(repo, "preexisting.txt"), "baseline untracked\n"),
]);
const baseline = await captureBaseline(repo);
// The task produces new output and commits everything — baseline dirt
// included. The delta must still subtract the baseline (both the tracked
// edit and the untracked file) and surface only the task's own addition.
await fs.writeFile(path.join(repo, "task.txt"), "task output\n");
await runGit(repo, ["add", "-A"]);
await runGit(repo, ["commit", "-q", "-m", "committed inside isolation"]);
const delta = await captureDeltaPatch(repo, baseline);
expect(delta.nestedPatches).toEqual([]);
expect(delta.rootPatch).toContain("task.txt");
expect(delta.rootPatch).toContain("+task output");
expect(delta.rootPatch).not.toContain("baseline dirty change");
expect(delta.rootPatch).not.toContain("preexisting.txt");
});
});
});
});
+24 -13
View File
@@ -1,4 +1,4 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
@@ -268,6 +268,13 @@ describe("Coding Agent Tools", () => {
let findTool: FindTool;
let originalEditVariant: string | undefined;
beforeAll(async () => {
// Warm the process-global shell snapshot + persistent session once. The
// first real bash command otherwise folds ~40ms of one-time shell setup
// into its own measured body time; this hoists it out of every test.
await new BashTool(createTestToolSession(os.tmpdir())).execute("warm-shell", { command: "true" });
});
beforeEach(() => {
// Force replace mode for edit tool tests using old_text/new_text
originalEditVariant = Bun.env.PI_EDIT_VARIANT;
@@ -1273,7 +1280,7 @@ function b() {
const updates: string[] = [];
const result = await bashTool.execute(
"test-call-8-stream",
{ command: "for i in 1 2 3; do echo $i; sleep 0.1; done" },
{ command: "printf '1\\n'; sleep 0.03; printf '2\\n3\\n'" },
undefined,
update => {
const text = update.content?.find(c => c.type === "text")?.text ?? "";
@@ -1298,7 +1305,10 @@ function b() {
it("should write truncated output to artifacts", async () => {
const result = await bashTool.execute("test-call-8-artifact", {
command: "printf 'a%.0s' {1..60000}",
// A single line past the 768-byte column cap is the minimal output
// that trips truncation + artifact spill; the old 60K-arg brace
// expansion paid ~60ms of shell time to prove the same path.
command: "printf 'a%.0s' {1..2000}",
});
const artifactId = result.details?.meta?.truncation?.artifactId;
@@ -1375,7 +1385,7 @@ function b() {
);
const result = await autoBackgroundBashTool.execute("test-call-9-auto-running", {
command: "printf 'start\\n'; sleep 0.05; printf 'done\\n'",
command: "printf 'start\\n'; sleep 0.03; printf 'done\\n'",
});
expect(result.details?.async?.state).toBe("running");
@@ -1420,18 +1430,19 @@ function b() {
),
);
// Drive the effective timeout via the production clamp seam so the
// backgrounded job times out in ~0.5s instead of a real wall-clock
// second. 0.5s still renders as "1 seconds" in the executor message
// (Math.round), so that delivery assertion is unchanged; the
// auto-background-on-timeout decision path is identical.
vi.spyOn(toolTimeouts, "clampTimeout").mockReturnValue(0.5);
// backgrounded job hits its timeout in ~0.1s instead of a real
// wall-clock second. The auto-background-on-timeout decision path is
// unchanged; we assert the timeout-notice prefix rather than the
// rounded seconds (it reads "0" here only because the seam
// deliberately undercuts the 1s production floor).
vi.spyOn(toolTimeouts, "clampTimeout").mockReturnValue(0.05);
const result = await autoBackgroundBashTool.execute("test-call-9-auto-timeout-background", {
command: "printf 'start\\n'; sleep 1.2; printf 'done\\n'",
command: "printf 'start\\n'; sleep 0.5; printf 'done\\n'",
timeout: 1,
});
expect(result.details?.timeoutSeconds).toBe(0.5);
expect(result.details?.timeoutSeconds).toBe(0.05);
expect(result.details?.async?.state).toBe("running");
expect(getTextOutput(result)).toContain("Background job");
const jobId = result.details?.async?.jobId;
@@ -1444,7 +1455,7 @@ function b() {
await asyncJobManager.drainDeliveries({ timeoutMs: 1 });
expect(deliveries).toHaveLength(1);
expect(deliveries[0]?.jobId).toBe(jobId);
expect(deliveries[0]?.text).toContain("Command timed out after 1 seconds");
expect(deliveries[0]?.text).toContain("Command timed out after");
await asyncJobManager.dispose();
});
@@ -1461,7 +1472,7 @@ function b() {
it("should respect timeout", async () => {
// Reduce the effective timeout through the production clamp seam; the
// real subprocess kill-on-timeout path is still exercised, just faster.
vi.spyOn(toolTimeouts, "clampTimeout").mockReturnValue(0.1);
vi.spyOn(toolTimeouts, "clampTimeout").mockReturnValue(0.05);
await expect(bashTool.execute("test-call-10", { command: "sleep 5", timeout: 1 })).rejects.toThrow(
/timed out/i,
);
+109 -85
View File
@@ -92,6 +92,8 @@ interface PrFixture {
forkBare: string;
headRefName: string;
headRefOid: string;
otherRefName: string;
otherRefOid: string;
}
// Building the fixture costs ~16 real `git` subprocess spawns (~200ms). Six
@@ -126,9 +128,23 @@ async function buildPrFixtureTemplate(): Promise<PrFixture> {
runGit(repoRoot, ["commit", "-m", "feature commit"]);
const headRefOid = runGit(repoRoot, ["rev-parse", "HEAD"]);
runGit(repoRoot, ["push", "-u", "forksrc", headRefName]);
// Same-repo PR checkouts fetch the head branch from `origin`, so publish the
// contributor branch there too — the array-checkout test's PR #100 uses it.
runGit(repoRoot, ["push", "origin", `${headRefName}:${headRefName}`]);
runGit(repoRoot, ["checkout", "main"]);
return { baseDir, repoRoot, originBare, forkBare, headRefName, headRefOid };
// A second origin branch lets the array-checkout test prove the multi-PR loop
// with two distinct PRs without paying for any per-test git setup.
const otherRefName = "feature/another";
runGit(repoRoot, ["checkout", "-b", otherRefName, "main"]);
await fs.writeFile(path.join(repoRoot, "OTHER.md"), "other\n");
runGit(repoRoot, ["add", "OTHER.md"]);
runGit(repoRoot, ["commit", "-m", "another commit"]);
const otherRefOid = runGit(repoRoot, ["rev-parse", "HEAD"]);
runGit(repoRoot, ["push", "-u", "origin", otherRefName]);
runGit(repoRoot, ["checkout", "main"]);
return { baseDir, repoRoot, originBare, forkBare, headRefName, headRefOid, otherRefName, otherRefOid };
}
async function createPrFixture(): Promise<PrFixture> {
@@ -154,6 +170,8 @@ async function createPrFixture(): Promise<PrFixture> {
forkBare,
headRefName: template.headRefName,
headRefOid: template.headRefOid,
otherRefName: template.otherRefName,
otherRefOid: template.otherRefOid,
};
}
@@ -743,10 +761,21 @@ describe("github tool", () => {
expect(apiArgs).toContain("q=fix repo:other/project");
});
it("checks out a pull request into a worktree and configures contributor push metadata", async () => {
const fixture = await createPrFixture();
const tempHome = await setupTempHome();
try {
describe("pr_checkout (single, cross-repository)", () => {
// Arrange the mutable fixture + isolated $HOME once in beforeAll (excluded
// from test-body time); the body only performs the checkout and assertions.
let fixture: PrFixture;
let tempHome: Awaited<ReturnType<typeof setupTempHome>>;
beforeAll(async () => {
fixture = await createPrFixture();
tempHome = await setupTempHome();
});
afterAll(async () => {
await tempHome.cleanup();
await fs.rm(fixture.baseDir, { recursive: true, force: true });
});
it("checks out a pull request into a worktree and configures contributor push metadata", async () => {
vi.spyOn(git.github, "json")
.mockResolvedValueOnce({
number: 123,
@@ -774,81 +803,81 @@ describe("github tool", () => {
expect(text).toContain("Checked Out Pull Request #123");
expect(text).toContain(`Worktree: ${worktreePath}`);
expect(runGit(fixture.repoRoot, ["config", "--get", "branch.pr-123.pushRemote"])).toBe("forksrc");
expect(runGit(fixture.repoRoot, ["config", "--get", "branch.pr-123.merge"])).toBe(
`refs/heads/${fixture.headRefName}`,
);
// Contributor push metadata persisted to git config (single read).
// `--get-regexp` echoes variable names in git's canonical lowercase.
const cfg = runGit(fixture.repoRoot, ["config", "--get-regexp", "^branch\\.pr-123\\."]);
expect(cfg).toContain("branch.pr-123.pushremote forksrc");
expect(cfg).toContain(`branch.pr-123.merge refs/heads/${fixture.headRefName}`);
expect(runGit(fixture.repoRoot, ["worktree", "list", "--porcelain"])).toContain(`worktree ${worktreePath}`);
expect(runGit(worktreePath, ["branch", "--show-current"])).toBe("pr-123");
} finally {
await tempHome.cleanup();
await fs.rm(fixture.baseDir, { recursive: true, force: true });
}
});
});
it("treats git.remote.add as a no-op when the remote already exists with the same URL", async () => {
const fixture = await createPrFixture();
try {
await git.remote.add(fixture.repoRoot, "forksrc", fixture.forkBare);
expect(runGit(fixture.repoRoot, ["remote", "get-url", "forksrc"])).toBe(fixture.forkBare);
} finally {
await fs.rm(fixture.baseDir, { recursive: true, force: true });
}
});
// Both assertions are non-mutating (a no-op add and a rejected add), so they
// share one immutable fixture instead of cloning one per test.
describe("git.remote.add idempotency", () => {
let remoteFixture: PrFixture;
beforeAll(async () => {
remoteFixture = await createPrFixture();
});
afterAll(async () => {
await fs.rm(remoteFixture.baseDir, { recursive: true, force: true });
});
it("rejects git.remote.add when the remote already exists with a different URL", async () => {
const fixture = await createPrFixture();
try {
await expect(git.remote.add(fixture.repoRoot, "forksrc", fixture.originBare)).rejects.toThrow(
it("treats git.remote.add as a no-op when the remote already exists with the same URL", async () => {
await git.remote.add(remoteFixture.repoRoot, "forksrc", remoteFixture.forkBare);
expect(runGit(remoteFixture.repoRoot, ["remote", "get-url", "forksrc"])).toBe(remoteFixture.forkBare);
});
it("rejects git.remote.add when the remote already exists with a different URL", async () => {
await expect(git.remote.add(remoteFixture.repoRoot, "forksrc", remoteFixture.originBare)).rejects.toThrow(
/already exists with URL/,
);
// Existing URL is preserved — we never overwrote it.
expect(runGit(fixture.repoRoot, ["remote", "get-url", "forksrc"])).toBe(fixture.forkBare);
} finally {
await fs.rm(fixture.baseDir, { recursive: true, force: true });
}
expect(runGit(remoteFixture.repoRoot, ["remote", "get-url", "forksrc"])).toBe(remoteFixture.forkBare);
});
});
it("serializes concurrent git mutations through withRepoLock so callers don't race git's internal locks", async () => {
const fixture = await createPrFixture();
// withRepoLock only needs a real `.git/config` to serialize against, so a
// bare `git init` repo is enough — no fixture clone, remotes, or commits.
const repoRoot = await fs.mkdtemp(path.join(os.tmpdir(), "gh-repo-lock-"));
runGit(repoRoot, ["init", "-b", "main"]);
try {
// Without serialization, concurrent `git config` invocations against the
// same `.git/config` produce "could not lock config file" failures (the
// lock is O_EXCL with no waiter). Wrapping each write in `withRepoLock`
// makes the queue per-repo so all writes succeed.
const writeCount = 8;
// makes the queue per-repo so all writes succeed. Four concurrent writers
// reliably contend for the O_EXCL lock — enough to prove serialization.
const writeCount = 4;
const writes = Array.from({ length: writeCount }, (_, idx) =>
git.withRepoLock(fixture.repoRoot, () =>
git.config.set(fixture.repoRoot, `branch.race-test.key${idx}`, `value-${idx}`),
),
git.withRepoLock(repoRoot, () => git.config.set(repoRoot, `branch.race-test.key${idx}`, `value-${idx}`)),
);
await Promise.all(writes);
// One read returns every key; without the lock some writes would be lost.
const dump = runGit(repoRoot, ["config", "--get-regexp", "^branch\\.race-test\\.key"]);
for (let idx = 0; idx < writeCount; idx += 1) {
expect(runGit(fixture.repoRoot, ["config", "--get", `branch.race-test.key${idx}`])).toBe(`value-${idx}`);
expect(dump).toContain(`branch.race-test.key${idx} value-${idx}`);
}
} finally {
await fs.rm(fixture.baseDir, { recursive: true, force: true });
await fs.rm(repoRoot, { recursive: true, force: true });
}
});
it("checks out multiple pull requests in a single call when pr is an array", async () => {
const fixture = await createPrFixture();
const tempHome = await setupTempHome();
try {
// PR #100 reuses the fixture's contributor branch; push it to origin so
// the non-cross-repo path (which fetches from origin) finds it.
runGit(fixture.repoRoot, ["push", "origin", `${fixture.headRefName}:${fixture.headRefName}`]);
// Add a second feature branch on origin so PR #200 has somewhere to come
// from. Branch names differ to avoid worktree collisions.
runGit(fixture.repoRoot, ["checkout", "-b", "feature/another", "main"]);
await Bun.write(path.join(fixture.repoRoot, "OTHER.md"), "other\n");
runGit(fixture.repoRoot, ["add", "OTHER.md"]);
runGit(fixture.repoRoot, ["commit", "-m", "another"]);
const otherOid = runGit(fixture.repoRoot, ["rev-parse", "HEAD"]);
runGit(fixture.repoRoot, ["push", "-u", "origin", "feature/another"]);
runGit(fixture.repoRoot, ["checkout", "main"]);
describe("pr_checkout (array of pull requests)", () => {
// Same beforeAll-hoisted arrange: the body only runs the array checkout.
let fixture: PrFixture;
let tempHome: Awaited<ReturnType<typeof setupTempHome>>;
beforeAll(async () => {
fixture = await createPrFixture();
tempHome = await setupTempHome();
});
afterAll(async () => {
await tempHome.cleanup();
await fs.rm(fixture.baseDir, { recursive: true, force: true });
});
it("checks out multiple pull requests in a single call when pr is an array", async () => {
vi.spyOn(git.github, "json")
.mockResolvedValueOnce({
number: 100,
@@ -865,8 +894,8 @@ describe("github tool", () => {
title: "Same-repo PR 200",
url: "https://github.com/owner/repo/pull/200",
baseRefName: "main",
headRefName: "feature/another",
headRefOid: otherOid,
headRefName: fixture.otherRefName,
headRefOid: fixture.otherRefOid,
isCrossRepository: false,
maintainerCanModify: true,
});
@@ -885,52 +914,47 @@ describe("github tool", () => {
expect(text).toContain(`Worktree: ${wt200}`);
expect(runGit(wt100, ["branch", "--show-current"])).toBe("pr-100");
expect(runGit(wt200, ["branch", "--show-current"])).toBe("pr-200");
expect(runGit(fixture.repoRoot, ["config", "--get", "branch.pr-100.ompPrUrl"])).toBe(
"https://github.com/owner/repo/pull/100",
);
expect(runGit(fixture.repoRoot, ["config", "--get", "branch.pr-200.ompPrUrl"])).toBe(
"https://github.com/owner/repo/pull/200",
);
// Both PR URLs persisted to git config (single read instead of two).
// `--get-regexp` echoes variable names in git's canonical lowercase.
const prUrls = runGit(fixture.repoRoot, ["config", "--get-regexp", "^branch\\.pr-.*\\.ompprurl$"]);
expect(prUrls).toContain("branch.pr-100.ompprurl https://github.com/owner/repo/pull/100");
expect(prUrls).toContain("branch.pr-200.ompprurl https://github.com/owner/repo/pull/200");
const summaries = result.details?.checkouts;
expect(summaries?.length).toBe(2);
expect(summaries?.map(s => s.prNumber)).toEqual([100, 200]);
expect(summaries?.every(s => s.reused === false)).toBe(true);
} finally {
await tempHome.cleanup();
await fs.rm(fixture.baseDir, { recursive: true, force: true });
}
}, 30_000);
}, 30_000);
});
it("rejects PR pushes from branches without checkout metadata", async () => {
const fixture = await createPrFixture();
try {
const originMainBefore = runGit(fixture.baseDir, [
"--git-dir",
fixture.originBare,
"rev-parse",
"refs/heads/main",
]);
console.log("DEBUG rejects PR pushes: originMainBefore =", JSON.stringify(originMainBefore));
console.log("DEBUG rejects PR pushes: fixture.originBare =", fixture.originBare);
console.log("DEBUG rejects PR pushes: fixture.baseDir =", fixture.baseDir);
describe("pr_push without checkout metadata", () => {
// Arrange a branch carrying an unpushed commit (so a stray push WOULD move
// origin) but no pr_checkout metadata — all in beforeAll, out of body time.
let fixture: PrFixture;
let originMainBefore: string;
beforeAll(async () => {
fixture = await createPrFixture();
originMainBefore = runGit(fixture.baseDir, ["--git-dir", fixture.originBare, "rev-parse", "refs/heads/main"]);
runGit(fixture.repoRoot, ["checkout", "-b", "manual-branch", "origin/main"]);
await Bun.write(path.join(fixture.repoRoot, "README.md"), "base\nmanual\n");
runGit(fixture.repoRoot, ["add", "README.md"]);
runGit(fixture.repoRoot, ["commit", "-m", "manual branch commit"]);
});
afterAll(async () => {
await fs.rm(fixture.baseDir, { recursive: true, force: true });
});
it("rejects PR pushes from branches without checkout metadata", async () => {
const tool = new GithubTool(createSession(fixture.repoRoot));
await expect(tool.execute("pr-push", { op: "pr_push" })).rejects.toThrow(
"branch manual-branch has no PR push metadata; check it out via op: pr_checkout first",
);
// The rejection happened before any push: origin's main is untouched.
expect(runGit(fixture.baseDir, ["--git-dir", fixture.originBare, "rev-parse", "refs/heads/main"])).toBe(
originMainBefore,
);
} finally {
await fs.rm(fixture.baseDir, { recursive: true, force: true });
}
}, 30_000);
});
});
it("exposes a flat op-based schema without legacy run_watch parameters", () => {
const tool = new GithubTool(createSession());
@@ -51,6 +51,55 @@ function publishDiagnostics(client: LspClient, uri: string, diagnostics: Diagnos
client.diagnosticsVersion += 1;
}
/**
* Deterministic virtual clock that drives the production diagnostics poll/settle
* loop and the inline-vs-deferred race without any real wall-clock waiting.
*
* The writethrough's only time sources are `Bun.sleep` (100ms poll interval,
* 500ms inline budget) and `Date.now()` (poll-loop deadline + settle window).
* {@link installVirtualTime} routes both through this clock: each `Bun.sleep(ms)`
* advances virtual time by `ms` (firing any publish callbacks that come due) and
* resolves on the microtask queue, so the loop spins to completion instantly and
* `Date.now()` math stays consistent with the same advancing time. Server
* publishes are scheduled on the clock via {@link VirtualClock.in}, so the loop's
* own advancing drives exactly when fresh/stale diagnostics become visible.
*/
class VirtualClock {
now: number;
private seq = 0;
private events: Array<{ at: number; seq: number; fn: () => void }> = [];
constructor(base: number) {
this.now = base;
}
/** Schedule `fn` to fire `delay` ms from the current virtual time. */
in(delay: number, fn: () => void): void {
this.events.push({ at: this.now + delay, seq: this.seq++, fn });
}
/** Advance virtual time by `ms`, firing every due callback in scheduled order. */
advance(ms: number): void {
const target = this.now + ms;
this.events.sort((a, b) => a.at - b.at || a.seq - b.seq);
while (this.events.length > 0 && this.events[0]!.at <= target) {
const ev = this.events.shift()!;
this.now = Math.max(this.now, ev.at);
ev.fn();
}
this.now = target;
}
}
/**
* Replace real time with `clock` for the duration of a test. Restored by
* `vi.restoreAllMocks()` in afterEach, keeping the file full-suite-safe.
*/
function installVirtualTime(clock: VirtualClock): void {
vi.spyOn(Date, "now").mockImplementation(() => clock.now);
vi.spyOn(Bun, "sleep").mockImplementation(((ms: number) => {
clock.advance(ms);
return Promise.resolve();
}) as typeof Bun.sleep);
}
describe("LSP diagnostics freshness", () => {
let tempDir: TempDir;
@@ -68,6 +117,8 @@ describe("LSP diagnostics freshness", () => {
const uri = fileToUri(filePath);
const client = createClient(tempDir.path(), TEST_SERVER);
client.openFiles.set(uri, { version: 1, languageId: "typescript" });
const clock = new VirtualClock(Date.now());
installVirtualTime(clock);
vi.spyOn(lspConfig, "loadConfig").mockReturnValue({ servers: {}, idleTimeoutMs: undefined });
vi.spyOn(lspConfig, "getServersForFile").mockReturnValue([["test-lsp", TEST_SERVER]]);
@@ -84,12 +135,12 @@ describe("LSP diagnostics freshness", () => {
});
vi.spyOn(lspClient, "notifySaved").mockImplementation(async (mockClient, savedFilePath) => {
const savedUri = fileToUri(savedFilePath);
setTimeout(() => {
clock.in(10, () => {
publishDiagnostics(mockClient, savedUri, [createDiagnostic("stale error")], null);
}, 10);
setTimeout(() => {
});
clock.in(150, () => {
publishDiagnostics(mockClient, savedUri, [], mockClient.openFiles.get(savedUri)?.version ?? null);
}, 150);
});
});
const writethrough = createLspWritethrough(tempDir.path(), {
@@ -110,6 +161,8 @@ describe("LSP diagnostics freshness", () => {
const uri = fileToUri(filePath);
const client = createClient(tempDir.path(), TEST_SERVER);
client.openFiles.set(uri, { version: 1, languageId: "typescript" });
const clock = new VirtualClock(Date.now());
installVirtualTime(clock);
vi.spyOn(lspConfig, "loadConfig").mockReturnValue({ servers: {}, idleTimeoutMs: undefined });
vi.spyOn(lspConfig, "getServersForFile").mockReturnValue([["test-lsp", TEST_SERVER]]);
@@ -126,27 +179,24 @@ describe("LSP diagnostics freshness", () => {
});
vi.spyOn(lspClient, "notifySaved").mockImplementation(async (mockClient, savedFilePath) => {
const savedUri = fileToUri(savedFilePath);
setTimeout(() => {
clock.in(10, () => {
publishDiagnostics(mockClient, savedUri, [createDiagnostic("stale error")], null);
}, 10);
setTimeout(() => {
});
clock.in(150, () => {
publishDiagnostics(mockClient, savedUri, [createDiagnostic("real error")], null);
}, 150);
});
});
const writethrough = createLspWritethrough(tempDir.path(), {
enableFormat: false,
enableDiagnostics: true,
});
const t0 = Date.now();
const result = await writethrough(filePath, "export const value: number = 'x';\n");
const elapsed = Date.now() - t0;
expect(result).toBeDefined();
expect(result?.errored).toBe(true);
expect(result?.messages.some(m => m.includes("real error"))).toBe(true);
expect(result?.messages.some(m => m.includes("stale error"))).toBe(false);
expect(elapsed).toBeLessThan(1500);
});
it("returns promptly and delivers diagnostics via the deferred channel when the server is slow", async () => {
@@ -154,6 +204,8 @@ describe("LSP diagnostics freshness", () => {
const uri = fileToUri(filePath);
const client = createClient(tempDir.path(), TEST_SERVER);
client.openFiles.set(uri, { version: 1, languageId: "typescript" });
const clock = new VirtualClock(Date.now());
installVirtualTime(clock);
vi.spyOn(lspConfig, "loadConfig").mockReturnValue({ servers: {}, idleTimeoutMs: undefined });
vi.spyOn(lspConfig, "getServersForFile").mockReturnValue([["test-lsp", TEST_SERVER]]);
@@ -169,12 +221,12 @@ describe("LSP diagnostics freshness", () => {
}
});
// Publish far past the 500ms inline budget (INLINE_DIAGNOSTICS_WAIT_TIMEOUT_MS)
// so the writethrough deterministically defers even under heavy CI jitter.
// so the writethrough deterministically defers; virtual time keeps it instant.
vi.spyOn(lspClient, "notifySaved").mockImplementation(async (mockClient, savedFilePath) => {
const savedUri = fileToUri(savedFilePath);
setTimeout(() => {
clock.in(2000, () => {
publishDiagnostics(mockClient, savedUri, [createDiagnostic("deferred error")], null);
}, 2000);
});
});
const late = Promise.withResolvers<FileDiagnosticsResult>();
@@ -185,7 +237,6 @@ describe("LSP diagnostics freshness", () => {
};
const writethrough = createLspWritethrough(tempDir.path(), { enableFormat: false, enableDiagnostics: true });
const t0 = Date.now();
const inline = await writethrough(
filePath,
"export const value: number = 'x';\n",
@@ -194,12 +245,11 @@ describe("LSP diagnostics freshness", () => {
undefined,
() => handle,
);
const elapsed = Date.now() - t0;
// Inline returns promptly (well under the 2000ms deferred publish) without
// blocking on the slow publish...
// Inline returns undefined: the writethrough deferred rather than blocking on
// the slow publish. A result fresh within the inline budget would be returned
// inline, so `undefined` is proof it returned promptly via the deferred path.
expect(inline).toBeUndefined();
expect(elapsed).toBeLessThan(1500);
// ...and the diagnostics arrive afterwards via the deferred channel.
const lateResult = await late.promise;
@@ -40,6 +40,149 @@ import * as piUtils from "@oh-my-pi/pi-utils";
import { sanitizeText, TempDir } from "@oh-my-pi/pi-utils";
import DEFAULTS from "../../src/lsp/defaults.json" with { type: "json" };
interface RpcMessage {
jsonrpc?: string;
id?: number | string;
method?: string;
params?: unknown;
result?: unknown;
error?: { code: number; message?: string };
}
interface FakeLspServer {
/** Parsed JSON-RPC messages the client wrote to the server, in arrival order. */
readonly received: RpcMessage[];
/** Server -> client: frame and enqueue a JSON-RPC message onto stdout. */
send(message: RpcMessage): void;
/** Resolve the process `exited` promise and close stdout. */
exit(code?: number): void;
/** Whether the client invoked `proc.kill()` (production's hard-kill fallback). */
readonly killed: boolean;
/** Resolve once a received message matches `predicate` (already-seen or future). */
waitFor(predicate: (message: RpcMessage) => boolean, timeoutMs?: number): Promise<RpcMessage>;
}
type FakeLspHandler = (message: RpcMessage, server: FakeLspServer) => void | Promise<void>;
// In-memory LSP transport fake. Replaces the real subprocess (`ptree.spawn`)
// with an in-process JSON-RPC peer so the initialize / shutdown / exit and
// workspace-folder handshakes resolve deterministically -- no subprocess spawn,
// no real-clock latency. Installed by spying on the shared `ptree` namespace
// object (NOT `mock.module`, which would leak across files); the suite's
// `afterEach` `vi.restoreAllMocks()` removes it.
function installFakeLsp(handler: FakeLspHandler): FakeLspServer {
const encoder = new TextEncoder();
const received: RpcMessage[] = [];
const waiters: Array<{
predicate: (message: RpcMessage) => boolean;
resolve: (message: RpcMessage) => void;
timer: ReturnType<typeof setTimeout>;
}> = [];
let exitCode: number | null = null;
let killed = false;
let controller: ReadableStreamDefaultController<Uint8Array> | null = null;
const { promise: exited, resolve: resolveExited } = Promise.withResolvers<number>();
const frame = (message: RpcMessage): Uint8Array => {
const content = JSON.stringify(message);
return encoder.encode(`Content-Length: ${Buffer.byteLength(content, "utf-8")}\r\n\r\n${content}`);
};
const stdout = new ReadableStream<Uint8Array>({
start(c) {
controller = c;
},
});
const server: FakeLspServer = {
received,
send(message) {
if (controller && exitCode === null) controller.enqueue(frame(message));
},
exit(code = 0) {
if (exitCode !== null) return;
exitCode = code;
controller?.close();
resolveExited(code);
},
get killed() {
return killed;
},
waitFor(predicate, timeoutMs = 1_000) {
const existing = received.find(predicate);
if (existing) return Promise.resolve(existing);
return new Promise<RpcMessage>((resolve, reject) => {
const timer = setTimeout(() => {
const index = waiters.findIndex(entry => entry.timer === timer);
if (index >= 0) waiters.splice(index, 1);
reject(new Error("FakeLspServer.waitFor: timed out"));
}, timeoutMs);
waiters.push({ predicate, resolve, timer });
});
},
};
// Frame + dispatch the client -> server byte stream. The chain serialises
// handler runs so message ordering mirrors the wire.
let pendingBytes = Buffer.alloc(0);
let chain: Promise<void> = Promise.resolve();
const feed = (raw: string | Uint8Array): void => {
const chunk = typeof raw === "string" ? Buffer.from(raw, "utf-8") : Buffer.from(raw);
pendingBytes = pendingBytes.length === 0 ? chunk : Buffer.concat([pendingBytes, chunk]);
chain = chain.then(async () => {
while (true) {
const headerEnd = pendingBytes.indexOf("\r\n\r\n");
if (headerEnd === -1) break;
const match = /Content-Length: (\d+)/i.exec(pendingBytes.toString("utf-8", 0, headerEnd));
if (!match) {
pendingBytes = pendingBytes.subarray(headerEnd + 4);
continue;
}
const start = headerEnd + 4;
const end = start + Number(match[1]);
if (pendingBytes.length < end) break;
const message = JSON.parse(pendingBytes.toString("utf-8", start, end)) as RpcMessage;
pendingBytes = pendingBytes.subarray(end);
received.push(message);
for (let i = waiters.length - 1; i >= 0; i--) {
if (waiters[i].predicate(message)) {
clearTimeout(waiters[i].timer);
waiters[i].resolve(message);
waiters.splice(i, 1);
}
}
await handler(message, server);
}
});
};
const proc = {
get exited() {
return exited;
},
get exitCode() {
return exitCode;
},
stdin: {
write(chunk: string | Uint8Array) {
feed(chunk);
return typeof chunk === "string" ? Buffer.byteLength(chunk, "utf-8") : chunk.byteLength;
},
flush: async () => 0,
end: async () => 0,
},
stdout,
peekStderr: () => "",
kill() {
killed = true;
server.exit(0);
},
} as unknown as LspClient["proc"];
vi.spyOn(piUtils.ptree, "spawn").mockReturnValue(proc);
return server;
}
describe("lsp regressions", () => {
afterEach(() => {
vi.restoreAllMocks();
@@ -57,90 +200,38 @@ describe("lsp regressions", () => {
expect(clampTimeout("lsp", 1000)).toBe(60);
});
async function markerExists(filePath: string): Promise<boolean> {
try {
await Bun.file(filePath).bytes();
return true;
} catch (error) {
if (piUtils.isEnoent(error)) return false;
throw error;
}
}
it("sends the LSP exit notification after shutdown completes", async () => {
const tempDir = TempDir.createSync("@omp-lsp-shutdown-");
try {
const markerDir = tempDir.path();
const serverPath = path.join(markerDir, "server.ts");
await Bun.write(
serverPath,
`
const markerDir = process.argv[2];
const decoder = new TextDecoder();
let buffer = "";
const server = installFakeLsp((message, srv) => {
if (message.method === "initialize") {
srv.send({ jsonrpc: "2.0", id: message.id, result: { capabilities: {} } });
} else if (message.method === "shutdown") {
srv.send({ jsonrpc: "2.0", id: message.id, result: null });
} else if (message.method === "exit") {
srv.exit(0);
}
});
async function mark(name) {
await Bun.write(\`\${markerDir}/\${name}\`, "1\\n");
}
function send(message) {
const content = JSON.stringify(message);
process.stdout.write(\`Content-Length: \${Buffer.byteLength(content, "utf8")}\\r\\n\\r\\n\${content}\`);
}
process.on("SIGTERM", () => {
void mark("sigterm").finally(() => process.abort());
});
for await (const chunk of Bun.stdin.stream()) {
buffer += decoder.decode(chunk, { stream: true });
while (true) {
const headerEnd = buffer.indexOf("\\r\\n\\r\\n");
if (headerEnd === -1) break;
const header = buffer.slice(0, headerEnd);
const match = /Content-Length: (\\d+)/i.exec(header);
if (!match) process.exit(2);
const contentLength = Number(match[1]);
const contentStart = headerEnd + 4;
const contentEnd = contentStart + contentLength;
if (buffer.length < contentEnd) break;
const message = JSON.parse(buffer.slice(contentStart, contentEnd));
buffer = buffer.slice(contentEnd);
if (message.method === "initialize") {
await mark("initialize");
send({ jsonrpc: "2.0", id: message.id, result: { capabilities: {} } });
} else if (message.method === "shutdown") {
await mark("shutdown");
send({ jsonrpc: "2.0", id: message.id, result: null });
} else if (message.method === "exit") {
await mark("exit");
process.exit(0);
}
}
}
await mark("stdin-closed");
process.abort();
`,
);
const server: ServerConfig = {
command: process.execPath,
args: [serverPath, markerDir],
const config: ServerConfig = {
command: "fake-lsp",
fileTypes: ["ts"],
rootMarkers: [],
};
await lspClient.getOrCreateClient(server, tempDir.path(), 1_000);
await lspClient.getOrCreateClient(config, tempDir.path(), 1_000);
await lspClient.shutdownAll();
expect(await markerExists(path.join(markerDir, "shutdown"))).toBe(true);
expect(await markerExists(path.join(markerDir, "exit"))).toBe(true);
expect(await markerExists(path.join(markerDir, "sigterm"))).toBe(false);
// Graceful handshake: the client sends `shutdown`, waits for its reply,
// then sends the `exit` notification -- and never resorts to the hard
// `proc.kill()` (production's SIGTERM fallback) because the server exits
// cleanly on `exit`.
const methods = server.received.map(message => message.method);
const shutdownIndex = methods.indexOf("shutdown");
const exitIndex = methods.indexOf("exit");
expect(shutdownIndex).toBeGreaterThanOrEqual(0);
expect(exitIndex).toBeGreaterThan(shutdownIndex);
expect(server.killed).toBe(false);
} finally {
await lspClient.shutdownAll();
tempDir.removeSync();
@@ -150,60 +241,26 @@ process.abort();
it("advertises workspace folder support during LSP initialization", async () => {
const tempDir = TempDir.createSync("@omp-lsp-workspace-folders-");
try {
const initPath = path.join(tempDir.path(), "initialize.json");
const serverPath = path.join(tempDir.path(), "server.ts");
await Bun.write(
serverPath,
`
const initPath = process.argv[2];
const decoder = new TextDecoder();
let buffer = "";
const server = installFakeLsp((message, srv) => {
if (message.method === "initialize") {
srv.send({ jsonrpc: "2.0", id: message.id, result: { capabilities: {} } });
} else if (message.method === "shutdown") {
srv.send({ jsonrpc: "2.0", id: message.id, result: null });
} else if (message.method === "exit") {
srv.exit(0);
}
});
function send(message) {
const content = JSON.stringify(message);
process.stdout.write(\`Content-Length: \${Buffer.byteLength(content, "utf8")}\\r\\n\\r\\n\${content}\`);
}
for await (const chunk of Bun.stdin.stream()) {
buffer += decoder.decode(chunk, { stream: true });
while (true) {
const headerEnd = buffer.indexOf("\\r\\n\\r\\n");
if (headerEnd === -1) break;
const header = buffer.slice(0, headerEnd);
const match = /Content-Length: (\\d+)/i.exec(header);
if (!match) process.exit(2);
const contentLength = Number(match[1]);
const contentStart = headerEnd + 4;
const contentEnd = contentStart + contentLength;
if (buffer.length < contentEnd) break;
const message = JSON.parse(buffer.slice(contentStart, contentEnd));
buffer = buffer.slice(contentEnd);
if (message.method === "initialize") {
await Bun.write(initPath, JSON.stringify(message.params));
send({ jsonrpc: "2.0", id: message.id, result: { capabilities: {} } });
} else if (message.method === "shutdown") {
send({ jsonrpc: "2.0", id: message.id, result: null });
} else if (message.method === "exit") {
process.exit(0);
}
}
}
`,
);
const server: ServerConfig = {
command: process.execPath,
args: [serverPath, initPath],
const config: ServerConfig = {
command: "fake-lsp",
fileTypes: ["rs"],
rootMarkers: [],
};
await lspClient.getOrCreateClient(server, tempDir.path(), 1_000);
const params = (await Bun.file(initPath).json()) as {
await lspClient.getOrCreateClient(config, tempDir.path(), 1_000);
const init = server.received.find(message => message.method === "initialize");
const params = init?.params as {
capabilities?: { workspace?: { workspaceFolders?: unknown } };
workspaceFolders?: unknown;
};
@@ -221,69 +278,26 @@ for await (const chunk of Bun.stdin.stream()) {
it("answers workspace/workspaceFolders requests with the current folder set", async () => {
const tempDir = TempDir.createSync("@omp-lsp-workspace-folders-request-");
try {
const responsePath = path.join(tempDir.path(), "folders-response.json");
const serverPath = path.join(tempDir.path(), "server.ts");
await Bun.write(
serverPath,
`
const responsePath = process.argv[2];
const decoder = new TextDecoder();
let buffer = "";
const server = installFakeLsp((message, srv) => {
if (message.method === "initialize") {
srv.send({ jsonrpc: "2.0", id: message.id, result: { capabilities: {} } });
// Server-initiated request: the client must answer with the folder set.
srv.send({ jsonrpc: "2.0", id: 9001, method: "workspace/workspaceFolders" });
} else if (message.method === "shutdown") {
srv.send({ jsonrpc: "2.0", id: message.id, result: null });
} else if (message.method === "exit") {
srv.exit(0);
}
});
function send(message) {
const content = JSON.stringify(message);
process.stdout.write(\`Content-Length: \${Buffer.byteLength(content, "utf8")}\\r\\n\\r\\n\${content}\`);
}
for await (const chunk of Bun.stdin.stream()) {
buffer += decoder.decode(chunk, { stream: true });
while (true) {
const headerEnd = buffer.indexOf("\\r\\n\\r\\n");
if (headerEnd === -1) break;
const header = buffer.slice(0, headerEnd);
const match = /Content-Length: (\\d+)/i.exec(header);
if (!match) process.exit(2);
const contentLength = Number(match[1]);
const contentStart = headerEnd + 4;
const contentEnd = contentStart + contentLength;
if (buffer.length < contentEnd) break;
const message = JSON.parse(buffer.slice(contentStart, contentEnd));
buffer = buffer.slice(contentEnd);
if (message.method === "initialize") {
send({ jsonrpc: "2.0", id: message.id, result: { capabilities: {} } });
send({ jsonrpc: "2.0", id: 9001, method: "workspace/workspaceFolders" });
} else if (message.id === 9001) {
await Bun.write(responsePath, JSON.stringify(message));
} else if (message.method === "shutdown") {
send({ jsonrpc: "2.0", id: message.id, result: null });
} else if (message.method === "exit") {
process.exit(0);
}
}
}
`,
);
const server: ServerConfig = {
command: process.execPath,
args: [serverPath, responsePath],
const config: ServerConfig = {
command: "fake-lsp",
fileTypes: ["rs"],
rootMarkers: [],
};
await lspClient.getOrCreateClient(server, tempDir.path(), 1_000);
const deadline = Date.now() + 1_000;
while (!fs.existsSync(responsePath) && Date.now() < deadline) {
await Bun.sleep(20);
}
const response = (await Bun.file(responsePath).json()) as {
error?: { code: number };
result?: unknown;
};
await lspClient.getOrCreateClient(config, tempDir.path(), 1_000);
const response = await server.waitFor(message => message.id === 9001 && message.method === undefined);
expect(response.error).toBeUndefined();
expect(response.result).toEqual([{ uri: fileToUri(tempDir.path()), name: path.basename(tempDir.path()) }]);
@@ -297,88 +311,67 @@ for await (const chunk of Bun.stdin.stream()) {
const tempDir = TempDir.createSync("@omp-lsp-rust-workspace-");
try {
const sourcePath = path.join(tempDir.path(), "src", "main.rs");
const serverPath = path.join(tempDir.path(), "server.ts");
const eventLogPath = path.join(tempDir.path(), "events.log");
const statusCountPath = path.join(tempDir.path(), "status-count.txt");
await Bun.write(path.join(tempDir.path(), "Cargo.toml"), '[package]\nname = "fixture"\nversion = "0.0.0"\n');
await Bun.write(sourcePath, "fn greet() {}\nfn main() { greet(); }\n");
await Bun.write(
serverPath,
`
const eventLogPath = process.argv[2];
const statusCountPath = process.argv[3];
const definitionUri = process.argv[4];
const decoder = new TextDecoder();
let buffer = "";
let statusRequests = 0;
let eventLog = "";
function send(message) {
const content = JSON.stringify(message);
process.stdout.write(\`Content-Length: \${Buffer.byteLength(content, "utf8")}\\r\\n\\r\\n\${content}\`);
}
for await (const chunk of Bun.stdin.stream()) {
buffer += decoder.decode(chunk, { stream: true });
while (true) {
const headerEnd = buffer.indexOf("\\r\\n\\r\\n");
if (headerEnd === -1) break;
const header = buffer.slice(0, headerEnd);
const match = /Content-Length: (\\d+)/i.exec(header);
if (!match) process.exit(2);
const contentLength = Number(match[1]);
const contentStart = headerEnd + 4;
const contentEnd = contentStart + contentLength;
if (buffer.length < contentEnd) break;
const message = JSON.parse(buffer.slice(contentStart, contentEnd));
buffer = buffer.slice(contentEnd);
if (message.method === "initialize") {
send({ jsonrpc: "2.0", id: message.id, result: { capabilities: { definitionProvider: true } } });
send({ jsonrpc: "2.0", method: "$/progress", params: { token: "workspace", value: { kind: "begin" } } });
send({ jsonrpc: "2.0", method: "$/progress", params: { token: "workspace", value: { kind: "end" } } });
} else if (message.method === "rust-analyzer/analyzerStatus") {
statusRequests++;
eventLog += "status\\n";
await Bun.write(eventLogPath, eventLog);
await Bun.write(statusCountPath, String(statusRequests));
if (statusRequests === 1) {
continue;
}
const result = statusRequests < 3 ? "No workspaces" : "Workspaces:\\nLoaded 1 package across 1 workspace.";
send({ jsonrpc: "2.0", id: message.id, result });
} else if (message.method === "textDocument/didOpen") {
eventLog += "open\\n";
await Bun.write(eventLogPath, eventLog);
} else if (message.method === "textDocument/definition") {
send({
jsonrpc: "2.0",
id: message.id,
result: [{ uri: definitionUri, range: { start: { line: 0, character: 3 }, end: { line: 0, character: 8 } } }],
const events: string[] = [];
let statusRequests = 0;
installFakeLsp((message, srv) => {
if (message.method === "initialize") {
srv.send({ jsonrpc: "2.0", id: message.id, result: { capabilities: { definitionProvider: true } } });
srv.send({
jsonrpc: "2.0",
method: "$/progress",
params: { token: "workspace", value: { kind: "begin" } },
});
srv.send({
jsonrpc: "2.0",
method: "$/progress",
params: { token: "workspace", value: { kind: "end" } },
});
} else if (message.method === "rust-analyzer/analyzerStatus") {
statusRequests++;
events.push("status");
// The first status request is intentionally dropped so the client
// must treat the request timeout as a retry signal
// (deadline-as-signal), then keep polling through "No workspaces"
// until the workspace reports ready.
if (statusRequests === 1) return;
const ready = statusRequests >= 3;
srv.send({
jsonrpc: "2.0",
id: message.id,
result: ready ? "Workspaces:\nLoaded 1 package across 1 workspace." : "No workspaces",
});
} else if (message.method === "textDocument/didOpen") {
events.push("open");
} else if (message.method === "textDocument/definition") {
srv.send({
jsonrpc: "2.0",
id: message.id,
result: [
{
uri: fileToUri(sourcePath),
range: { start: { line: 0, character: 3 }, end: { line: 0, character: 8 } },
},
],
});
} else if (message.method === "shutdown") {
srv.send({ jsonrpc: "2.0", id: message.id, result: null });
} else if (message.method === "exit") {
srv.exit(0);
}
});
} else if (message.method === "shutdown") {
send({ jsonrpc: "2.0", id: message.id, result: null });
} else if (message.method === "exit") {
process.exit(0);
}
}
}
`,
);
const server: ServerConfig = {
command: "rust-analyzer",
resolvedCommand: process.execPath,
args: [serverPath, eventLogPath, statusCountPath, fileToUri(sourcePath)],
fileTypes: ["rs"],
rootMarkers: [],
// Shrink the workspace-ready polling window so the test exercises the
// timeout→retry→ready sequence without waiting out the 2s production settle.
// The status-request timeout stays generous to avoid racing the subprocess.
workspaceReadyTimings: { timeoutMs: 5_000, pollMs: 10, settleMs: 20, statusRequestTimeoutMs: 150 },
// Drive the timeout -> retry -> ready loop without real-clock latency.
// The first status request still times out (proving deadline-as-signal),
// just on a tiny budget instead of the 2s production settle window.
workspaceReadyTimings: { timeoutMs: 5_000, pollMs: 1, settleMs: 2, statusRequestTimeoutMs: 20 },
};
vi.spyOn(lspConfig, "loadConfig").mockReturnValue({
@@ -400,11 +393,9 @@ for await (const chunk of Bun.stdin.stream()) {
.map(block => block.text)
.join("\n");
const eventLog = (await Bun.file(eventLogPath).text()).trim().split("\n");
expect(output).toContain("Found 1 definition(s)");
expect(eventLog[0]).toBe("open");
expect(eventLog.filter(line => line === "status").length).toBeGreaterThanOrEqual(3);
expect(Number(await Bun.file(statusCountPath).text())).toBeGreaterThanOrEqual(3);
expect(events[0]).toBe("open");
expect(events.filter(line => line === "status").length).toBeGreaterThanOrEqual(3);
} finally {
vi.restoreAllMocks();
await lspClient.shutdownAll();
@@ -416,70 +407,38 @@ for await (const chunk of Bun.stdin.stream()) {
const tempDir = TempDir.createSync("@omp-lsp-rust-standalone-");
try {
const sourcePath = path.join(tempDir.path(), "foo.rs");
const serverPath = path.join(tempDir.path(), "server.ts");
const eventLogPath = path.join(tempDir.path(), "events.log");
await Bun.write(sourcePath, 'fn greet() -> &\'static str { "hi" }\n');
await Bun.write(
serverPath,
`
const eventLogPath = process.argv[2];
const definitionUri = process.argv[3];
const decoder = new TextDecoder();
let buffer = "";
let eventLog = "";
function send(message) {
const content = JSON.stringify(message);
process.stdout.write(\`Content-Length: \${Buffer.byteLength(content, "utf8")}\\r\\n\\r\\n\${content}\`);
}
for await (const chunk of Bun.stdin.stream()) {
buffer += decoder.decode(chunk, { stream: true });
while (true) {
const headerEnd = buffer.indexOf("\\r\\n\\r\\n");
if (headerEnd === -1) break;
const header = buffer.slice(0, headerEnd);
const match = /Content-Length: (\\d+)/i.exec(header);
if (!match) process.exit(2);
const contentLength = Number(match[1]);
const contentStart = headerEnd + 4;
const contentEnd = contentStart + contentLength;
if (buffer.length < contentEnd) break;
const message = JSON.parse(buffer.slice(contentStart, contentEnd));
buffer = buffer.slice(contentEnd);
if (message.method === "initialize") {
send({ jsonrpc: "2.0", id: message.id, result: { capabilities: { definitionProvider: true } } });
} else if (message.method === "rust-analyzer/analyzerStatus") {
eventLog += "status\\n";
await Bun.write(eventLogPath, eventLog);
send({ jsonrpc: "2.0", id: message.id, result: "No workspaces" });
} else if (message.method === "textDocument/didOpen") {
eventLog += "open\\n";
await Bun.write(eventLogPath, eventLog);
} else if (message.method === "textDocument/definition") {
send({
jsonrpc: "2.0",
id: message.id,
result: [{ uri: definitionUri, range: { start: { line: 0, character: 3 }, end: { line: 0, character: 8 } } }],
const events: string[] = [];
installFakeLsp((message, srv) => {
if (message.method === "initialize") {
srv.send({ jsonrpc: "2.0", id: message.id, result: { capabilities: { definitionProvider: true } } });
} else if (message.method === "rust-analyzer/analyzerStatus") {
events.push("status");
srv.send({ jsonrpc: "2.0", id: message.id, result: "No workspaces" });
} else if (message.method === "textDocument/didOpen") {
events.push("open");
} else if (message.method === "textDocument/definition") {
srv.send({
jsonrpc: "2.0",
id: message.id,
result: [
{
uri: fileToUri(sourcePath),
range: { start: { line: 0, character: 3 }, end: { line: 0, character: 8 } },
},
],
});
} else if (message.method === "shutdown") {
srv.send({ jsonrpc: "2.0", id: message.id, result: null });
} else if (message.method === "exit") {
srv.exit(0);
}
});
} else if (message.method === "shutdown") {
send({ jsonrpc: "2.0", id: message.id, result: null });
} else if (message.method === "exit") {
process.exit(0);
}
}
}
`,
);
const server: ServerConfig = {
command: "rust-analyzer",
resolvedCommand: process.execPath,
args: [serverPath, eventLogPath, fileToUri(sourcePath)],
fileTypes: ["rs"],
rootMarkers: [],
};
@@ -491,7 +450,6 @@ for await (const chunk of Bun.stdin.stream()) {
vi.spyOn(lspConfig, "getServersForFile").mockReturnValue([["rust-analyzer", server]]);
const tool = new LspTool({ cwd: tempDir.path() } as ToolSession);
const started = Date.now();
const result = await tool.execute("rust-standalone-test", {
action: "definition",
file: sourcePath,
@@ -499,17 +457,19 @@ for await (const chunk of Bun.stdin.stream()) {
symbol: "greet",
timeout: 10,
});
const elapsed = Date.now() - started;
const output = result.content
.filter(block => block.type === "text")
.map(block => block.text)
.join("\n");
const eventLog = (await Bun.file(eventLogPath).text()).trim().split("\n");
// Standalone .rs (no Cargo workspace ancestor): the file is opened but the
// analyzerStatus readiness poll is skipped entirely. The direct
// `not.toContain("status")` assertion is the skip contract; the original
// `elapsed < 2000ms` wall-clock proxy is dropped -- it is vacuous once the
// transport is in-memory.
expect(output).toContain("Found 1 definition(s)");
expect(eventLog).toContain("open");
expect(eventLog).not.toContain("status");
expect(elapsed).toBeLessThan(2_000);
expect(events).toContain("open");
expect(events).not.toContain("status");
} finally {
vi.restoreAllMocks();
await lspClient.shutdownAll();
@@ -1,4 +1,4 @@
import { afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test";
import { afterAll, beforeAll, beforeEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
@@ -126,16 +126,17 @@ describe("tool path arrays", () => {
beforeAll(async () => {
await initTheme(false, undefined, undefined, "dark", "light");
});
beforeEach(async () => {
treeEntryCounter = 0;
resetSettingsForTest();
await Settings.init({ inMemory: true });
tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "search-path-lists-"));
await createSearchFixture(tempDir);
});
afterEach(async () => {
beforeEach(() => {
treeEntryCounter = 0;
});
afterAll(async () => {
await fs.rm(tempDir, { recursive: true, force: true });
resetSettingsForTest();
});
@@ -259,10 +260,11 @@ describe("tool path arrays", () => {
// Create `apps/[id]/page.tsx` — `[id]` is glob char-class syntax but here it
// is a literal directory name. The literal path must take precedence over
// the glob interpretation, otherwise the lookup returns no matches.
await fs.mkdir(path.join(tempDir, "apps", "[id]"), { recursive: true });
await Bun.write(path.join(tempDir, "apps", "[id]", "page.tsx"), "bracket-needle\n");
const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "search-path-lists-"));
await fs.mkdir(path.join(tmp, "apps", "[id]"), { recursive: true });
await Bun.write(path.join(tmp, "apps", "[id]", "page.tsx"), "bracket-needle\n");
const tools = await createTools(createTestSession(tempDir));
const tools = await createTools(createTestSession(tmp));
const tool = tools.find(entry => entry.name === "search");
if (!tool) throw new Error("Missing search tool");
@@ -277,6 +279,7 @@ describe("tool path arrays", () => {
paths: ["apps/[id]"],
});
expect(getText(dir)).toContain("bracket-needle");
await fs.rm(tmp, { recursive: true, force: true });
});
it("search pending renderer accepts a single string path", () => {
@@ -290,7 +293,8 @@ describe("tool path arrays", () => {
expect((component as Text).getText()).toContain("in folder with spaces/");
});
it("agent hub chat renders a single-string search path summary", async () => {
const sessionFile = await makeJsonlSessionFile(tempDir, [
const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "search-path-lists-"));
const sessionFile = await makeJsonlSessionFile(tmp, [
{ type: "session", version: 3, id: "search-overlay-session", timestamp: new Date().toISOString() },
{
type: "message",
@@ -365,6 +369,7 @@ describe("tool path arrays", () => {
// single-string `paths` arg shows up as the "in <paths>" scope meta on the
// pending call line (a completed result merges the call line away).
expect(rendered).toContain("in folder with spaces/");
await fs.rm(tmp, { recursive: true, force: true });
});
it("tree selector renders a single-string search path summary", () => {
@@ -475,12 +480,13 @@ describe("tool path arrays", () => {
});
it("write reports absolute in-cwd targets relative to cwd", async () => {
const tools = await createTools(createTestSession(tempDir));
const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "search-path-lists-"));
const tools = await createTools(createTestSession(tmp));
const tool = tools.find(entry => entry.name === "write");
expect(tool).toBeDefined();
if (!tool) throw new Error("Missing write tool");
const absoluteTarget = path.join(tempDir, "written.txt");
const absoluteTarget = path.join(tmp, "written.txt");
const result = await tool.execute("write-absolute-in-cwd", {
path: absoluteTarget,
content: "written\n",
@@ -488,8 +494,9 @@ describe("tool path arrays", () => {
const text = getText(result);
expect(text).toContain("Successfully wrote 8 bytes to written.txt");
expect(text).not.toContain(tempDir);
expect(text).not.toContain(tmp);
expect(await Bun.file(absoluteTarget).text()).toBe("written\n");
await fs.rm(tmp, { recursive: true, force: true });
});
it("read expands comma-delimited paths", async () => {
@@ -599,9 +606,11 @@ describe("tool path arrays", () => {
});
it("ast_edit applies across an explicit path array", async () => {
const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "search-path-lists-"));
await createSearchFixture(tmp);
const queue = new ToolChoiceQueue();
const tools = await createTools(
createTestSession(tempDir, {
createTestSession(tmp, {
getToolChoiceQueue: () => queue,
buildToolChoice: () => ({ type: "tool" as const, name: "resolve" }),
steer: () => {},
@@ -630,16 +639,13 @@ describe("tool path arrays", () => {
if (!invoker) throw new Error("Expected pending resolve invoker");
await invoker({ action: "apply", reason: "apply multi-path ast edit" });
expect(await Bun.file(path.join(tempDir, "apps", "ast.ts")).text()).toContain("modernWrap(appsValue, appsArg)");
expect(await Bun.file(path.join(tempDir, "packages", "ast.ts")).text()).toContain(
expect(await Bun.file(path.join(tmp, "apps", "ast.ts")).text()).toContain("modernWrap(appsValue, appsArg)");
expect(await Bun.file(path.join(tmp, "packages", "ast.ts")).text()).toContain(
"modernWrap(packagesValue, packagesArg)",
);
expect(await Bun.file(path.join(tempDir, "phases", "ast.ts")).text()).toContain(
"modernWrap(phasesValue, phasesArg)",
);
expect(await Bun.file(path.join(tempDir, "other", "ast.ts")).text()).toContain(
"legacyWrap(otherValue, otherArg)",
);
expect(await Bun.file(path.join(tmp, "phases", "ast.ts")).text()).toContain("modernWrap(phasesValue, phasesArg)");
expect(await Bun.file(path.join(tmp, "other", "ast.ts")).text()).toContain("legacyWrap(otherValue, otherArg)");
await fs.rm(tmp, { recursive: true, force: true });
});
it("find accepts explicit path arrays", async () => {
@@ -804,13 +810,14 @@ describe("tool path arrays", () => {
});
it("grep keeps explicit files exact", async () => {
await fs.mkdir(path.join(tempDir, "nested"), { recursive: true });
await Bun.write(path.join(tempDir, "alpha.txt"), "exact-needle alpha\n");
await Bun.write(path.join(tempDir, "beta.txt"), "exact-needle beta\n");
await Bun.write(path.join(tempDir, "nested", "alpha.txt"), "exact-needle nested alpha\n");
await Bun.write(path.join(tempDir, "nested", "beta.txt"), "exact-needle nested beta\n");
const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "search-path-lists-"));
await fs.mkdir(path.join(tmp, "nested"), { recursive: true });
await Bun.write(path.join(tmp, "alpha.txt"), "exact-needle alpha\n");
await Bun.write(path.join(tmp, "beta.txt"), "exact-needle beta\n");
await Bun.write(path.join(tmp, "nested", "alpha.txt"), "exact-needle nested alpha\n");
await Bun.write(path.join(tmp, "nested", "beta.txt"), "exact-needle nested beta\n");
const tools = await createTools(createTestSession(tempDir));
const tools = await createTools(createTestSession(tmp));
const tool = tools.find(entry => entry.name === "search");
expect(tool).toBeDefined();
if (!tool) throw new Error("Missing search tool");
@@ -829,6 +836,7 @@ describe("tool path arrays", () => {
expect(text).not.toContain("nested");
expect(details?.fileCount).toBe(2);
expect(details?.scopePath).toBe("alpha.txt, beta.txt");
await fs.rm(tmp, { recursive: true, force: true });
});
it("grep renders only file headings that have child lines", async () => {
@@ -856,10 +864,11 @@ describe("tool path arrays", () => {
});
it("grep explains match and context gutters with new format", async () => {
await Bun.write(path.join(tempDir, "context.txt"), "#if FLAG\nneedle\n#endif\n");
const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "search-path-lists-"));
await Bun.write(path.join(tmp, "context.txt"), "#if FLAG\nneedle\n#endif\n");
const tools = await createTools(
createTestSession(tempDir, {
createTestSession(tmp, {
settings: Settings.isolated({ "search.contextBefore": 1, "search.contextAfter": 1 }),
}),
);
@@ -876,5 +885,6 @@ describe("tool path arrays", () => {
expect(text).toMatch(/ 1:#if FLAG/);
expect(text).toMatch(/\*2:needle/);
expect(text).toMatch(/ 3:#endif/);
await fs.rm(tmp, { recursive: true, force: true });
});
});
+99 -123
View File
@@ -1,5 +1,5 @@
import { Database } from "bun:sqlite";
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import { afterAll, beforeAll, describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
@@ -40,8 +40,14 @@ function createSession(cwd: string, overrides: Partial<SessionLike> = {}): Sessi
} as SessionLike;
}
function createFixtureDatabase(dbPath: string): void {
const db = new Database(dbPath);
/**
* Builds the fixture database once in memory and serializes it to bytes. Tests
* stamp these bytes onto disk with a single `writeFile` (one fsync) instead of
* re-running the table creation and ~13 autocommit inserts (≈14 fsyncs) per
* test — the original per-test on-disk build dominated the suite's wall time.
*/
function buildFixtureBytes(): Uint8Array {
const db = new Database(":memory:");
try {
db.run(`
CREATE TABLE users (
@@ -70,52 +76,30 @@ function createFixtureDatabase(dbPath: string): void {
);
`);
db.prepare("INSERT INTO users (name, email, status, created) VALUES (?, ?, ?, ?)").run(
"Alice",
"alice@example.com",
"active",
1,
);
db.prepare("INSERT INTO users (name, email, status, created) VALUES (?, ?, ?, ?)").run(
"Bob",
"bob@example.com",
"inactive",
2,
);
db.prepare("INSERT INTO users (name, email, status, created) VALUES (?, ?, ?, ?)").run(
"Carol",
"carol@example.com",
"active",
3,
);
db.prepare("INSERT INTO users (name, email, status, created) VALUES (?, ?, ?, ?)").run(
"Dave",
"dave@example.com",
"inactive",
4,
);
db.prepare("INSERT INTO users (name, email, status, created) VALUES (?, ?, ?, ?)").run(
"Eve",
"eve@example.com",
"active",
5,
);
db.prepare("INSERT INTO users (name, email, status, created) VALUES (?, ?, ?, ?)").run(
"Frank",
"frank@example.com",
"active",
6,
);
const insertUser = db.prepare("INSERT INTO users (name, email, status, created) VALUES (?, ?, ?, ?)");
const insertSlug = db.prepare("INSERT INTO slugs (slug, title) VALUES (?, ?)");
const insertNote = db.prepare("INSERT INTO notes (body) VALUES (?)");
const seed = db.transaction(() => {
insertUser.run("Alice", "alice@example.com", "active", 1);
insertUser.run("Bob", "bob@example.com", "inactive", 2);
insertUser.run("Carol", "carol@example.com", "active", 3);
insertUser.run("Dave", "dave@example.com", "inactive", 4);
insertUser.run("Eve", "eve@example.com", "active", 5);
insertUser.run("Frank", "frank@example.com", "active", 6);
db.prepare("INSERT INTO slugs (slug, title) VALUES (?, ?)").run("welcome", "Welcome");
db.prepare("INSERT INTO slugs (slug, title) VALUES (?, ?)").run("about", "About");
insertSlug.run("welcome", "Welcome");
insertSlug.run("about", "About");
db.prepare("INSERT INTO notes (body) VALUES (?)").run("First note");
db.prepare("INSERT INTO notes (body) VALUES (?)").run("Second note");
db.prepare("INSERT INTO notes (body) VALUES (?)").run("Third; note");
insertNote.run("First note");
insertNote.run("Second note");
insertNote.run("Third; note");
db.prepare("INSERT INTO composite (team_id, user_id, value) VALUES (?, ?, ?)").run(1, 2, "pair");
db.prepare("INSERT INTO wide_rows (id, payload) VALUES (?, ?)").run(1, "x".repeat(320));
db.prepare("INSERT INTO composite (team_id, user_id, value) VALUES (?, ?, ?)").run(1, 2, "pair");
db.prepare("INSERT INTO wide_rows (id, payload) VALUES (?, ?)").run(1, "x".repeat(320));
});
seed();
return db.serialize();
} finally {
db.close();
}
@@ -156,11 +140,21 @@ describe("SQLite tool support", () => {
let sqlitePath: string;
let sqliteDbPath: string;
let invalidDbPath: string;
let fixtureBytes: Uint8Array;
let readTool: ReadTool;
let writeTool: WriteTool;
let originalEditVariant: string | undefined;
beforeEach(async () => {
// The shared fixture is only ever read by most tests; the few tests that
// mutate a database stamp their own fresh copy via `stampFreshDb`, so the
// shared file stays pristine and can be created exactly once.
async function stampFreshDb(name: string): Promise<string> {
const dbPath = path.join(tmpDir, name);
await fs.writeFile(dbPath, fixtureBytes);
return dbPath;
}
beforeAll(async () => {
tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "sqlite-tool-test-"));
sqlitePath = path.join(tmpDir, "app.sqlite");
sqliteDbPath = path.join(tmpDir, "app.db");
@@ -168,16 +162,17 @@ describe("SQLite tool support", () => {
originalEditVariant = Bun.env.PI_EDIT_VARIANT;
Bun.env.PI_EDIT_VARIANT = "replace";
createFixtureDatabase(sqlitePath);
await fs.copyFile(sqlitePath, sqliteDbPath);
await Bun.write(invalidDbPath, "not sqlite\nstill text\n");
fixtureBytes = buildFixtureBytes();
await fs.writeFile(sqlitePath, fixtureBytes);
await fs.writeFile(sqliteDbPath, fixtureBytes);
await fs.writeFile(invalidDbPath, "not sqlite\nstill text\n");
const session = createSession(tmpDir);
readTool = new ReadTool(session);
writeTool = new WriteTool(session);
});
afterEach(async () => {
afterAll(async () => {
if (originalEditVariant === undefined) {
delete Bun.env.PI_EDIT_VARIANT;
} else {
@@ -330,7 +325,10 @@ describe("SQLite tool support", () => {
});
it("caps raw ?q= queries at the row limit and surfaces a LIMIT hint", async () => {
const db = new Database(sqlitePath);
// Dedicated database: this is the only test that needs >1000 rows, and it
// must not pollute the shared fixture. A single transaction = one commit.
const capDbPath = path.join(tmpDir, "rawcap.sqlite");
const db = new Database(capDbPath);
try {
db.run("CREATE TABLE big (id INTEGER PRIMARY KEY, value TEXT NOT NULL)");
const insert = db.prepare("INSERT INTO big (value) VALUES (?)");
@@ -344,7 +342,7 @@ describe("SQLite tool support", () => {
db.close();
}
const result = await readTool.execute("sqlite-raw-row-cap", { path: `${sqlitePath}?q=SELECT * FROM big` });
const result = await readTool.execute("sqlite-raw-row-cap", { path: `${capDbPath}?q=SELECT * FROM big` });
const text = getText(result);
expect(text).toContain("val_1000_end");
@@ -373,34 +371,37 @@ describe("SQLite tool support", () => {
});
it("inserts rows through the write tool with JSON5 content", async () => {
const dbPath = await stampFreshDb("write-insert.sqlite");
await writeTool.execute("sqlite-write-insert", {
path: `${sqlitePath}:users`,
path: `${dbPath}:users`,
content: "{ name: 'Grace', email: 'grace@example.com', status: 'active', created: 7 }",
});
expect(readUserByEmail(sqlitePath, "grace@example.com")).toEqual({
expect(readUserByEmail(dbPath, "grace@example.com")).toEqual({
name: "Grace",
email: "grace@example.com",
});
});
it("updates rows through the write tool by primary key", async () => {
const dbPath = await stampFreshDb("write-update.sqlite");
await writeTool.execute("sqlite-write-update", {
path: `${sqlitePath}:users:2`,
path: `${dbPath}:users:2`,
content: "{ email: 'bob+new@example.com' }",
});
expect(readUserEmail(sqlitePath, 2)).toBe("bob+new@example.com");
expect(readUserEmail(dbPath, 2)).toBe("bob+new@example.com");
});
it("deletes rows through the write tool with empty content", async () => {
const dbPath = await stampFreshDb("write-delete.sqlite");
await writeTool.execute("sqlite-write-delete", {
path: `${sqlitePath}:users:2`,
path: `${dbPath}:users:2`,
content: " ",
});
expect(readUserCount(sqlitePath)).toBe(5);
expect(readUserEmail(sqlitePath, 2)).toBeNull();
expect(readUserCount(dbPath)).toBe(5);
expect(readUserEmail(dbPath, 2)).toBeNull();
});
it("enforces plan mode for SQLite writes", async () => {
@@ -449,79 +450,54 @@ describe("SQLite tool support", () => {
});
describe("SQLite table listing row counts", () => {
let tmpDir: string;
let dbPath: string;
// These tests exercise `listTables`/`renderTableList` directly against a
// `Database` handle, so an in-memory database preserves the row-count
// contract with zero disk I/O. `base` is never analyzed (exact / lower-bound
// behavior); `analyzed` carries planner estimates.
let base: Database;
let analyzed: Database;
beforeEach(async () => {
tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "sqlite-count-test-"));
dbPath = path.join(tmpDir, "counts.db");
});
afterEach(async () => {
await fs.rm(tmpDir, { recursive: true, force: true });
});
function seed(rowsPerTable: { big: number; small: number }): void {
const db = new Database(dbPath);
try {
db.run("CREATE TABLE big (id INTEGER PRIMARY KEY, v TEXT NOT NULL)");
db.run("CREATE TABLE small (id INTEGER PRIMARY KEY)");
const bigStmt = db.prepare("INSERT INTO big (v) VALUES (?)");
for (let i = 0; i < rowsPerTable.big; i++) bigStmt.run("x");
const smallStmt = db.prepare("INSERT INTO small DEFAULT VALUES");
for (let i = 0; i < rowsPerTable.small; i++) smallStmt.run();
} finally {
db.close();
}
function buildCountsDb(analyze: boolean): Database {
const db = new Database(":memory:");
db.run("CREATE TABLE big (id INTEGER PRIMARY KEY, v TEXT NOT NULL)");
db.run("CREATE TABLE small (id INTEGER PRIMARY KEY)");
const bigStmt = db.prepare("INSERT INTO big (v) VALUES (?)");
for (let i = 0; i < 10; i++) bigStmt.run("x");
const smallStmt = db.prepare("INSERT INTO small DEFAULT VALUES");
for (let i = 0; i < 2; i++) smallStmt.run();
if (analyze) db.run("ANALYZE");
return db;
}
function analyze(): void {
const db = new Database(dbPath);
try {
db.run("ANALYZE");
} finally {
db.close();
}
}
beforeAll(() => {
base = buildCountsDb(false);
analyzed = buildCountsDb(true);
});
afterAll(() => {
base.close();
analyzed.close();
});
it("counts small tables exactly", () => {
seed({ big: 10, small: 2 });
const db = new Database(dbPath, { readonly: true });
try {
const rendered = renderTableList(listTables(db, { probeCap: 100 }));
expect(rendered).toContain("big (10 rows)");
expect(rendered).toContain("small (2 rows)");
} finally {
db.close();
}
const rendered = renderTableList(listTables(base, { probeCap: 100 }));
expect(rendered).toContain("big (10 rows)");
expect(rendered).toContain("small (2 rows)");
});
it("reports the planner estimate for tables larger than the probe cap", () => {
seed({ big: 10, small: 2 });
analyze();
const db = new Database(dbPath, { readonly: true });
try {
// probeCap=5: big (estimate 10) exceeds it and is reported as an estimate
// without scanning; small (estimate 2) is counted exactly.
const rendered = renderTableList(listTables(db, { probeCap: 5 }));
expect(rendered).toContain("big (~10 rows)");
expect(rendered).toContain("small (2 rows)");
} finally {
db.close();
}
// probeCap=5: big (estimate 10) exceeds it and is reported as an estimate
// without scanning; small (estimate 2) is counted exactly.
const rendered = renderTableList(listTables(analyzed, { probeCap: 5 }));
expect(rendered).toContain("big (~10 rows)");
expect(rendered).toContain("small (2 rows)");
});
it("reports a lower bound when an unanalyzed table exceeds the probe cap", () => {
seed({ big: 10, small: 2 });
const db = new Database(dbPath, { readonly: true });
try {
// No ANALYZE, so no estimate exists; the bounded probe stops at the cap
// and reports a lower bound instead of scanning the whole table.
const rendered = renderTableList(listTables(db, { probeCap: 3 }));
expect(rendered).toContain("big (3+ rows)");
expect(rendered).toContain("small (2 rows)");
} finally {
db.close();
}
// No ANALYZE, so no estimate exists; the bounded probe stops at the cap
// and reports a lower bound instead of scanning the whole table.
const rendered = renderTableList(listTables(base, { probeCap: 3 }));
expect(rendered).toContain("big (3+ rows)");
expect(rendered).toContain("small (2 rows)");
});
});
@@ -1,4 +1,4 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import { afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test";
import { resizeImage } from "@oh-my-pi/pi-coding-agent/utils/image-resize";
// 1x1 red PNG (69 bytes) — used as a Bun.Image seed to synthesize larger fixtures
@@ -21,25 +21,42 @@ async function makeRedWebP(width: number, height: number): Promise<string> {
return Buffer.from(upscaled).toBase64();
}
// Fixtures are synthesized once and shared read-only — `resizeImage` never mutates
// its input, so a single decodable source per shape serves every test. Real image
// encode/decode is the only cost here, so each source is the smallest solid-red
// image that still crosses the threshold under test:
// - oversizedPng: a thin strip whose long edge exceeds the 1568 default cap, so
// re-encodes touch ~1568×98 px instead of 1568×1568. A uniform red keeps format
// selection deterministic (WebP is always smallest; PNG always beats JPEG), so
// the strip exercises the same format/budget logic as a large square.
// - smallPng / smallWebp: 200×200, comfortably inside every default cap (fast path).
let oversizedPng: string;
let smallPng: string;
let smallWebp: string;
beforeAll(async () => {
[oversizedPng, smallPng, smallWebp] = await Promise.all([
makeRedPng(1600, 100),
makeRedPng(200, 200),
makeRedWebP(200, 200),
]);
});
describe("resizeImage defaults", () => {
it("downscales inputs larger than 1568px on the long edge", async () => {
// 2000x1500 — exceeds the default 1568 cap on width
const data = await makeRedPng(2000, 1500);
const result = await resizeImage({ type: "image", data, mimeType: "image/png" });
// 1600px wide — exceeds the default 1568 cap on the long edge.
const result = await resizeImage({ type: "image", data: oversizedPng, mimeType: "image/png" });
expect(result.wasResized).toBe(true);
expect(result.width).toBeLessThanOrEqual(1568);
expect(result.height).toBeLessThanOrEqual(1568);
// Aspect ratio preserved (with rounding tolerance)
expect(Math.abs(result.width / result.height - 2000 / 1500)).toBeLessThan(0.01);
// Aspect ratio of the 1600x100 source preserved (with rounding tolerance).
expect(Math.abs(result.width / result.height - 1600 / 100)).toBeLessThan(0.01);
});
it("preserves inputs already within budget and dimensions (fast path)", async () => {
// 200x200 red square encodes to ~few hundred bytes — well below budget/4
const data = await makeRedPng(200, 200);
const result = await resizeImage({ type: "image", data, mimeType: "image/png" });
// 200x200 red square encodes to ~few hundred bytes — well below budget/4.
const result = await resizeImage({ type: "image", data: smallPng, mimeType: "image/png" });
expect(result.wasResized).toBe(false);
expect(result.width).toBe(200);
@@ -48,11 +65,9 @@ describe("resizeImage defaults", () => {
});
it("respects custom maxWidth/maxHeight overrides (browser-tool case)", async () => {
// 1600x1200 — exceeds the 1024 cap from the browser screenshot override
const data = await makeRedPng(1600, 1200);
// 1600px wide — exceeds the 1024 cap from the browser screenshot override.
const result = await resizeImage(
{ type: "image", data, mimeType: "image/png" },
{ type: "image", data: oversizedPng, mimeType: "image/png" },
{ maxWidth: 1024, maxHeight: 1024, maxBytes: 150 * 1024, jpegQuality: 70 },
);
@@ -63,12 +78,11 @@ describe("resizeImage defaults", () => {
});
it("respects custom maxBytes override even when dimensions already fit", async () => {
// 800x600 within both default and override dimensions, but a tight 4KB
// budget forces re-encoding/dimension reduction.
const data = await makeRedPng(800, 600);
const originalBytes = Buffer.from(data, "base64").length;
// 200x200 sits within every dimension cap, but a byte budget below the
// source size (after the /4 fast-path headroom) forces a re-encode.
const originalBytes = Buffer.from(smallPng, "base64").length;
const result = await resizeImage({ type: "image", data, mimeType: "image/png" }, { maxBytes: 4 * 1024 });
const result = await resizeImage({ type: "image", data: smallPng, mimeType: "image/png" }, { maxBytes: 1024 });
// Either the result fits the budget, or the algorithm exhausted its
// fallbacks and shipped its smallest variant — but in both cases the
@@ -77,24 +91,23 @@ describe("resizeImage defaults", () => {
});
it("uses lossy WebP or JPEG (not PNG) for oversized inputs", async () => {
// 2000x2000 red PNG — exceeds dimension cap, triggers encodeSmallest.
// Oversized red strip exceeds the dimension cap, triggering encodeSmallest.
// Lossy formats (JPEG/WebP) should win over PNG for a solid-color image
// at this dimension because they compress more aggressively.
const data = await makeRedPng(2000, 2000);
const result = await resizeImage({ type: "image", data, mimeType: "image/png" });
// because they compress more aggressively.
const result = await resizeImage({ type: "image", data: oversizedPng, mimeType: "image/png" });
expect(result.wasResized).toBe(true);
// The result should be a lossy format (JPEG or WebP), not PNG,
// because lossy encoding at <=1568px for a solid square is trivially small.
// because lossy encoding for a solid strip is trivially small.
expect(["image/jpeg", "image/webp"]).toContain(result.mimeType);
expect(result.buffer.length).toBeLessThanOrEqual(500 * 1024);
});
it("excludes WebP when excludeWebP option is true", async () => {
const data = await makeRedPng(2000, 2000);
const result = await resizeImage({ type: "image", data, mimeType: "image/png" }, { excludeWebP: true });
const result = await resizeImage(
{ type: "image", data: oversizedPng, mimeType: "image/png" },
{ excludeWebP: true },
);
expect(result.wasResized).toBe(true);
expect(["image/png", "image/jpeg"]).toContain(result.mimeType);
@@ -105,9 +118,10 @@ describe("resizeImage defaults", () => {
// 200x200 WebP — well below 1568px and ~tiny bytes, so it would hit the
// fast path and pass through as image/webp. excludeWebP MUST force a
// re-encode to a non-WebP format.
const data = await makeRedWebP(200, 200);
const result = await resizeImage({ type: "image", data, mimeType: "image/webp" }, { excludeWebP: true });
const result = await resizeImage(
{ type: "image", data: smallWebp, mimeType: "image/webp" },
{ excludeWebP: true },
);
expect(result.mimeType).not.toBe("image/webp");
expect(["image/png", "image/jpeg"]).toContain(result.mimeType);
@@ -127,23 +141,20 @@ describe("resizeImage env wiring", () => {
});
it("treats OMP_NO_WEBP=1 set at call time as exclusion (not baked at module load)", async () => {
const data = await makeRedWebP(200, 200);
Bun.env.OMP_NO_WEBP = "1";
const result = await resizeImage({ type: "image", data, mimeType: "image/webp" });
const result = await resizeImage({ type: "image", data: smallWebp, mimeType: "image/webp" });
expect(result.mimeType).not.toBe("image/webp");
});
it("treats OMP_NO_WEBP='' / '0' as NOT excluded", async () => {
const data = await makeRedWebP(200, 200);
Bun.env.OMP_NO_WEBP = "";
const empty = await resizeImage({ type: "image", data, mimeType: "image/webp" });
const empty = await resizeImage({ type: "image", data: smallWebp, mimeType: "image/webp" });
expect(empty.mimeType).toBe("image/webp");
Bun.env.OMP_NO_WEBP = "0";
const zero = await resizeImage({ type: "image", data, mimeType: "image/webp" });
const zero = await resizeImage({ type: "image", data: smallWebp, mimeType: "image/webp" });
expect(zero.mimeType).toBe("image/webp");
});
});