ci(test): budgeted bun test parallelism across the chunk pool

The runner stacked two independent parallelism knobs: OMP_TEST_CONCURRENCY
spawned N `bun test` processes while each process ran its own --parallel=M
test files, so the workspace bucket put 4 x 8 = 32 files in flight on a
4-core runner. bun's per-test timeout is wall-clock, so CPU-starved suites
crossed the 5s default and failed at random - mnemopi's sqlite/CLI files
tripped a different pair every run.

- TestCommand now carries a `parallel` request instead of baking the flag
  into argv; the dispatcher resolves it against one shared budget
  (availableParallelism x 2, split by the live pool width) so total
  in-flight files track the machine. A chunk that runs alone still gets
  its full requested width, leaving the sequential CI path unchanged.
- Raised the per-test timeout to 30s (OMP_TEST_TIMEOUT to override).
  Suites here build real SQLite schemas and spawn CLIs, already running
  1-4s per case on a quiet runner; the 10-minute chunk watchdog stays the
  backstop for an actual hang.
- --dry-run now resolves the same budget, so it prints the argv a real run
  would use, and the pool header reports cores plus the granted width.
- Dropped the dead `{ smol: true }` argument: workspaceTestCommand never
  accepted it, and this file's own findings say a smaller heap makes bun
  1.3.14's GC crash more often, so it should not be wired up.
This commit is contained in:
can1357
2026-08-08 07:06:07 +02:00
parent 7cebe901b7
commit 08819b279c
+72 -8
View File
@@ -21,7 +21,10 @@ type CodingAgentBucket = "singleton" | "ui" | "runtime" | "native";
interface TestCommand {
label: string;
cwd: string;
/** argv without `--parallel`; the runner appends it from `parallel` and the pool's CPU budget. */
command: string[];
/** `bun test --parallel` width this chunk wants when it has the machine to itself. */
parallel?: number;
}
type CodingAgentTestPartition = Record<CodingAgentBucket, string[]>;
@@ -204,7 +207,8 @@ function workspaceTestCommand(pkg: string, parallel: number, options: { extraArg
return {
label: pkg,
cwd: pkg,
command: ["bun", "test", `--parallel=${parallel}`, ...extraArgs],
command: ["bun", "test", ...extraArgs],
parallel,
};
}
@@ -313,7 +317,8 @@ async function codingAgentTestCommands(bucket: CodingAgentBucket): Promise<TestC
commands.push({
label: `packages/coding-agent (${plan.label}; ${testFiles.length} files; parallel=${plan.parallel}${chunkLabel}; ${chunk.length} files)`,
cwd: "packages/coding-agent",
command: ["bun", "test", `--parallel=${plan.parallel}`, ...onlyFailuresArgs, ...chunk],
command: ["bun", "test", ...onlyFailuresArgs, ...chunk],
parallel: plan.parallel,
});
}
return commands;
@@ -324,7 +329,7 @@ async function commandsForMode(mode: Mode): Promise<TestCommand[]> {
case "workspace":
return fastWorkspacePackages.map(pkg => workspaceTestCommand(pkg, 8));
case "native":
return nativeAndIntegrationPackages.map(pkg => workspaceTestCommand(pkg, 4, { smol: true }));
return nativeAndIntegrationPackages.map(pkg => workspaceTestCommand(pkg, 4));
case "coding-agent-singleton":
return await codingAgentTestCommands("singleton");
case "coding-agent-ui":
@@ -528,6 +533,58 @@ function testConcurrency(total: number): number {
throw new Error(`Invalid OMP_TEST_CONCURRENCY=${JSON.stringify(raw)}; expected a positive integer, all, or max`);
}
// Test files interleave real IO — sqlite writes, temp dirs, spawned CLIs — with
// CPU, so keeping every core busy needs more in-flight files than cores. This is
// the factor by which the shared budget exceeds `availableParallelism()`.
const FILE_OVERSUBSCRIBE = 2;
// Two independent parallelism knobs stack multiplicatively: the chunk pool runs
// `poolWidth` `bun test` processes at once, and each of those runs its own
// `--parallel=N` test files concurrently, so up to `poolWidth * N` files are in
// flight. Left unbudgeted that oversubscribes the runner by design — the
// workspace bucket asked for 4 x 8 = 32 files on a 4-core box — and because
// bun's per-test timeout is wall-clock, CPU-starved suites blow it and fail at
// random (mnemopi's sqlite/CLI files did, a different set each run). Spend one
// budget instead: each live chunk gets an equal share, never below 1 and never
// above the width it asked for. A chunk that runs alone still gets everything,
// so the sequential CI path is unchanged.
function budgetedParallel(requested: number, poolWidth: number): number {
const budget = Math.max(1, os.availableParallelism()) * FILE_OVERSUBSCRIBE;
return Math.max(1, Math.min(requested, Math.floor(budget / poolWidth)));
}
// Bun's 5s default per-test timeout is a unit-test default, and this repo's
// suites are not unit tests: mnemopi builds real SQLite schemas per case, the
// coding-agent suites drive sessions and subprocesses. Those cases already run
// 1-4s on a quiet CI runner, so any scheduling hiccup crosses 5s and reports a
// timeout that says nothing about the code. Timing out is still worth catching,
// so keep a ceiling — just one loose enough to only fire on a real hang. The
// per-chunk watchdog (chunkTimeoutMs) remains the backstop for a wedged process.
// Override with OMP_TEST_TIMEOUT (seconds); per-test `it(name, fn, ms)` still wins.
function testTimeoutMs(): number {
const raw = Number(Bun.env.OMP_TEST_TIMEOUT?.trim());
if (Number.isFinite(raw) && raw >= 1) return raw * 1000;
return 30_000;
}
// Materialize each chunk's argv against the pool width it will actually run at,
// rewriting `parallel` from the requested width to the granted one so later
// reporting reads the truth. A `parallel` request marks the command as a `bun
// test` invocation, so that is also where the shared per-test timeout is
// applied; the Rust task, which has neither, passes through untouched.
function applyChunkBudget(commands: TestCommand[], poolWidth: number): TestCommand[] {
const timeout = testTimeoutMs();
return commands.map(testCommand => {
if (testCommand.parallel === undefined) return testCommand;
const parallel = budgetedParallel(testCommand.parallel, poolWidth);
return {
...testCommand,
command: [...testCommand.command, `--parallel=${parallel}`, `--timeout=${timeout}`],
parallel,
};
});
}
// ANSI styling for interactive runs only; disabled when stdout is not a TTY or
// NO_COLOR is set, so CI logs and piped/aggregated output stay plain text.
const useColor = Boolean(process.stdout.isTTY) && !process.env.NO_COLOR;
@@ -694,9 +751,11 @@ export async function runTestCommandsInParallel(commands: TestCommand[], concurr
const queue = [...commands];
const failures: ChunkOutcome[] = [];
let completed = 0;
const fileWidths = [...new Set(commands.map(c => c.parallel).filter(p => p !== undefined))].sort((a, b) => a - b);
console.log(
`Running ${commands.length} test command(s), up to ${concurrency} in parallel ` +
`(OMP_TEST_CONCURRENCY=<n>|all to change).`,
`(OMP_TEST_CONCURRENCY=<n>|all to change); ${os.availableParallelism()} cores, ` +
`--parallel=${fileWidths.join("/") || "n/a"} per chunk.`,
);
// Incremental, cancellable drain into a mutable sink, so a watchdog-killed
@@ -847,13 +906,18 @@ if (import.meta.main) {
);
}
const testCommands = await commandsForMode(requestedMode as Mode);
const requestedCommands = await commandsForMode(requestedMode as Mode);
const explicitConcurrency = Boolean(Bun.env.OMP_TEST_CONCURRENCY?.trim());
// CI defaults to one process at a time, but memory-sized workflow buckets
// explicitly opt into bounded process concurrency. Local runs fan out by
// default and may use the same override.
if (!isDryRun && testCommands.length > 1 && (!isCI() || explicitConcurrency)) {
await runTestCommandsInParallel(testCommands, testConcurrency(testCommands.length));
// default and may use the same override. Resolved before the dry-run check so
// `--dry-run` prints the argv the real run would use, budget included.
const pooled = requestedCommands.length > 1 && (!isCI() || explicitConcurrency);
// The sequential path is a pool of one, so a lone chunk keeps the whole budget.
const poolWidth = pooled ? testConcurrency(requestedCommands.length) : 1;
const testCommands = applyChunkBudget(requestedCommands, poolWidth);
if (pooled && !isDryRun) {
await runTestCommandsInParallel(testCommands, poolWidth);
} else {
for (const testCommand of testCommands) {
await runTestCommand(testCommand);