fix(typescript-edit-benchmark): resolved ts-edit-benchmark exit behavior

- Handled successful `main()` resolution by calling `process.exit(0)`.
- Preserved existing benchmark failure handling by logging the error and exiting with status 1.
This commit is contained in:
can1357
2026-05-26 11:48:45 +02:00
parent 2590bd4e28
commit 8b41c87ded
3 changed files with 56 additions and 43 deletions
@@ -12,7 +12,7 @@ import {
type CreateAgentSessionResult,
createAgentSession,
discoverAuthStorage,
type ModelRegistry,
ModelRegistry,
SessionManager,
Settings,
} from "@oh-my-pi/pi-coding-agent";
@@ -49,24 +49,28 @@ export interface DiscoverSharedInfraOptions {
/** Discover shared infrastructure once for the entire benchmark run. */
export async function discoverSharedInfra(options: DiscoverSharedInfraOptions = {}): Promise<SharedInfra> {
const { ModelRegistry: MR } = await import("@oh-my-pi/pi-coding-agent");
const authStorage = await discoverAuthStorage();
const modelRegistry = new MR(authStorage);
try {
const modelRegistry = new ModelRegistry(authStorage);
// Initialize global Settings singleton (required by code paths that use the global `settings` proxy)
const overrides: Record<string, unknown> = {};
if (options.editVariant && options.editVariant !== "auto") {
overrides["edit.mode"] = options.editVariant;
}
if (options.editFuzzy !== undefined && options.editFuzzy !== "auto") {
overrides["edit.fuzzyMatch"] = options.editFuzzy;
}
if (options.editFuzzyThreshold !== undefined && options.editFuzzyThreshold !== "auto") {
overrides["edit.fuzzyThreshold"] = options.editFuzzyThreshold;
}
await Settings.init({ cwd: options.cwd, overrides });
// Initialize global Settings singleton (required by code paths that use the global `settings` proxy)
const overrides: Record<string, unknown> = {};
if (options.editVariant && options.editVariant !== "auto") {
overrides["edit.mode"] = options.editVariant;
}
if (options.editFuzzy !== undefined && options.editFuzzy !== "auto") {
overrides["edit.fuzzyMatch"] = options.editFuzzy;
}
if (options.editFuzzyThreshold !== undefined && options.editFuzzyThreshold !== "auto") {
overrides["edit.fuzzyThreshold"] = options.editFuzzyThreshold;
}
await Settings.init({ cwd: options.cwd, overrides });
return { authStorage, modelRegistry };
return { authStorage, modelRegistry };
} catch (error) {
authStorage.close();
throw error;
}
}
/**
@@ -529,6 +529,11 @@ async function main(): Promise<void> {
if (cleanup) {
await cleanup();
}
// In-process benchmark runs can leave provider keep-alive sockets and
// background AgentSession timers alive after the report is written. Treat the
// final report as the CLI boundary so the command returns to the shell.
await postmortem.quit(0);
}
class LiveProgress {
@@ -711,7 +716,7 @@ class LiveProgress {
}
}
main().catch(err => {
main().catch(async err => {
console.error("Benchmark failed:", err);
process.exit(1);
await postmortem.quit(1);
});
@@ -42,7 +42,7 @@ type ConversationDumpSessionState = {
/** Common interface for both RPC and in-process clients */
interface BenchmarkClient {
start(): Promise<void>;
setThinkingLevel(level: import("@oh-my-pi/pi-agent-core").ResolvedThinkingLevel): Promise<void>;
setThinkingLevel(level: ResolvedThinkingLevel): Promise<void>;
onEvent(listener: (event: { type: string; [key: string]: unknown }) => void): () => void;
prompt(text: string): Promise<void>;
followUp(text: string): Promise<void>;
@@ -1887,32 +1887,36 @@ export async function runBenchmark(
})
: undefined;
const runItems: TaskRunItem[] = tasks.flatMap(task =>
Array.from({ length: config.runsPerTask }, (_, runIndex) => ({ task, runIndex })),
);
try {
const runItems: TaskRunItem[] = tasks.flatMap(task =>
Array.from({ length: config.runsPerTask }, (_, runIndex) => ({ task, runIndex })),
);
const pending = shuffle(runItems);
const resultsByTask = new Map<string, TaskRunResult[]>();
const concurrency = Math.max(1, Math.floor(config.taskConcurrency));
const running: Promise<void>[] = [];
const pending = shuffle(runItems);
const resultsByTask = new Map<string, TaskRunResult[]>();
const concurrency = Math.max(1, Math.floor(config.taskConcurrency));
const running: Promise<void>[] = [];
const runNext = async (): Promise<void> => {
const nextItem = pending.shift();
if (!nextItem) return;
const { task, result } = await runConcurrentBenchmarkRun(nextItem, config, onProgress, shared);
const list = resultsByTask.get(task.id) ?? [];
list.push(result);
resultsByTask.set(task.id, list);
onResultSnapshot?.(buildBenchmarkResult({ tasks, config, resultsByTask, startTime }));
await runNext();
};
const runNext = async (): Promise<void> => {
const nextItem = pending.shift();
if (!nextItem) return;
const { task, result } = await runConcurrentBenchmarkRun(nextItem, config, onProgress, shared);
const list = resultsByTask.get(task.id) ?? [];
list.push(result);
resultsByTask.set(task.id, list);
onResultSnapshot?.(buildBenchmarkResult({ tasks, config, resultsByTask, startTime }));
await runNext();
};
const slots = Math.min(concurrency, pending.length);
for (let i = 0; i < slots; i++) {
running.push(runNext());
const slots = Math.min(concurrency, pending.length);
for (let i = 0; i < slots; i++) {
running.push(runNext());
}
await Promise.all(running);
return buildBenchmarkResult({ tasks, config, resultsByTask, startTime });
} finally {
shared?.authStorage.close();
}
await Promise.all(running);
return buildBenchmarkResult({ tasks, config, resultsByTask, startTime });
}