fix(robomp): run sandbox setup/teardown off the event loop safely
Workspace setup/teardown (git clone/fetch, worktree add/remove, chown) ran synchronously on the asyncio dispatcher loop, so one stalled subprocess froze the entire process. - Offload every ensure_workspace/remove_workspace call to a worker thread via a new _run_workspace_op helper that drains the thread to completion on cancellation, so a cancelled event cannot reap/release a slot the setup thread still owns. - Serialize same-repo setup with a per-repo threading.RLock while letting distinct repos run concurrently. - Bound the direct git/chown subprocesses with a 120s timeout (returncode 124); treat a timed-out branch probe as an error rather than "branch absent" to avoid silently rebasing a follow-up onto the default branch and losing the PR's commits. - When a timed-out worktree remove leaves the checkout behind, rmtree it and run `git worktree prune` so the pool's dangling registration cannot trip a later worktree add for the same path. Adds regression tests for event-loop liveness, cancellation-safe offload, per-repo lock serialization, subprocess timeout mapping, the branch-probe timeout guard, and worktree-prune after a failed remove. Op: correct Restores: spec:dispatcher-event-loop-never-blocks-on-workspace-io
This commit is contained in:
+52
-12
@@ -2,9 +2,10 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from collections.abc import Mapping
|
||||
from typing import Any
|
||||
from collections.abc import Callable, Mapping
|
||||
from typing import Any, TypeVar
|
||||
|
||||
from robomp import persona
|
||||
from robomp.config import Settings
|
||||
@@ -23,6 +24,38 @@ from robomp.worker import DirectiveInfo, TaskInputs, ThreadMessage, run_task
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
_T = TypeVar("_T")
|
||||
|
||||
|
||||
async def _run_workspace_op(func: Callable[..., _T], /, **kwargs: object) -> _T:
|
||||
"""Offload a blocking sandbox workspace op to a worker thread, uncancellably.
|
||||
|
||||
Workspace setup/teardown (git clone/fetch, worktree add/remove, chown) is
|
||||
blocking, so it runs off the event loop via a thread. Unlike a bare
|
||||
``await asyncio.to_thread(...)``, cancelling the awaiting coroutine here does
|
||||
NOT detach the still-running thread: the thread holds the per-repo lock and
|
||||
owns the slot mid-setup, and ``_run_event``'s ``finally`` reaps/releases that
|
||||
slot on cancellation. If the await detached, the reaped slot could be reused
|
||||
while the thread is still touching it. So on cancellation we drain the thread
|
||||
to completion before propagating, gating the caller's ``finally`` behind the
|
||||
thread. The inner subprocesses are timeout-bounded, so completion is assured.
|
||||
"""
|
||||
inner = asyncio.ensure_future(asyncio.to_thread(func, **kwargs))
|
||||
try:
|
||||
return await asyncio.shield(inner)
|
||||
except asyncio.CancelledError:
|
||||
# A repeated cancel can interrupt even a shielded await, so loop until
|
||||
# the thread is actually done; swallow the inner's own outcome and
|
||||
# re-raise the cancellation the caller expects.
|
||||
while not inner.done():
|
||||
try:
|
||||
await asyncio.shield(inner)
|
||||
except asyncio.CancelledError:
|
||||
continue
|
||||
except BaseException:
|
||||
break
|
||||
raise
|
||||
|
||||
|
||||
def _comment_from_payload(payload: Mapping[str, Any]) -> CommentInfo:
|
||||
c = payload.get("comment") or {}
|
||||
@@ -267,7 +300,8 @@ async def triage_issue(
|
||||
return
|
||||
db.upsert_issue(key=key, repo=repo.full_name, number=issue.number, state="reproducing")
|
||||
clone_url = repo.clone_url
|
||||
workspace = sandbox.ensure_workspace(
|
||||
workspace = await _run_workspace_op(
|
||||
sandbox.ensure_workspace,
|
||||
repo=repo.full_name,
|
||||
number=issue.number,
|
||||
title=issue.title,
|
||||
@@ -341,7 +375,8 @@ async def review_pr(
|
||||
)
|
||||
|
||||
db.upsert_issue(key=key, repo=repo.full_name, number=pr_number, state="reviewing", pr_number=pr_number)
|
||||
workspace = sandbox.ensure_workspace(
|
||||
workspace = await _run_workspace_op(
|
||||
sandbox.ensure_workspace,
|
||||
repo=repo.full_name,
|
||||
number=pr_number,
|
||||
title=issue.title,
|
||||
@@ -406,7 +441,8 @@ async def handle_comment(
|
||||
# first and executes the directive in the same RPC turn.
|
||||
log.info("directive bootstrap", extra={"key": key, "author": directive.author})
|
||||
db.upsert_issue(key=key, repo=repo.full_name, number=issue.number, state="reproducing")
|
||||
workspace = sandbox.ensure_workspace(
|
||||
workspace = await _run_workspace_op(
|
||||
sandbox.ensure_workspace,
|
||||
repo=repo.full_name,
|
||||
number=issue.number,
|
||||
title=issue.title,
|
||||
@@ -456,9 +492,10 @@ async def handle_comment(
|
||||
# Maintainer reopen: tear down stale workspace, reset state, branch
|
||||
# afresh from default. The old branch may have been merged/deleted.
|
||||
log.info("directive reopen", extra={"key": key, "from_state": existing.state, "author": directive.author})
|
||||
sandbox.remove_workspace(repo=repo.full_name, number=issue.number)
|
||||
await _run_workspace_op(sandbox.remove_workspace, repo=repo.full_name, number=issue.number)
|
||||
db.upsert_issue(key=key, repo=repo.full_name, number=issue.number, state="reproducing")
|
||||
workspace = sandbox.ensure_workspace(
|
||||
workspace = await _run_workspace_op(
|
||||
sandbox.ensure_workspace,
|
||||
repo=repo.full_name,
|
||||
number=issue.number,
|
||||
title=issue.title,
|
||||
@@ -493,7 +530,8 @@ async def handle_comment(
|
||||
await run_task(task_kind="handle_comment", inputs=inputs, comment=comment, directive=directive)
|
||||
return
|
||||
|
||||
workspace = sandbox.ensure_workspace(
|
||||
workspace = await _run_workspace_op(
|
||||
sandbox.ensure_workspace,
|
||||
repo=repo.full_name,
|
||||
number=issue.number,
|
||||
title=issue.title,
|
||||
@@ -567,7 +605,8 @@ async def handle_review(
|
||||
log.warning("review fetch failed", extra={"err": str(exc)})
|
||||
return
|
||||
clone_url = repo.clone_url
|
||||
workspace = sandbox.ensure_workspace(
|
||||
workspace = await _run_workspace_op(
|
||||
sandbox.ensure_workspace,
|
||||
repo=repo.full_name,
|
||||
number=issue.number,
|
||||
title=issue.title,
|
||||
@@ -677,7 +716,7 @@ async def handle_pr_conversation(
|
||||
"directive reopen (pr)",
|
||||
extra={"key": issue_row.key, "from_state": issue_row.state, "author": directive.author},
|
||||
)
|
||||
sandbox.remove_workspace(repo=issue_row.repo, number=issue_row.number)
|
||||
await _run_workspace_op(sandbox.remove_workspace, repo=issue_row.repo, number=issue_row.number)
|
||||
db.upsert_issue(key=issue_row.key, repo=issue_row.repo, number=issue_row.number, state="reproducing")
|
||||
issue_row = db.get_issue(issue_row.key) or issue_row
|
||||
# Bare @mention with no request body — the route stashes an empty
|
||||
@@ -713,7 +752,8 @@ async def handle_pr_conversation(
|
||||
if existing_branch is None and not (directive and issue_row.state == "reproducing"):
|
||||
log.info("skip: pr-conversation PR missing branch mapping", extra={"repo": repo_full, "pr": pr_number})
|
||||
return
|
||||
workspace = sandbox.ensure_workspace(
|
||||
workspace = await _run_workspace_op(
|
||||
sandbox.ensure_workspace,
|
||||
repo=repo.full_name,
|
||||
number=issue.number,
|
||||
title=issue.title,
|
||||
@@ -797,7 +837,7 @@ async def cleanup_workspace(
|
||||
issue_row = db.get_issue(issue_key(repo_full, number))
|
||||
if issue_row is None:
|
||||
return
|
||||
sandbox.remove_workspace(repo=issue_row.repo, number=issue_row.number)
|
||||
await _run_workspace_op(sandbox.remove_workspace, repo=issue_row.repo, number=issue_row.number)
|
||||
db.set_issue_state(issue_row.key, target_state)
|
||||
log.info("cleanup", extra={"key": issue_row.key, "state": target_state})
|
||||
|
||||
|
||||
Reference in New Issue
Block a user