fix(robomp): run sandbox setup/teardown off the event loop safely

Workspace setup/teardown (git clone/fetch, worktree add/remove, chown)
ran synchronously on the asyncio dispatcher loop, so one stalled
subprocess froze the entire process.

- Offload every ensure_workspace/remove_workspace call to a worker thread
  via a new _run_workspace_op helper that drains the thread to completion
  on cancellation, so a cancelled event cannot reap/release a slot the
  setup thread still owns.
- Serialize same-repo setup with a per-repo threading.RLock while letting
  distinct repos run concurrently.
- Bound the direct git/chown subprocesses with a 120s timeout
  (returncode 124); treat a timed-out branch probe as an error rather
  than "branch absent" to avoid silently rebasing a follow-up onto the
  default branch and losing the PR's commits.
- When a timed-out worktree remove leaves the checkout behind, rmtree it
  and run `git worktree prune` so the pool's dangling registration cannot
  trip a later worktree add for the same path.

Adds regression tests for event-loop liveness, cancellation-safe offload,
per-repo lock serialization, subprocess timeout mapping, the branch-probe
timeout guard, and worktree-prune after a failed remove.

Op: correct
Restores: spec:dispatcher-event-loop-never-blocks-on-workspace-io
This commit is contained in:
metaphorics
2026-07-02 07:46:36 +09:00
parent 0823892295
commit 6d16c19a04
5 changed files with 655 additions and 152 deletions
+52 -12
View File
@@ -2,9 +2,10 @@
from __future__ import annotations
import asyncio
import logging
from collections.abc import Mapping
from typing import Any
from collections.abc import Callable, Mapping
from typing import Any, TypeVar
from robomp import persona
from robomp.config import Settings
@@ -23,6 +24,38 @@ from robomp.worker import DirectiveInfo, TaskInputs, ThreadMessage, run_task
log = logging.getLogger(__name__)
_T = TypeVar("_T")
async def _run_workspace_op(func: Callable[..., _T], /, **kwargs: object) -> _T:
"""Offload a blocking sandbox workspace op to a worker thread, uncancellably.
Workspace setup/teardown (git clone/fetch, worktree add/remove, chown) is
blocking, so it runs off the event loop via a thread. Unlike a bare
``await asyncio.to_thread(...)``, cancelling the awaiting coroutine here does
NOT detach the still-running thread: the thread holds the per-repo lock and
owns the slot mid-setup, and ``_run_event``'s ``finally`` reaps/releases that
slot on cancellation. If the await detached, the reaped slot could be reused
while the thread is still touching it. So on cancellation we drain the thread
to completion before propagating, gating the caller's ``finally`` behind the
thread. The inner subprocesses are timeout-bounded, so completion is assured.
"""
inner = asyncio.ensure_future(asyncio.to_thread(func, **kwargs))
try:
return await asyncio.shield(inner)
except asyncio.CancelledError:
# A repeated cancel can interrupt even a shielded await, so loop until
# the thread is actually done; swallow the inner's own outcome and
# re-raise the cancellation the caller expects.
while not inner.done():
try:
await asyncio.shield(inner)
except asyncio.CancelledError:
continue
except BaseException:
break
raise
def _comment_from_payload(payload: Mapping[str, Any]) -> CommentInfo:
c = payload.get("comment") or {}
@@ -267,7 +300,8 @@ async def triage_issue(
return
db.upsert_issue(key=key, repo=repo.full_name, number=issue.number, state="reproducing")
clone_url = repo.clone_url
workspace = sandbox.ensure_workspace(
workspace = await _run_workspace_op(
sandbox.ensure_workspace,
repo=repo.full_name,
number=issue.number,
title=issue.title,
@@ -341,7 +375,8 @@ async def review_pr(
)
db.upsert_issue(key=key, repo=repo.full_name, number=pr_number, state="reviewing", pr_number=pr_number)
workspace = sandbox.ensure_workspace(
workspace = await _run_workspace_op(
sandbox.ensure_workspace,
repo=repo.full_name,
number=pr_number,
title=issue.title,
@@ -406,7 +441,8 @@ async def handle_comment(
# first and executes the directive in the same RPC turn.
log.info("directive bootstrap", extra={"key": key, "author": directive.author})
db.upsert_issue(key=key, repo=repo.full_name, number=issue.number, state="reproducing")
workspace = sandbox.ensure_workspace(
workspace = await _run_workspace_op(
sandbox.ensure_workspace,
repo=repo.full_name,
number=issue.number,
title=issue.title,
@@ -456,9 +492,10 @@ async def handle_comment(
# Maintainer reopen: tear down stale workspace, reset state, branch
# afresh from default. The old branch may have been merged/deleted.
log.info("directive reopen", extra={"key": key, "from_state": existing.state, "author": directive.author})
sandbox.remove_workspace(repo=repo.full_name, number=issue.number)
await _run_workspace_op(sandbox.remove_workspace, repo=repo.full_name, number=issue.number)
db.upsert_issue(key=key, repo=repo.full_name, number=issue.number, state="reproducing")
workspace = sandbox.ensure_workspace(
workspace = await _run_workspace_op(
sandbox.ensure_workspace,
repo=repo.full_name,
number=issue.number,
title=issue.title,
@@ -493,7 +530,8 @@ async def handle_comment(
await run_task(task_kind="handle_comment", inputs=inputs, comment=comment, directive=directive)
return
workspace = sandbox.ensure_workspace(
workspace = await _run_workspace_op(
sandbox.ensure_workspace,
repo=repo.full_name,
number=issue.number,
title=issue.title,
@@ -567,7 +605,8 @@ async def handle_review(
log.warning("review fetch failed", extra={"err": str(exc)})
return
clone_url = repo.clone_url
workspace = sandbox.ensure_workspace(
workspace = await _run_workspace_op(
sandbox.ensure_workspace,
repo=repo.full_name,
number=issue.number,
title=issue.title,
@@ -677,7 +716,7 @@ async def handle_pr_conversation(
"directive reopen (pr)",
extra={"key": issue_row.key, "from_state": issue_row.state, "author": directive.author},
)
sandbox.remove_workspace(repo=issue_row.repo, number=issue_row.number)
await _run_workspace_op(sandbox.remove_workspace, repo=issue_row.repo, number=issue_row.number)
db.upsert_issue(key=issue_row.key, repo=issue_row.repo, number=issue_row.number, state="reproducing")
issue_row = db.get_issue(issue_row.key) or issue_row
# Bare @mention with no request body — the route stashes an empty
@@ -713,7 +752,8 @@ async def handle_pr_conversation(
if existing_branch is None and not (directive and issue_row.state == "reproducing"):
log.info("skip: pr-conversation PR missing branch mapping", extra={"repo": repo_full, "pr": pr_number})
return
workspace = sandbox.ensure_workspace(
workspace = await _run_workspace_op(
sandbox.ensure_workspace,
repo=repo.full_name,
number=issue.number,
title=issue.title,
@@ -797,7 +837,7 @@ async def cleanup_workspace(
issue_row = db.get_issue(issue_key(repo_full, number))
if issue_row is None:
return
sandbox.remove_workspace(repo=issue_row.repo, number=issue_row.number)
await _run_workspace_op(sandbox.remove_workspace, repo=issue_row.repo, number=issue_row.number)
db.set_issue_state(issue_row.key, target_state)
log.info("cleanup", extra={"key": issue_row.key, "state": target_state})