feat(robomp): added repo-scoped issue search to issue triage flow
- Added `search_issues` support to the GitHub backend and client, including `state_reason` and `is_pull_request` in issue summaries. - Added the `gh_search_issues` host tool with repo-prefixed query handling, non-empty/restricted `repo:` validation, default and bounded `limit` values, and inbound issue filtering. - Added proxy integration for issue search with a new `/gh/v1/search_issues` endpoint and matching proxy-client method/response parsing. - Updated triage prompts to perform pre-`classify_issue` duplicate and already-fixed checks via search, and added tests for search query formatting, validation, and state-aware match rendering.
This commit is contained in:
@@ -46,6 +46,8 @@ class GitHubBackend(Protocol):
|
||||
limit: int = 30,
|
||||
) -> list[IssueSummary]: ...
|
||||
|
||||
async def search_issues(self, repo: str, query: str, *, limit: int = 10) -> list[IssueSummary]: ...
|
||||
|
||||
async def list_comments(self, repo: str, number: int) -> list[CommentInfo]: ...
|
||||
|
||||
async def list_review_comments(self, repo: str, pr_number: int) -> list[ReviewCommentInfo]: ...
|
||||
|
||||
@@ -116,6 +116,10 @@ class IssueSummary:
|
||||
updated_at: str
|
||||
created_at: str
|
||||
html_url: str
|
||||
# `completed` / `not_planned` / `reopened` when closed; empty otherwise.
|
||||
state_reason: str = ""
|
||||
# Search results mix issues and PRs; list_issues always yields issues.
|
||||
is_pull_request: bool = False
|
||||
|
||||
|
||||
@dataclass(slots=True, frozen=True)
|
||||
@@ -335,24 +339,26 @@ class GitHubClient:
|
||||
for item in data or []:
|
||||
if "pull_request" in item:
|
||||
continue # GitHub's /issues endpoint also returns PRs; skip them.
|
||||
user = item.get("user") or {}
|
||||
labels_raw = item.get("labels") or []
|
||||
out.append(
|
||||
IssueSummary(
|
||||
repo=repo,
|
||||
number=int(item["number"]),
|
||||
title=str(item.get("title") or ""),
|
||||
state=str(item.get("state") or "open"),
|
||||
author=str(user.get("login") or ""),
|
||||
labels=tuple(str(lbl["name"]) if isinstance(lbl, dict) else str(lbl) for lbl in labels_raw),
|
||||
comments=int(item.get("comments") or 0),
|
||||
updated_at=str(item.get("updated_at") or ""),
|
||||
created_at=str(item.get("created_at") or ""),
|
||||
html_url=str(item.get("html_url") or ""),
|
||||
)
|
||||
)
|
||||
out.append(_summary_from_item(repo, item))
|
||||
return out
|
||||
|
||||
async def search_issues(self, repo: str, query: str, *, limit: int = 10) -> list[IssueSummary]:
|
||||
"""Search issues AND pull requests in `repo` using GitHub issue-search syntax.
|
||||
|
||||
`query` takes bare keywords plus qualifiers (`is:pr`, `is:closed`,
|
||||
`label:bug`, `in:title`, …); the `repo:` scope is applied here. Results
|
||||
come back in GitHub's best-match order. `limit` is capped at 30 — this
|
||||
serves triage lookups (duplicates, prior fixes), not pagination.
|
||||
"""
|
||||
per_page = max(1, min(int(limit), 30))
|
||||
data = await self.request(
|
||||
"GET",
|
||||
"/search/issues",
|
||||
params={"q": f"repo:{repo} {query}".strip(), "per_page": per_page},
|
||||
)
|
||||
items = (data or {}).get("items") or []
|
||||
return [_summary_from_item(repo, item) for item in items]
|
||||
|
||||
async def list_comments(self, repo: str, number: int) -> list[CommentInfo]:
|
||||
data = await self.request("GET", f"/repos/{repo}/issues/{number}/comments", params={"per_page": 100})
|
||||
return [_comment_from_payload(item) for item in (data or [])]
|
||||
@@ -576,6 +582,26 @@ def _pr_review_from_payload(data: Mapping[str, Any]) -> PullRequestReviewInfo:
|
||||
)
|
||||
|
||||
|
||||
def _summary_from_item(repo: str, item: Mapping[str, Any]) -> IssueSummary:
|
||||
"""Build an `IssueSummary` from a REST issue object (list or search shape)."""
|
||||
user = item.get("user") or {}
|
||||
labels_raw = item.get("labels") or []
|
||||
return IssueSummary(
|
||||
repo=repo,
|
||||
number=int(item["number"]),
|
||||
title=str(item.get("title") or ""),
|
||||
state=str(item.get("state") or "open"),
|
||||
author=str(user.get("login") or ""),
|
||||
labels=tuple(str(lbl["name"]) if isinstance(lbl, dict) else str(lbl) for lbl in labels_raw),
|
||||
comments=int(item.get("comments") or 0),
|
||||
updated_at=str(item.get("updated_at") or ""),
|
||||
created_at=str(item.get("created_at") or ""),
|
||||
html_url=str(item.get("html_url") or ""),
|
||||
state_reason=str(item.get("state_reason") or ""),
|
||||
is_pull_request="pull_request" in item,
|
||||
)
|
||||
|
||||
|
||||
def _pr_file_from_payload(data: Mapping[str, Any]) -> PullRequestFileInfo:
|
||||
return PullRequestFileInfo(
|
||||
path=str(data.get("filename") or data.get("path") or ""),
|
||||
|
||||
@@ -1256,6 +1256,75 @@ def _build_fetch_thread(bindings: ToolBindings) -> HostTool[Any, Any]:
|
||||
)
|
||||
|
||||
|
||||
# ---------- gh_search_issues ----------
|
||||
_REPO_QUALIFIER_RE = re.compile(r"(?i)\brepo:")
|
||||
|
||||
|
||||
def _build_search_issues(bindings: ToolBindings) -> HostTool[Any, Any]:
|
||||
"""Read-only issue/PR search scoped to the current repo.
|
||||
|
||||
Exists so triage can find duplicates and already-merged fixes instead of
|
||||
classifying blind; the inbound issue itself is filtered out of results.
|
||||
"""
|
||||
|
||||
def execute(args: dict[str, Any], _ctx: HostToolContext[Any]) -> str:
|
||||
query = args.get("query")
|
||||
if not isinstance(query, str) or not query.strip():
|
||||
msg = "gh_search_issues requires a non-empty 'query'."
|
||||
_audit(bindings, "gh_search_issues", args, error=msg)
|
||||
_raise_command(msg)
|
||||
query = query.strip()
|
||||
if _REPO_QUALIFIER_RE.search(query):
|
||||
msg = "gh_search_issues scopes to the current repo automatically; drop the 'repo:' qualifier."
|
||||
_audit(bindings, "gh_search_issues", args, error=msg)
|
||||
_raise_command(msg)
|
||||
limit_raw = args.get("limit")
|
||||
limit = max(1, min(int(limit_raw), 20)) if isinstance(limit_raw, int) else 10
|
||||
try:
|
||||
results = _run_coro(
|
||||
bindings.loop,
|
||||
bindings.github.search_issues(bindings.repo.full_name, query, limit=limit),
|
||||
)
|
||||
except GitHubError as exc:
|
||||
_audit(bindings, "gh_search_issues", args, error=str(exc))
|
||||
_raise_command(f"GitHub search failed: {exc.status} {exc.message}")
|
||||
results = [s for s in results if s.is_pull_request or s.number != bindings.issue.number]
|
||||
if not results:
|
||||
_audit(bindings, "gh_search_issues", args, result={"matches": 0})
|
||||
return f"No issues or PRs in {bindings.repo.full_name} match {query!r}."
|
||||
lines = [f"# {len(results)} match(es) for {query!r} in {bindings.repo.full_name}"]
|
||||
for s in results:
|
||||
kind = "PR" if s.is_pull_request else "issue"
|
||||
state = f"{s.state} ({s.state_reason})" if s.state_reason else s.state
|
||||
labels = f" [{', '.join(s.labels)}]" if s.labels else ""
|
||||
lines.append(
|
||||
f"- #{s.number} ({kind}, {state}) {s.title} — @{s.author}, updated {s.updated_at[:10]}{labels}"
|
||||
)
|
||||
_audit(bindings, "gh_search_issues", args, result={"matches": len(results)})
|
||||
return "\n".join(lines)
|
||||
|
||||
return host_tool(
|
||||
name="gh_search_issues",
|
||||
description=persona.host_tool_description("gh_search_issues"),
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"query": {
|
||||
"type": "string",
|
||||
"description": persona.host_tool_parameter_description("gh_search_issues", "query"),
|
||||
},
|
||||
"limit": {
|
||||
"type": "integer",
|
||||
"description": persona.host_tool_parameter_description("gh_search_issues", "limit"),
|
||||
},
|
||||
},
|
||||
"required": ["query"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
execute=execute,
|
||||
)
|
||||
|
||||
|
||||
_PRIMARY_TYPES = ("bug", "enhancement", "question", "proposal", "documentation", "wontfix", "invalid", "duplicate")
|
||||
_AUTO_PR_CLASSIFICATIONS = frozenset({"bug", "documentation"})
|
||||
_PRIORITIES = ("prio:p0", "prio:p1", "prio:p2", "prio:p3")
|
||||
@@ -1835,6 +1904,7 @@ def build(bindings: ToolBindings) -> tuple[HostTool[Any, Any], ...]:
|
||||
_build_mark_unable(bindings),
|
||||
_build_abort_task(bindings),
|
||||
_build_fetch_thread(bindings),
|
||||
_build_search_issues(bindings),
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -72,6 +72,13 @@ reason = "Internal diagnosis for the operator. Concrete, specific, blameless. NE
|
||||
[fetch_issue_thread]
|
||||
description = "Refetch the originating issue and its comments. Use sparingly."
|
||||
|
||||
[gh_search_issues]
|
||||
description = "Search issues AND pull requests in the current repo (GitHub issue-search syntax; repo scope applied automatically). Use during triage to find duplicates and to check whether a merged PR already fixed the reported problem before classifying."
|
||||
|
||||
[gh_search_issues.parameters]
|
||||
query = "GitHub issue-search syntax: bare keywords plus qualifiers like `is:pr`, `is:closed`, `is:merged`, `label:bug`, `in:title`, `author:<login>`. NEVER include a `repo:` qualifier — scope is applied automatically."
|
||||
limit = "Max results, 1-20. Default 10."
|
||||
|
||||
[set_issue_labels]
|
||||
description = "Append labels to the originating issue/PR. NEVER removes existing labels."
|
||||
|
||||
|
||||
@@ -16,7 +16,9 @@ Worktree is at cwd; the branch above is checked out and ready for commits **if**
|
||||
the classification calls for code. Drive the todo list to completion:
|
||||
|
||||
1. **Triage first.** Read the body and any comments via `read` /
|
||||
`fetch_issue_thread`, then call
|
||||
`fetch_issue_thread`. Run `gh_search_issues` for duplicates and
|
||||
already-merged fixes — the reporter may be on an older release than your
|
||||
worktree. Then call
|
||||
`classify_issue(primary=..., priority=..., functional=[...], rationale=...)`.
|
||||
Apply the **merit gate** from the system prompt before picking `bug`:
|
||||
broken contract, demonstrated impact, deliberate-tradeoff check, upstream
|
||||
|
||||
@@ -21,7 +21,14 @@ Pick exactly ONE primary label per issue:
|
||||
| `proposal` | Design/process proposal requiring maintainer decision. Comment with thoughts; no PR. |
|
||||
| `question` | How-to, clarification, or usage question. Answer in one comment. |
|
||||
| `invalid` | Spam, off-topic, or not actionable. One brief explanatory comment. |
|
||||
| `duplicate` | Clear duplicate of another issue. Cite the original; no PR. |
|
||||
| `duplicate` | Duplicate of another issue, or already fixed by a merged PR / newer release. Cite the original or the fixing PR; no new PR. |
|
||||
|
||||
## Duplicate & already-fixed check
|
||||
|
||||
Before `classify_issue`, run `gh_search_issues` with the report's key terms (retry with synonyms and an `is:pr` variant — one search proves nothing):
|
||||
|
||||
- **Prior issue on the same problem** → `duplicate`, cite it. A prior closure as not-planned/`wontfix` on the same complaint is binding precedent — adopt that verdict; NEVER relitigate it.
|
||||
- **Already fixed.** Your worktree is the CURRENT default branch; reporters often run older releases. When the reported version lags the latest release (topmost released section of the relevant `packages/*/CHANGELOG.md`), check the changelog and merged PRs (`is:pr is:merged <keywords>`) for an existing fix, and try the repro against the worktree — failing on the reporter's version but passing here means it is already fixed. Classify `duplicate`: cite the fixing PR, name the release carrying it (or say it ships in the next release when still under `[Unreleased]`), and tell the reporter to update. NEVER re-fix what main already fixed.
|
||||
|
||||
## Merit gate — `bug` vs `wontfix` vs `enhancement`
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
name = "Classify"
|
||||
tasks = [
|
||||
"Read the issue body + every prior comment",
|
||||
"gh_search_issues for duplicates and already-merged fixes",
|
||||
"Call classify_issue with primary type + labels",
|
||||
]
|
||||
|
||||
|
||||
@@ -513,6 +513,16 @@ def create_proxy_app(settings: Settings) -> FastAPI:
|
||||
return _gh_error_response(exc)
|
||||
return JSONResponse({"items": [_serialize(s) for s in items]})
|
||||
|
||||
@app.get("/gh/v1/search_issues")
|
||||
async def search_issues(request: Request, repo: str, q: str, limit: int = 10) -> JSONResponse:
|
||||
await _authenticate(request)
|
||||
github: GitHubClient = request.app.state.github
|
||||
try:
|
||||
items = await github.search_issues(repo, q, limit=limit)
|
||||
except GitHubError as exc:
|
||||
return _gh_error_response(exc)
|
||||
return JSONResponse({"items": [_serialize(s) for s in items]})
|
||||
|
||||
@app.get("/gh/v1/comments")
|
||||
async def list_comments(request: Request, repo: str, number: int) -> JSONResponse:
|
||||
await _authenticate(request)
|
||||
|
||||
@@ -204,6 +204,14 @@ class GitHubProxyClient:
|
||||
)
|
||||
return [_issue_summary_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []]
|
||||
|
||||
async def search_issues(self, repo: str, query: str, *, limit: int = 10) -> list[IssueSummary]:
|
||||
data = await self._request(
|
||||
"GET",
|
||||
"/gh/v1/search_issues",
|
||||
params={"repo": repo, "q": query, "limit": limit},
|
||||
)
|
||||
return [_issue_summary_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []]
|
||||
|
||||
async def list_comments(self, repo: str, number: int) -> list[CommentInfo]:
|
||||
data = await self._request("GET", "/gh/v1/comments", params={"repo": repo, "number": number})
|
||||
return [_comment_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []]
|
||||
@@ -498,6 +506,8 @@ def _issue_summary_from(data: Any) -> IssueSummary:
|
||||
updated_at=str(data.get("updated_at") or ""),
|
||||
created_at=str(data.get("created_at") or ""),
|
||||
html_url=str(data.get("html_url") or ""),
|
||||
state_reason=str(data.get("state_reason") or ""),
|
||||
is_pull_request=bool(data.get("is_pull_request")),
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -198,7 +198,7 @@ def _ensure_agent_run_dir() -> None:
|
||||
return
|
||||
try:
|
||||
run_dir.mkdir(parents=True, exist_ok=True)
|
||||
for root, dirs, files in os.walk(run_dir):
|
||||
for root, _dirs, files in os.walk(run_dir):
|
||||
root_path = Path(root)
|
||||
os.chown(root_path, -1, gid)
|
||||
root_path.chmod(0o2770)
|
||||
|
||||
@@ -861,6 +861,82 @@ def test_classify_issue_wontfix_takes_comment_only_path(db: Database, tmp_path:
|
||||
assert row is not None and row.classification == "wontfix"
|
||||
|
||||
|
||||
def test_gh_search_issues_scopes_repo_and_renders_matches(db: Database, tmp_path: Path) -> None:
|
||||
"""Search auto-prefixes the repo scope, surfaces PR/state_reason so triage can
|
||||
spot prior fixes and not-planned precedents, and filters the inbound issue."""
|
||||
captured: dict[str, Any] = {}
|
||||
|
||||
def handler(request: httpx.Request) -> httpx.Response:
|
||||
captured["q"] = request.url.params["q"]
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"total_count": 3,
|
||||
"items": [
|
||||
{
|
||||
"number": 42, # the inbound issue itself — must be filtered
|
||||
"title": "boom",
|
||||
"state": "open",
|
||||
"user": {"login": "alice"},
|
||||
"labels": [],
|
||||
"comments": 0,
|
||||
"updated_at": "2026-07-01T00:00:00Z",
|
||||
"created_at": "2026-07-01T00:00:00Z",
|
||||
"html_url": "https://example/42",
|
||||
},
|
||||
{
|
||||
"number": 30,
|
||||
"title": "same crash on resize",
|
||||
"state": "closed",
|
||||
"state_reason": "not_planned",
|
||||
"user": {"login": "bob"},
|
||||
"labels": [{"name": "wontfix"}],
|
||||
"comments": 3,
|
||||
"updated_at": "2026-06-01T00:00:00Z",
|
||||
"created_at": "2026-05-01T00:00:00Z",
|
||||
"html_url": "https://example/30",
|
||||
},
|
||||
{
|
||||
"number": 31,
|
||||
"title": "fix: resize crash",
|
||||
"state": "closed",
|
||||
"state_reason": "completed",
|
||||
"user": {"login": "bot"},
|
||||
"labels": [],
|
||||
"comments": 1,
|
||||
"updated_at": "2026-06-02T00:00:00Z",
|
||||
"created_at": "2026-06-02T00:00:00Z",
|
||||
"html_url": "https://example/pull/31",
|
||||
"pull_request": {"url": "https://example/pull/31"},
|
||||
},
|
||||
],
|
||||
},
|
||||
)
|
||||
|
||||
bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(handler))
|
||||
try:
|
||||
tool = next(x for x in build(bindings) if x.name == "gh_search_issues")
|
||||
result = tool.execute({"query": "resize crash"}, _ctx())
|
||||
finally:
|
||||
_stop_loop(loop, t)
|
||||
assert captured["q"] == "repo:octo/widget resize crash"
|
||||
assert "#42" not in result # inbound issue filtered out
|
||||
assert "#30 (issue, closed (not_planned))" in result
|
||||
assert "#31 (PR, closed (completed))" in result
|
||||
|
||||
|
||||
def test_gh_search_issues_rejects_repo_qualifier_and_empty_query(db: Database, tmp_path: Path) -> None:
|
||||
bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(lambda r: httpx.Response(500)))
|
||||
try:
|
||||
tool = next(x for x in build(bindings) if x.name == "gh_search_issues")
|
||||
with pytest.raises(RpcCommandError):
|
||||
tool.execute({"query": "repo:evil/elsewhere secrets"}, _ctx())
|
||||
with pytest.raises(RpcCommandError):
|
||||
tool.execute({"query": " "}, _ctx())
|
||||
finally:
|
||||
_stop_loop(loop, t)
|
||||
|
||||
|
||||
def test_classify_issue_rejects_bug_without_priority(db: Database, tmp_path: Path) -> None:
|
||||
bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(lambda r: httpx.Response(500)))
|
||||
try:
|
||||
|
||||
@@ -251,6 +251,29 @@ def round_trip_app(proxy_settings: Settings):
|
||||
}
|
||||
],
|
||||
)
|
||||
if path == "/search/issues" and req.method == "GET":
|
||||
assert req.url.params["q"].startswith("repo:octo/widget ")
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"total_count": 1,
|
||||
"items": [
|
||||
{
|
||||
"number": 9,
|
||||
"title": "fixed it",
|
||||
"state": "closed",
|
||||
"state_reason": "completed",
|
||||
"user": {"login": "bob"},
|
||||
"labels": [{"name": "bug"}],
|
||||
"comments": 2,
|
||||
"updated_at": "2026-02-01T00:00:00Z",
|
||||
"created_at": "2026-01-15T00:00:00Z",
|
||||
"html_url": "https://example/9",
|
||||
"pull_request": {"url": "https://example/pull/9"},
|
||||
}
|
||||
],
|
||||
},
|
||||
)
|
||||
if path == "/repos/octo/widget/issues/1/comments" and req.method == "GET":
|
||||
return httpx.Response(
|
||||
200,
|
||||
@@ -365,6 +388,10 @@ async def test_round_trip_all_endpoints(round_trip_app) -> None:
|
||||
issues = await client.list_issues("octo/widget")
|
||||
assert len(issues) == 1 and isinstance(issues[0], IssueSummary)
|
||||
|
||||
found = await client.search_issues("octo/widget", "colon selector is:pr")
|
||||
assert len(found) == 1 and isinstance(found[0], IssueSummary)
|
||||
assert found[0].is_pull_request and found[0].state_reason == "completed"
|
||||
|
||||
comments = await client.list_comments("octo/widget", 1)
|
||||
assert len(comments) == 1 and isinstance(comments[0], CommentInfo)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user