feat(robomp): added repo-scoped issue search to issue triage flow

- Added `search_issues` support to the GitHub backend and client, including `state_reason` and `is_pull_request` in issue summaries.
- Added the `gh_search_issues` host tool with repo-prefixed query handling, non-empty/restricted `repo:` validation, default and bounded `limit` values, and inbound issue filtering.
- Added proxy integration for issue search with a new `/gh/v1/search_issues` endpoint and matching proxy-client method/response parsing.
- Updated triage prompts to perform pre-`classify_issue` duplicate and already-fixed checks via search, and added tests for search query formatting, validation, and state-aware match rendering.
This commit is contained in:
can1357
2026-07-15 00:30:15 +02:00
parent 539c132094
commit 3e6ae36c7c
12 changed files with 257 additions and 19 deletions
+2
View File
@@ -46,6 +46,8 @@ class GitHubBackend(Protocol):
limit: int = 30,
) -> list[IssueSummary]: ...
async def search_issues(self, repo: str, query: str, *, limit: int = 10) -> list[IssueSummary]: ...
async def list_comments(self, repo: str, number: int) -> list[CommentInfo]: ...
async def list_review_comments(self, repo: str, pr_number: int) -> list[ReviewCommentInfo]: ...
+42 -16
View File
@@ -116,6 +116,10 @@ class IssueSummary:
updated_at: str
created_at: str
html_url: str
# `completed` / `not_planned` / `reopened` when closed; empty otherwise.
state_reason: str = ""
# Search results mix issues and PRs; list_issues always yields issues.
is_pull_request: bool = False
@dataclass(slots=True, frozen=True)
@@ -335,24 +339,26 @@ class GitHubClient:
for item in data or []:
if "pull_request" in item:
continue # GitHub's /issues endpoint also returns PRs; skip them.
user = item.get("user") or {}
labels_raw = item.get("labels") or []
out.append(
IssueSummary(
repo=repo,
number=int(item["number"]),
title=str(item.get("title") or ""),
state=str(item.get("state") or "open"),
author=str(user.get("login") or ""),
labels=tuple(str(lbl["name"]) if isinstance(lbl, dict) else str(lbl) for lbl in labels_raw),
comments=int(item.get("comments") or 0),
updated_at=str(item.get("updated_at") or ""),
created_at=str(item.get("created_at") or ""),
html_url=str(item.get("html_url") or ""),
)
)
out.append(_summary_from_item(repo, item))
return out
async def search_issues(self, repo: str, query: str, *, limit: int = 10) -> list[IssueSummary]:
"""Search issues AND pull requests in `repo` using GitHub issue-search syntax.
`query` takes bare keywords plus qualifiers (`is:pr`, `is:closed`,
`label:bug`, `in:title`, …); the `repo:` scope is applied here. Results
come back in GitHub's best-match order. `limit` is capped at 30 — this
serves triage lookups (duplicates, prior fixes), not pagination.
"""
per_page = max(1, min(int(limit), 30))
data = await self.request(
"GET",
"/search/issues",
params={"q": f"repo:{repo} {query}".strip(), "per_page": per_page},
)
items = (data or {}).get("items") or []
return [_summary_from_item(repo, item) for item in items]
async def list_comments(self, repo: str, number: int) -> list[CommentInfo]:
data = await self.request("GET", f"/repos/{repo}/issues/{number}/comments", params={"per_page": 100})
return [_comment_from_payload(item) for item in (data or [])]
@@ -576,6 +582,26 @@ def _pr_review_from_payload(data: Mapping[str, Any]) -> PullRequestReviewInfo:
)
def _summary_from_item(repo: str, item: Mapping[str, Any]) -> IssueSummary:
"""Build an `IssueSummary` from a REST issue object (list or search shape)."""
user = item.get("user") or {}
labels_raw = item.get("labels") or []
return IssueSummary(
repo=repo,
number=int(item["number"]),
title=str(item.get("title") or ""),
state=str(item.get("state") or "open"),
author=str(user.get("login") or ""),
labels=tuple(str(lbl["name"]) if isinstance(lbl, dict) else str(lbl) for lbl in labels_raw),
comments=int(item.get("comments") or 0),
updated_at=str(item.get("updated_at") or ""),
created_at=str(item.get("created_at") or ""),
html_url=str(item.get("html_url") or ""),
state_reason=str(item.get("state_reason") or ""),
is_pull_request="pull_request" in item,
)
def _pr_file_from_payload(data: Mapping[str, Any]) -> PullRequestFileInfo:
return PullRequestFileInfo(
path=str(data.get("filename") or data.get("path") or ""),
+70
View File
@@ -1256,6 +1256,75 @@ def _build_fetch_thread(bindings: ToolBindings) -> HostTool[Any, Any]:
)
# ---------- gh_search_issues ----------
_REPO_QUALIFIER_RE = re.compile(r"(?i)\brepo:")
def _build_search_issues(bindings: ToolBindings) -> HostTool[Any, Any]:
"""Read-only issue/PR search scoped to the current repo.
Exists so triage can find duplicates and already-merged fixes instead of
classifying blind; the inbound issue itself is filtered out of results.
"""
def execute(args: dict[str, Any], _ctx: HostToolContext[Any]) -> str:
query = args.get("query")
if not isinstance(query, str) or not query.strip():
msg = "gh_search_issues requires a non-empty 'query'."
_audit(bindings, "gh_search_issues", args, error=msg)
_raise_command(msg)
query = query.strip()
if _REPO_QUALIFIER_RE.search(query):
msg = "gh_search_issues scopes to the current repo automatically; drop the 'repo:' qualifier."
_audit(bindings, "gh_search_issues", args, error=msg)
_raise_command(msg)
limit_raw = args.get("limit")
limit = max(1, min(int(limit_raw), 20)) if isinstance(limit_raw, int) else 10
try:
results = _run_coro(
bindings.loop,
bindings.github.search_issues(bindings.repo.full_name, query, limit=limit),
)
except GitHubError as exc:
_audit(bindings, "gh_search_issues", args, error=str(exc))
_raise_command(f"GitHub search failed: {exc.status} {exc.message}")
results = [s for s in results if s.is_pull_request or s.number != bindings.issue.number]
if not results:
_audit(bindings, "gh_search_issues", args, result={"matches": 0})
return f"No issues or PRs in {bindings.repo.full_name} match {query!r}."
lines = [f"# {len(results)} match(es) for {query!r} in {bindings.repo.full_name}"]
for s in results:
kind = "PR" if s.is_pull_request else "issue"
state = f"{s.state} ({s.state_reason})" if s.state_reason else s.state
labels = f" [{', '.join(s.labels)}]" if s.labels else ""
lines.append(
f"- #{s.number} ({kind}, {state}) {s.title} — @{s.author}, updated {s.updated_at[:10]}{labels}"
)
_audit(bindings, "gh_search_issues", args, result={"matches": len(results)})
return "\n".join(lines)
return host_tool(
name="gh_search_issues",
description=persona.host_tool_description("gh_search_issues"),
parameters={
"type": "object",
"properties": {
"query": {
"type": "string",
"description": persona.host_tool_parameter_description("gh_search_issues", "query"),
},
"limit": {
"type": "integer",
"description": persona.host_tool_parameter_description("gh_search_issues", "limit"),
},
},
"required": ["query"],
"additionalProperties": False,
},
execute=execute,
)
_PRIMARY_TYPES = ("bug", "enhancement", "question", "proposal", "documentation", "wontfix", "invalid", "duplicate")
_AUTO_PR_CLASSIFICATIONS = frozenset({"bug", "documentation"})
_PRIORITIES = ("prio:p0", "prio:p1", "prio:p2", "prio:p3")
@@ -1835,6 +1904,7 @@ def build(bindings: ToolBindings) -> tuple[HostTool[Any, Any], ...]:
_build_mark_unable(bindings),
_build_abort_task(bindings),
_build_fetch_thread(bindings),
_build_search_issues(bindings),
)
@@ -72,6 +72,13 @@ reason = "Internal diagnosis for the operator. Concrete, specific, blameless. NE
[fetch_issue_thread]
description = "Refetch the originating issue and its comments. Use sparingly."
[gh_search_issues]
description = "Search issues AND pull requests in the current repo (GitHub issue-search syntax; repo scope applied automatically). Use during triage to find duplicates and to check whether a merged PR already fixed the reported problem before classifying."
[gh_search_issues.parameters]
query = "GitHub issue-search syntax: bare keywords plus qualifiers like `is:pr`, `is:closed`, `is:merged`, `label:bug`, `in:title`, `author:<login>`. NEVER include a `repo:` qualifier — scope is applied automatically."
limit = "Max results, 1-20. Default 10."
[set_issue_labels]
description = "Append labels to the originating issue/PR. NEVER removes existing labels."
+3 -1
View File
@@ -16,7 +16,9 @@ Worktree is at cwd; the branch above is checked out and ready for commits **if**
the classification calls for code. Drive the todo list to completion:
1. **Triage first.** Read the body and any comments via `read` /
`fetch_issue_thread`, then call
`fetch_issue_thread`. Run `gh_search_issues` for duplicates and
already-merged fixes — the reporter may be on an older release than your
worktree. Then call
`classify_issue(primary=..., priority=..., functional=[...], rationale=...)`.
Apply the **merit gate** from the system prompt before picking `bug`:
broken contract, demonstrated impact, deliberate-tradeoff check, upstream
+8 -1
View File
@@ -21,7 +21,14 @@ Pick exactly ONE primary label per issue:
| `proposal` | Design/process proposal requiring maintainer decision. Comment with thoughts; no PR. |
| `question` | How-to, clarification, or usage question. Answer in one comment. |
| `invalid` | Spam, off-topic, or not actionable. One brief explanatory comment. |
| `duplicate` | Clear duplicate of another issue. Cite the original; no PR. |
| `duplicate` | Duplicate of another issue, or already fixed by a merged PR / newer release. Cite the original or the fixing PR; no new PR. |
## Duplicate & already-fixed check
Before `classify_issue`, run `gh_search_issues` with the report's key terms (retry with synonyms and an `is:pr` variant — one search proves nothing):
- **Prior issue on the same problem** → `duplicate`, cite it. A prior closure as not-planned/`wontfix` on the same complaint is binding precedent — adopt that verdict; NEVER relitigate it.
- **Already fixed.** Your worktree is the CURRENT default branch; reporters often run older releases. When the reported version lags the latest release (topmost released section of the relevant `packages/*/CHANGELOG.md`), check the changelog and merged PRs (`is:pr is:merged <keywords>`) for an existing fix, and try the repro against the worktree — failing on the reporter's version but passing here means it is already fixed. Classify `duplicate`: cite the fixing PR, name the release carrying it (or say it ships in the next release when still under `[Unreleased]`), and tell the reporter to update. NEVER re-fix what main already fixed.
## Merit gate — `bug` vs `wontfix` vs `enhancement`
@@ -2,6 +2,7 @@
name = "Classify"
tasks = [
"Read the issue body + every prior comment",
"gh_search_issues for duplicates and already-merged fixes",
"Call classify_issue with primary type + labels",
]
+10
View File
@@ -513,6 +513,16 @@ def create_proxy_app(settings: Settings) -> FastAPI:
return _gh_error_response(exc)
return JSONResponse({"items": [_serialize(s) for s in items]})
@app.get("/gh/v1/search_issues")
async def search_issues(request: Request, repo: str, q: str, limit: int = 10) -> JSONResponse:
await _authenticate(request)
github: GitHubClient = request.app.state.github
try:
items = await github.search_issues(repo, q, limit=limit)
except GitHubError as exc:
return _gh_error_response(exc)
return JSONResponse({"items": [_serialize(s) for s in items]})
@app.get("/gh/v1/comments")
async def list_comments(request: Request, repo: str, number: int) -> JSONResponse:
await _authenticate(request)
+10
View File
@@ -204,6 +204,14 @@ class GitHubProxyClient:
)
return [_issue_summary_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []]
async def search_issues(self, repo: str, query: str, *, limit: int = 10) -> list[IssueSummary]:
data = await self._request(
"GET",
"/gh/v1/search_issues",
params={"repo": repo, "q": query, "limit": limit},
)
return [_issue_summary_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []]
async def list_comments(self, repo: str, number: int) -> list[CommentInfo]:
data = await self._request("GET", "/gh/v1/comments", params={"repo": repo, "number": number})
return [_comment_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []]
@@ -498,6 +506,8 @@ def _issue_summary_from(data: Any) -> IssueSummary:
updated_at=str(data.get("updated_at") or ""),
created_at=str(data.get("created_at") or ""),
html_url=str(data.get("html_url") or ""),
state_reason=str(data.get("state_reason") or ""),
is_pull_request=bool(data.get("is_pull_request")),
)
+1 -1
View File
@@ -198,7 +198,7 @@ def _ensure_agent_run_dir() -> None:
return
try:
run_dir.mkdir(parents=True, exist_ok=True)
for root, dirs, files in os.walk(run_dir):
for root, _dirs, files in os.walk(run_dir):
root_path = Path(root)
os.chown(root_path, -1, gid)
root_path.chmod(0o2770)
+76
View File
@@ -861,6 +861,82 @@ def test_classify_issue_wontfix_takes_comment_only_path(db: Database, tmp_path:
assert row is not None and row.classification == "wontfix"
def test_gh_search_issues_scopes_repo_and_renders_matches(db: Database, tmp_path: Path) -> None:
"""Search auto-prefixes the repo scope, surfaces PR/state_reason so triage can
spot prior fixes and not-planned precedents, and filters the inbound issue."""
captured: dict[str, Any] = {}
def handler(request: httpx.Request) -> httpx.Response:
captured["q"] = request.url.params["q"]
return httpx.Response(
200,
json={
"total_count": 3,
"items": [
{
"number": 42, # the inbound issue itself — must be filtered
"title": "boom",
"state": "open",
"user": {"login": "alice"},
"labels": [],
"comments": 0,
"updated_at": "2026-07-01T00:00:00Z",
"created_at": "2026-07-01T00:00:00Z",
"html_url": "https://example/42",
},
{
"number": 30,
"title": "same crash on resize",
"state": "closed",
"state_reason": "not_planned",
"user": {"login": "bob"},
"labels": [{"name": "wontfix"}],
"comments": 3,
"updated_at": "2026-06-01T00:00:00Z",
"created_at": "2026-05-01T00:00:00Z",
"html_url": "https://example/30",
},
{
"number": 31,
"title": "fix: resize crash",
"state": "closed",
"state_reason": "completed",
"user": {"login": "bot"},
"labels": [],
"comments": 1,
"updated_at": "2026-06-02T00:00:00Z",
"created_at": "2026-06-02T00:00:00Z",
"html_url": "https://example/pull/31",
"pull_request": {"url": "https://example/pull/31"},
},
],
},
)
bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(handler))
try:
tool = next(x for x in build(bindings) if x.name == "gh_search_issues")
result = tool.execute({"query": "resize crash"}, _ctx())
finally:
_stop_loop(loop, t)
assert captured["q"] == "repo:octo/widget resize crash"
assert "#42" not in result # inbound issue filtered out
assert "#30 (issue, closed (not_planned))" in result
assert "#31 (PR, closed (completed))" in result
def test_gh_search_issues_rejects_repo_qualifier_and_empty_query(db: Database, tmp_path: Path) -> None:
bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(lambda r: httpx.Response(500)))
try:
tool = next(x for x in build(bindings) if x.name == "gh_search_issues")
with pytest.raises(RpcCommandError):
tool.execute({"query": "repo:evil/elsewhere secrets"}, _ctx())
with pytest.raises(RpcCommandError):
tool.execute({"query": " "}, _ctx())
finally:
_stop_loop(loop, t)
def test_classify_issue_rejects_bug_without_priority(db: Database, tmp_path: Path) -> None:
bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(lambda r: httpx.Response(500)))
try:
+27
View File
@@ -251,6 +251,29 @@ def round_trip_app(proxy_settings: Settings):
}
],
)
if path == "/search/issues" and req.method == "GET":
assert req.url.params["q"].startswith("repo:octo/widget ")
return httpx.Response(
200,
json={
"total_count": 1,
"items": [
{
"number": 9,
"title": "fixed it",
"state": "closed",
"state_reason": "completed",
"user": {"login": "bob"},
"labels": [{"name": "bug"}],
"comments": 2,
"updated_at": "2026-02-01T00:00:00Z",
"created_at": "2026-01-15T00:00:00Z",
"html_url": "https://example/9",
"pull_request": {"url": "https://example/pull/9"},
}
],
},
)
if path == "/repos/octo/widget/issues/1/comments" and req.method == "GET":
return httpx.Response(
200,
@@ -365,6 +388,10 @@ async def test_round_trip_all_endpoints(round_trip_app) -> None:
issues = await client.list_issues("octo/widget")
assert len(issues) == 1 and isinstance(issues[0], IssueSummary)
found = await client.search_issues("octo/widget", "colon selector is:pr")
assert len(found) == 1 and isinstance(found[0], IssueSummary)
assert found[0].is_pull_request and found[0].state_reason == "completed"
comments = await client.list_comments("octo/widget", 1)
assert len(comments) == 1 and isinstance(comments[0], CommentInfo)