mirror of
https://github.com/OpenHands/OpenHands.git
synced 2026-10-07 16:38:34 +08:00
Add issue duplicate automation workflow (#14444)
Co-authored-by: Engel Nyst <engel.nyst@gmail.com> Co-authored-by: openhands <openhands@all-hands.dev>
This commit is contained in:
co-authored by
Engel Nyst
openhands
parent
68083f275b
commit
e3d9abfd01
@@ -0,0 +1,473 @@
|
||||
#!/usr/bin/env python3
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from datetime import UTC, datetime, timedelta
|
||||
from typing import Any
|
||||
|
||||
GITHUB_API_BASE_URL = 'https://api.github.com'
|
||||
MAX_PAGES = 100
|
||||
DUPLICATE_CANDIDATE_LABEL = 'duplicate-candidate'
|
||||
DUPLICATE_VETO_MARKER = '<!-- openhands-duplicate-veto -->'
|
||||
AUTOMATION_BOT_LOGINS = {'all-hands-bot'}
|
||||
REPOSITORY_PATTERN = re.compile(r'^[a-zA-Z0-9_.-]+/[a-zA-Z0-9_.-]+$')
|
||||
DUPLICATE_MARKER_RE = re.compile(
|
||||
r'<!-- openhands-duplicate-check canonical=(?P<canonical>\d+) '
|
||||
r'auto-close=(?P<auto_close>true|false) -->'
|
||||
)
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description='Auto-close issues previously flagged as duplicate candidates.'
|
||||
)
|
||||
parser.add_argument('--repository', required=True)
|
||||
parser.add_argument('--close-after-days', type=int, default=3)
|
||||
parser.add_argument('--dry-run', action='store_true')
|
||||
args = parser.parse_args()
|
||||
if not REPOSITORY_PATTERN.fullmatch(args.repository):
|
||||
parser.error(f'Invalid repository format: {args.repository}')
|
||||
return args
|
||||
|
||||
|
||||
def github_headers() -> dict[str, str]:
|
||||
token = os.environ.get('GITHUB_TOKEN')
|
||||
if not token:
|
||||
raise RuntimeError('GITHUB_TOKEN environment variable is required')
|
||||
return {
|
||||
'Authorization': f'Bearer {token}',
|
||||
'Accept': 'application/vnd.github+json',
|
||||
'User-Agent': 'openhands-duplicate-auto-close',
|
||||
'X-GitHub-Api-Version': '2022-11-28',
|
||||
}
|
||||
|
||||
|
||||
def request_json(
|
||||
path: str,
|
||||
*,
|
||||
method: str = 'GET',
|
||||
body: dict[str, Any] | None = None,
|
||||
) -> Any:
|
||||
request_body = None
|
||||
headers = github_headers()
|
||||
if body is not None:
|
||||
request_body = json.dumps(body).encode('utf-8')
|
||||
headers['Content-Type'] = 'application/json'
|
||||
|
||||
request = urllib.request.Request(
|
||||
f'{GITHUB_API_BASE_URL}{path}',
|
||||
data=request_body,
|
||||
headers=headers,
|
||||
method=method,
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=60) as response:
|
||||
payload = response.read().decode('utf-8')
|
||||
except urllib.error.HTTPError as exc:
|
||||
error_body = exc.read().decode('utf-8', errors='replace')
|
||||
raise RuntimeError(
|
||||
f'{method} {path} failed with HTTP {exc.code}: {error_body}'
|
||||
) from exc
|
||||
except urllib.error.URLError as exc:
|
||||
raise RuntimeError(f'{method} {path} failed: {exc}') from exc
|
||||
|
||||
if not payload:
|
||||
return None
|
||||
try:
|
||||
return json.loads(payload)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise RuntimeError(f'Failed to parse JSON from {path}: {exc}') from exc
|
||||
|
||||
|
||||
def parse_timestamp(value: str) -> datetime:
|
||||
try:
|
||||
return datetime.fromisoformat(value.replace('Z', '+00:00'))
|
||||
except ValueError as exc:
|
||||
raise ValueError(f'Failed to parse timestamp {value!r}: {exc}') from exc
|
||||
|
||||
|
||||
def ensure_page_limit(page: int, resource_name: str) -> None:
|
||||
if page > MAX_PAGES:
|
||||
raise RuntimeError(f'Exceeded pagination limit while listing {resource_name}')
|
||||
|
||||
|
||||
def list_open_issues(repository: str) -> list[dict[str, Any]]:
|
||||
issues: list[dict[str, Any]] = []
|
||||
page = 1
|
||||
label_query = urllib.parse.quote(DUPLICATE_CANDIDATE_LABEL)
|
||||
while True:
|
||||
ensure_page_limit(page, f'open issues for {repository}')
|
||||
payload = request_json(
|
||||
f'/repos/{repository}/issues?state=open&labels={label_query}&per_page=100&page={page}'
|
||||
)
|
||||
if not isinstance(payload, list):
|
||||
raise RuntimeError(
|
||||
f'Expected list response while listing open issues for {repository}, '
|
||||
f'got {type(payload).__name__}'
|
||||
)
|
||||
if not payload:
|
||||
return issues
|
||||
for issue in payload:
|
||||
if issue.get('pull_request'):
|
||||
continue
|
||||
issues.append(issue)
|
||||
page += 1
|
||||
|
||||
|
||||
def list_issue_comments(repository: str, issue_number: int) -> list[dict[str, Any]]:
|
||||
comments: list[dict[str, Any]] = []
|
||||
page = 1
|
||||
while True:
|
||||
ensure_page_limit(page, f'comments for issue #{issue_number}')
|
||||
payload = request_json(
|
||||
f'/repos/{repository}/issues/{issue_number}/comments?per_page=100&page={page}'
|
||||
)
|
||||
if not isinstance(payload, list):
|
||||
raise RuntimeError(
|
||||
'Expected list response while listing comments for issue '
|
||||
f'#{issue_number}, got {type(payload).__name__}'
|
||||
)
|
||||
if not payload:
|
||||
return comments
|
||||
comments.extend(payload)
|
||||
page += 1
|
||||
|
||||
|
||||
def list_comment_reactions(repository: str, comment_id: int) -> list[dict[str, Any]]:
|
||||
reactions: list[dict[str, Any]] = []
|
||||
page = 1
|
||||
while True:
|
||||
ensure_page_limit(page, f'reactions for comment {comment_id}')
|
||||
payload = request_json(
|
||||
f'/repos/{repository}/issues/comments/{comment_id}/reactions?per_page=100&page={page}'
|
||||
)
|
||||
if not isinstance(payload, list):
|
||||
raise RuntimeError(
|
||||
'Expected list response while listing reactions for comment '
|
||||
f'{comment_id}, got {type(payload).__name__}'
|
||||
)
|
||||
if not payload:
|
||||
return reactions
|
||||
reactions.extend(payload)
|
||||
page += 1
|
||||
|
||||
|
||||
def extract_duplicate_metadata(comment_body: str) -> tuple[int | None, bool]:
|
||||
match = DUPLICATE_MARKER_RE.search(comment_body)
|
||||
if not match:
|
||||
return None, False
|
||||
return int(match.group('canonical')), match.group('auto_close') == 'true'
|
||||
|
||||
|
||||
def find_latest_auto_close_comment(
|
||||
comments: list[dict[str, Any]],
|
||||
) -> tuple[dict[str, Any] | None, int | None]:
|
||||
latest_comment: dict[str, Any] | None = None
|
||||
latest_canonical_issue: int | None = None
|
||||
latest_created_at: str | None = None
|
||||
for comment in comments:
|
||||
canonical_issue, auto_close = extract_duplicate_metadata(
|
||||
comment.get('body') or ''
|
||||
)
|
||||
if canonical_issue is None or not auto_close:
|
||||
continue
|
||||
comment_created_at = comment.get('created_at')
|
||||
if not isinstance(comment_created_at, str):
|
||||
comment_created_at = None
|
||||
if latest_comment is None:
|
||||
latest_comment = comment
|
||||
latest_canonical_issue = canonical_issue
|
||||
latest_created_at = comment_created_at
|
||||
continue
|
||||
if comment_created_at is None:
|
||||
continue
|
||||
if latest_created_at is not None:
|
||||
try:
|
||||
if parse_timestamp(comment_created_at) < parse_timestamp(
|
||||
latest_created_at
|
||||
):
|
||||
continue
|
||||
except ValueError:
|
||||
continue
|
||||
latest_comment = comment
|
||||
latest_canonical_issue = canonical_issue
|
||||
latest_created_at = comment_created_at
|
||||
return latest_comment, latest_canonical_issue
|
||||
|
||||
|
||||
def issue_has_label(issue: dict[str, Any], label_name: str) -> bool:
|
||||
labels = issue.get('labels') or []
|
||||
for label in labels:
|
||||
if label == label_name:
|
||||
return True
|
||||
if isinstance(label, dict) and label.get('name') == label_name:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def user_id_from_item(item: dict[str, Any]) -> int | None:
|
||||
user = item.get('user')
|
||||
if not isinstance(user, dict):
|
||||
return None
|
||||
user_id = user.get('id')
|
||||
return user_id if isinstance(user_id, int) else None
|
||||
|
||||
|
||||
def has_reaction_from_user(
|
||||
reactions: list[dict[str, Any]], user_id: int | None, content: str
|
||||
) -> bool:
|
||||
if user_id is None:
|
||||
return False
|
||||
return any(
|
||||
user_id_from_item(reaction) == user_id and reaction.get('content') == content
|
||||
for reaction in reactions
|
||||
)
|
||||
|
||||
|
||||
def has_veto_note(comments: list[dict[str, Any]]) -> bool:
|
||||
return any(
|
||||
DUPLICATE_VETO_MARKER in (comment.get('body') or '') for comment in comments
|
||||
)
|
||||
|
||||
|
||||
def is_non_bot_comment(comment: dict[str, Any]) -> bool:
|
||||
if user_id_from_item(comment) is None:
|
||||
return False
|
||||
user = comment.get('user')
|
||||
if not isinstance(user, dict):
|
||||
return False
|
||||
login = user.get('login')
|
||||
if not isinstance(login, str):
|
||||
return False
|
||||
login = login.lower()
|
||||
return (
|
||||
user.get('type') != 'Bot'
|
||||
and not login.endswith('[bot]')
|
||||
and login not in AUTOMATION_BOT_LOGINS
|
||||
)
|
||||
|
||||
|
||||
def remove_candidate_label(
|
||||
repository: str, issue_number: int, *, dry_run: bool
|
||||
) -> bool:
|
||||
if dry_run:
|
||||
return True
|
||||
try:
|
||||
request_json(
|
||||
f'/repos/{repository}/issues/{issue_number}/labels/{DUPLICATE_CANDIDATE_LABEL}',
|
||||
method='DELETE',
|
||||
)
|
||||
except RuntimeError as exc:
|
||||
if 'HTTP 404' in str(exc):
|
||||
return False
|
||||
raise
|
||||
return True
|
||||
|
||||
|
||||
def post_veto_note(repository: str, issue_number: int, *, dry_run: bool) -> bool:
|
||||
if dry_run:
|
||||
return True
|
||||
request_json(
|
||||
f'/repos/{repository}/issues/{issue_number}/comments',
|
||||
method='POST',
|
||||
body={
|
||||
'body': (
|
||||
'Thanks — leaving this open and removing the '
|
||||
f'{DUPLICATE_CANDIDATE_LABEL} label.\n\n'
|
||||
f'{DUPLICATE_VETO_MARKER}\n'
|
||||
'_This comment was created by an AI assistant '
|
||||
'(OpenHands) on behalf of the repository maintainer._'
|
||||
)
|
||||
},
|
||||
)
|
||||
return True
|
||||
|
||||
|
||||
def close_issue_as_duplicate(
|
||||
repository: str,
|
||||
issue_number: int,
|
||||
canonical_issue_number: int,
|
||||
*,
|
||||
dry_run: bool,
|
||||
) -> None:
|
||||
if dry_run:
|
||||
return
|
||||
|
||||
request_json(
|
||||
f'/repos/{repository}/issues/{issue_number}',
|
||||
method='PATCH',
|
||||
body={'state': 'closed', 'state_reason': 'duplicate'},
|
||||
)
|
||||
request_json(
|
||||
f'/repos/{repository}/issues/{issue_number}/comments',
|
||||
method='POST',
|
||||
body={
|
||||
'body': (
|
||||
'This issue has been automatically closed as a duplicate of '
|
||||
f'#{canonical_issue_number}.\n\n'
|
||||
'If this is incorrect, please add a comment and it can be '
|
||||
'reopened.\n\n'
|
||||
'_This comment was created by an AI assistant '
|
||||
'(OpenHands) on behalf of the repository maintainer._'
|
||||
)
|
||||
},
|
||||
)
|
||||
remove_candidate_label(repository, issue_number, dry_run=False)
|
||||
|
||||
|
||||
def keep_open_due_to_newer_comments(
|
||||
repository: str,
|
||||
issue: dict[str, Any],
|
||||
issue_number: int,
|
||||
*,
|
||||
dry_run: bool,
|
||||
) -> dict[str, Any]:
|
||||
label_removed = False
|
||||
if issue_has_label(issue, DUPLICATE_CANDIDATE_LABEL):
|
||||
label_removed = remove_candidate_label(
|
||||
repository,
|
||||
issue_number,
|
||||
dry_run=dry_run,
|
||||
)
|
||||
return {
|
||||
'issue_number': issue_number,
|
||||
'action': 'kept-open',
|
||||
'reason': 'newer-comment-after-duplicate-notice',
|
||||
'label_removed': label_removed,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
now = datetime.now(UTC)
|
||||
cutoff = now - timedelta(days=args.close_after_days)
|
||||
|
||||
summary: list[dict[str, Any]] = []
|
||||
for issue in list_open_issues(args.repository):
|
||||
issue_number = issue.get('number')
|
||||
if issue_number is None:
|
||||
continue
|
||||
try:
|
||||
issue_number = int(issue_number)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
|
||||
comments = list_issue_comments(args.repository, issue_number)
|
||||
latest_comment, canonical_issue_number = find_latest_auto_close_comment(
|
||||
comments
|
||||
)
|
||||
if latest_comment is None or canonical_issue_number is None:
|
||||
continue
|
||||
|
||||
comment_created_at_str = latest_comment.get('created_at')
|
||||
comment_id = latest_comment.get('id')
|
||||
if not comment_created_at_str or comment_id is None:
|
||||
continue
|
||||
try:
|
||||
comment_id = int(comment_id)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
try:
|
||||
comment_created_at = parse_timestamp(comment_created_at_str)
|
||||
except ValueError as exc:
|
||||
print(
|
||||
'Warning: Skipping issue '
|
||||
f'#{issue_number} due to invalid duplicate-comment timestamp: {exc}',
|
||||
file=sys.stderr,
|
||||
)
|
||||
continue
|
||||
if comment_created_at > cutoff:
|
||||
continue
|
||||
|
||||
author_id = user_id_from_item(issue)
|
||||
reactions = list_comment_reactions(args.repository, comment_id)
|
||||
author_thumbs_down = has_reaction_from_user(reactions, author_id, '-1')
|
||||
author_thumbs_up = has_reaction_from_user(reactions, author_id, '+1')
|
||||
if author_thumbs_down:
|
||||
label_removed = False
|
||||
if issue_has_label(issue, DUPLICATE_CANDIDATE_LABEL):
|
||||
label_removed = remove_candidate_label(
|
||||
args.repository,
|
||||
issue_number,
|
||||
dry_run=args.dry_run,
|
||||
)
|
||||
veto_note_posted = False
|
||||
if not has_veto_note(comments):
|
||||
veto_note_posted = post_veto_note(
|
||||
args.repository,
|
||||
issue_number,
|
||||
dry_run=args.dry_run,
|
||||
)
|
||||
summary.append(
|
||||
{
|
||||
'issue_number': issue_number,
|
||||
'action': 'kept-open',
|
||||
'reason': 'author-thumbed-down-duplicate-comment',
|
||||
'label_removed': label_removed,
|
||||
'veto_note_posted': veto_note_posted,
|
||||
'author_thumbs_up': author_thumbs_up,
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
newer_comments = []
|
||||
for comment in comments:
|
||||
created_at = comment.get('created_at')
|
||||
if not created_at or not is_non_bot_comment(comment):
|
||||
continue
|
||||
try:
|
||||
newer_comment_created_at = parse_timestamp(created_at)
|
||||
except ValueError as exc:
|
||||
print(
|
||||
'Warning: Ignoring newer comment with invalid timestamp on '
|
||||
f'issue #{issue_number}: {exc}',
|
||||
file=sys.stderr,
|
||||
)
|
||||
continue
|
||||
if newer_comment_created_at > comment_created_at:
|
||||
newer_comments.append(comment)
|
||||
if newer_comments:
|
||||
summary.append(
|
||||
keep_open_due_to_newer_comments(
|
||||
args.repository,
|
||||
issue,
|
||||
issue_number,
|
||||
dry_run=args.dry_run,
|
||||
)
|
||||
)
|
||||
continue
|
||||
|
||||
close_issue_as_duplicate(
|
||||
args.repository,
|
||||
issue_number,
|
||||
canonical_issue_number,
|
||||
dry_run=args.dry_run,
|
||||
)
|
||||
summary.append(
|
||||
{
|
||||
'issue_number': issue_number,
|
||||
'action': 'closed-as-duplicate'
|
||||
if not args.dry_run
|
||||
else 'would-close-as-duplicate',
|
||||
'canonical_issue_number': canonical_issue_number,
|
||||
'author_thumbs_up': author_thumbs_up,
|
||||
}
|
||||
)
|
||||
|
||||
print(json.dumps({'repository': args.repository, 'results': summary}, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
raise SystemExit(main())
|
||||
except Exception as exc: # noqa: BLE001
|
||||
print(f'error: {exc}', file=sys.stderr)
|
||||
raise
|
||||
@@ -0,0 +1,628 @@
|
||||
#!/usr/bin/env python3
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
OPENHANDS_BASE_URL = os.environ.get('OPENHANDS_BASE_URL', 'https://app.all-hands.dev')
|
||||
REPOSITORY_PATTERN = re.compile(r'^[a-zA-Z0-9_.-]+/[a-zA-Z0-9_.-]+$')
|
||||
GITHUB_API_BASE_URL = os.environ.get('GITHUB_API_BASE_URL', 'https://api.github.com')
|
||||
FAILED_EXECUTION_STATUSES = {
|
||||
'error',
|
||||
'errored',
|
||||
'failed',
|
||||
'stopped',
|
||||
}
|
||||
SUCCESSFUL_TERMINAL_EXECUTION_STATUSES = {
|
||||
'completed',
|
||||
'finished',
|
||||
}
|
||||
TERMINAL_EXECUTION_STATUSES = (
|
||||
FAILED_EXECUTION_STATUSES | SUCCESSFUL_TERMINAL_EXECUTION_STATUSES
|
||||
)
|
||||
EVENT_SEARCH_LIMIT = 1000
|
||||
EVENT_SEARCH_LIMIT_HIT_MESSAGE = (
|
||||
f'Event search returned at least {EVENT_SEARCH_LIMIT} events; results may be '
|
||||
'incomplete'
|
||||
)
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description=(
|
||||
'Start an OpenHands Cloud conversation that checks a GitHub issue '
|
||||
'for duplicates.'
|
||||
)
|
||||
)
|
||||
parser.add_argument(
|
||||
'--repository', required=True, help='Repository in owner/repo form'
|
||||
)
|
||||
parser.add_argument(
|
||||
'--issue-number', required=True, type=int, help='Issue number to inspect'
|
||||
)
|
||||
parser.add_argument(
|
||||
'--output',
|
||||
default='duplicate-check-result.json',
|
||||
help='Path where the JSON result should be written',
|
||||
)
|
||||
parser.add_argument(
|
||||
'--poll-interval-seconds',
|
||||
default=5,
|
||||
type=int,
|
||||
help='Polling interval while waiting for the conversation to finish',
|
||||
)
|
||||
parser.add_argument(
|
||||
'--max-wait-seconds',
|
||||
default=900,
|
||||
type=int,
|
||||
help=(
|
||||
'Maximum time to wait per polling phase; if a start task must be awaited '
|
||||
'first, the total runtime can approach twice this value'
|
||||
),
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def github_headers() -> dict[str, str]:
|
||||
headers = {
|
||||
'Accept': 'application/vnd.github+json',
|
||||
'User-Agent': 'openhands-issue-duplicate-check',
|
||||
'X-GitHub-Api-Version': '2022-11-28',
|
||||
}
|
||||
github_token = os.environ.get('GITHUB_TOKEN')
|
||||
if github_token:
|
||||
headers['Authorization'] = f'Bearer {github_token}'
|
||||
return headers
|
||||
|
||||
|
||||
def openhands_headers() -> dict[str, str]:
|
||||
api_key = os.environ.get('OPENHANDS_API_KEY')
|
||||
if not api_key:
|
||||
raise RuntimeError('OPENHANDS_API_KEY environment variable is required')
|
||||
return {
|
||||
'Authorization': f'Bearer {api_key}',
|
||||
'Content-Type': 'application/json',
|
||||
}
|
||||
|
||||
|
||||
def request_json(
|
||||
base_url: str,
|
||||
path: str,
|
||||
*,
|
||||
method: str = 'GET',
|
||||
headers: dict[str, str] | None = None,
|
||||
body: dict[str, Any] | None = None,
|
||||
) -> Any:
|
||||
data = json.dumps(body).encode('utf-8') if body is not None else None
|
||||
request = urllib.request.Request(
|
||||
f'{base_url}{path}',
|
||||
data=data,
|
||||
headers=headers or {},
|
||||
method=method,
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=60) as response:
|
||||
return json.load(response)
|
||||
except urllib.error.HTTPError as exc:
|
||||
error_body = exc.read().decode('utf-8', errors='replace')
|
||||
raise RuntimeError(
|
||||
f'{method} {base_url}{path} failed with HTTP {exc.code}: {error_body}'
|
||||
) from exc
|
||||
except json.JSONDecodeError as exc:
|
||||
raise RuntimeError(
|
||||
f'Failed to parse JSON from {method} {base_url}{path}: {exc}'
|
||||
) from exc
|
||||
except urllib.error.URLError as exc:
|
||||
raise RuntimeError(f'{method} {base_url}{path} failed: {exc}') from exc
|
||||
|
||||
|
||||
def fetch_issue(repository: str, issue_number: int) -> dict[str, Any]:
|
||||
if not REPOSITORY_PATTERN.fullmatch(repository):
|
||||
raise ValueError(f'Invalid repository format: {repository}')
|
||||
return request_json(
|
||||
GITHUB_API_BASE_URL,
|
||||
f'/repos/{repository}/issues/{issue_number}',
|
||||
headers=github_headers(),
|
||||
)
|
||||
|
||||
|
||||
def escape_json_text(value: str | None) -> str:
|
||||
return json.dumps(value or '', ensure_ascii=False)
|
||||
|
||||
|
||||
def build_prompt(repository: str, issue: dict[str, Any]) -> str:
|
||||
issue_number = issue['number']
|
||||
issue_title = issue.get('title', '')
|
||||
issue_body = issue.get('body') or ''
|
||||
issue_url = issue.get('html_url', '')
|
||||
issue_title_json = escape_json_text(issue_title)
|
||||
issue_body_json = escape_json_text(issue_body)
|
||||
|
||||
return '\n'.join(
|
||||
[
|
||||
'You are investigating whether a GitHub issue should be redirected '
|
||||
'to an existing issue because it is either:',
|
||||
'- an exact or near-exact duplicate, or',
|
||||
'- so overlapping in scope that discussion or fix planning would '
|
||||
'likely be better kept in one canonical issue.',
|
||||
'',
|
||||
'Be conservative about auto-close decisions, but do investigate '
|
||||
'seriously before deciding.',
|
||||
'',
|
||||
f'Repository: {repository}',
|
||||
f'New issue number: #{issue_number}',
|
||||
f'New issue URL: {issue_url}',
|
||||
f'New issue title (JSON-escaped string): {issue_title_json}',
|
||||
f'New issue body (JSON-escaped string): {issue_body_json}',
|
||||
'',
|
||||
'Task:',
|
||||
'1. Understand the core problem, user-facing outcome, likely root '
|
||||
'cause, and requested fix or behavior.',
|
||||
"2. Investigate this repository's open issues and issues closed "
|
||||
'in the last 90 days for exact duplicates, near-duplicates, or '
|
||||
'strong scope overlap.',
|
||||
'3. Use multiple search approaches with diverse keywords and '
|
||||
'phrasings rather than a single literal search.',
|
||||
'4. Ignore pull requests.',
|
||||
'5. Distinguish carefully between:',
|
||||
' - duplicate: essentially the same report, request, or root cause',
|
||||
' - overlapping-scope: not identical, but likely to fragment '
|
||||
'discussion or produce competing fixes',
|
||||
' - related-but-distinct: similar area, but should stay separate',
|
||||
' - no-match: no strong candidate worth redirecting to',
|
||||
'6. Inspect the strongest 1-3 candidates carefully. If needed, '
|
||||
'inspect comments on the strongest candidates to disambiguate '
|
||||
'false positives.',
|
||||
'7. Do not post comments, do not modify files, and do not change '
|
||||
'repository state.',
|
||||
'8. Useful API shapes include:',
|
||||
f' - GET https://api.github.com/repos/{repository}/issues?state=open&per_page=100',
|
||||
' - GET https://api.github.com/repos/'
|
||||
f'{repository}/issues?state=closed&since=<ISO-8601 timestamp>&per_page=100',
|
||||
' - GET https://api.github.com/search/issues?q=<query>',
|
||||
f' - GET https://api.github.com/repos/{repository}/issues/<number>/comments',
|
||||
'9. Return exactly one JSON object and nothing else. Do not wrap '
|
||||
'it in markdown fences.',
|
||||
'',
|
||||
'Return schema:',
|
||||
'{',
|
||||
f' "issue_number": {issue_number},',
|
||||
' "should_comment": true or false,',
|
||||
' "is_duplicate": true or false,',
|
||||
' "auto_close_candidate": true or false,',
|
||||
' "classification": "duplicate" | "overlapping-scope" | '
|
||||
'"related-but-distinct" | "no-match",',
|
||||
' "confidence": "high" | "medium" | "low",',
|
||||
' "summary": "short explanation",',
|
||||
' "canonical_issue_number": 123 or null,',
|
||||
' "candidate_issues": [',
|
||||
' {',
|
||||
' "number": 123,',
|
||||
f' "url": "https://github.com/{repository}/issues/123",',
|
||||
' "title": "issue title",',
|
||||
' "state": "open or closed",',
|
||||
' "closed_at": "ISO timestamp or null",',
|
||||
' "similarity_reason": "why it looks similar"',
|
||||
' }',
|
||||
' ]',
|
||||
'}',
|
||||
'',
|
||||
'Rules:',
|
||||
'- `should_comment` should be true only when redirecting the '
|
||||
'author would likely help.',
|
||||
'- `is_duplicate` should be true only for exact or near-exact duplicates.',
|
||||
'- `auto_close_candidate` should be true only when:',
|
||||
' - classification is `duplicate`',
|
||||
' - confidence is `high`',
|
||||
' - one canonical issue clearly stands out',
|
||||
' - a maintainer would likely be comfortable closing this issue '
|
||||
'after a waiting period',
|
||||
'- For `overlapping-scope`, `auto_close_candidate` must be false.',
|
||||
'- `candidate_issues` must contain at most 3 issues, sorted best-first.',
|
||||
'- If no strong match exists, return `should_comment: false`, '
|
||||
'`classification: "no-match"`, `canonical_issue_number: null`, '
|
||||
'and an empty candidate list.',
|
||||
'- Be especially careful not to collapse broad meta, tracking, '
|
||||
'feedback, or umbrella issues with specific bug reports unless '
|
||||
'the new issue clearly belongs in that exact thread.',
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
def start_conversation(
|
||||
prompt: str, repository: str, issue_number: int
|
||||
) -> dict[str, Any]:
|
||||
body = {
|
||||
'title': f'Issue duplicate check #{issue_number}',
|
||||
'selected_repository': repository,
|
||||
'initial_message': {
|
||||
'content': [
|
||||
{
|
||||
'type': 'text',
|
||||
'text': prompt,
|
||||
}
|
||||
]
|
||||
},
|
||||
}
|
||||
return request_json(
|
||||
OPENHANDS_BASE_URL,
|
||||
'/api/v1/app-conversations',
|
||||
method='POST',
|
||||
headers=openhands_headers(),
|
||||
body=body,
|
||||
)
|
||||
|
||||
|
||||
def extract_first_item(payload: Any) -> dict[str, Any] | None:
|
||||
if isinstance(payload, list):
|
||||
first_item = payload[0] if payload else None
|
||||
return first_item if isinstance(first_item, dict) else None
|
||||
if not isinstance(payload, dict):
|
||||
return None
|
||||
|
||||
items = payload.get('items')
|
||||
if isinstance(items, list):
|
||||
first_item = items[0] if items else None
|
||||
return first_item if isinstance(first_item, dict) else None
|
||||
return payload
|
||||
|
||||
|
||||
def poll_start_task(
|
||||
start_task_id: str, poll_interval_seconds: int, max_wait_seconds: int
|
||||
) -> dict[str, Any]:
|
||||
deadline = time.time() + max_wait_seconds
|
||||
while time.time() < deadline:
|
||||
payload = request_json(
|
||||
OPENHANDS_BASE_URL,
|
||||
f'/api/v1/app-conversations/start-tasks?ids={urllib.parse.quote(start_task_id)}',
|
||||
headers={'Authorization': openhands_headers()['Authorization']},
|
||||
)
|
||||
item = extract_first_item(payload)
|
||||
if item is None:
|
||||
time.sleep(poll_interval_seconds)
|
||||
continue
|
||||
status = item.get('status')
|
||||
if status == 'READY' and item.get('app_conversation_id'):
|
||||
return item
|
||||
if status in {'ERROR', 'FAILED'}:
|
||||
raise RuntimeError(f'OpenHands start task failed: {json.dumps(item)}')
|
||||
time.sleep(poll_interval_seconds)
|
||||
raise TimeoutError(
|
||||
f'Timed out waiting for start task {start_task_id} to become ready'
|
||||
)
|
||||
|
||||
|
||||
def poll_conversation(
|
||||
app_conversation_id: str, poll_interval_seconds: int, max_wait_seconds: int
|
||||
) -> dict[str, Any]:
|
||||
deadline = time.time() + max_wait_seconds
|
||||
while time.time() < deadline:
|
||||
payload = request_json(
|
||||
OPENHANDS_BASE_URL,
|
||||
f'/api/v1/app-conversations?ids={urllib.parse.quote(app_conversation_id)}',
|
||||
headers={'Authorization': openhands_headers()['Authorization']},
|
||||
)
|
||||
item = extract_first_item(payload)
|
||||
if item is None:
|
||||
time.sleep(poll_interval_seconds)
|
||||
continue
|
||||
execution_status = str(item.get('execution_status', '')).lower()
|
||||
if execution_status in FAILED_EXECUTION_STATUSES:
|
||||
raise RuntimeError(
|
||||
'OpenHands conversation ended with '
|
||||
f'{execution_status}: {json.dumps(item)}'
|
||||
)
|
||||
if execution_status in SUCCESSFUL_TERMINAL_EXECUTION_STATUSES:
|
||||
return item
|
||||
time.sleep(poll_interval_seconds)
|
||||
raise TimeoutError(
|
||||
f'Timed out waiting for conversation {app_conversation_id} to finish running'
|
||||
)
|
||||
|
||||
|
||||
def validate_event_search_results(events: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
if len(events) >= EVENT_SEARCH_LIMIT:
|
||||
raise RuntimeError(EVENT_SEARCH_LIMIT_HIT_MESSAGE)
|
||||
return events
|
||||
|
||||
|
||||
def fetch_app_server_events(app_conversation_id: str) -> list[dict[str, Any]]:
|
||||
payload = request_json(
|
||||
OPENHANDS_BASE_URL,
|
||||
f'/api/v1/conversation/{urllib.parse.quote(app_conversation_id)}/events/search?limit={EVENT_SEARCH_LIMIT}',
|
||||
headers={'Authorization': openhands_headers()['Authorization']},
|
||||
)
|
||||
if isinstance(payload, dict):
|
||||
items = payload.get('items')
|
||||
return validate_event_search_results(items) if isinstance(items, list) else []
|
||||
if isinstance(payload, list):
|
||||
return validate_event_search_results(payload)
|
||||
return []
|
||||
|
||||
|
||||
def fetch_agent_server_events(
|
||||
app_conversation_id: str, agent_server_url: str, session_api_key: str
|
||||
) -> list[dict[str, Any]]:
|
||||
payload = request_json(
|
||||
agent_server_url,
|
||||
f'/api/conversations/{urllib.parse.quote(app_conversation_id)}/events/search?limit={EVENT_SEARCH_LIMIT}',
|
||||
headers={'X-Session-API-Key': session_api_key},
|
||||
)
|
||||
if isinstance(payload, dict):
|
||||
items = payload.get('items')
|
||||
return validate_event_search_results(items) if isinstance(items, list) else []
|
||||
if isinstance(payload, list):
|
||||
return validate_event_search_results(payload)
|
||||
return []
|
||||
|
||||
|
||||
def fetch_agent_server_final_response(
|
||||
app_conversation_id: str, agent_server_url: str, session_api_key: str
|
||||
) -> str:
|
||||
payload = request_json(
|
||||
agent_server_url,
|
||||
f'/api/conversations/{urllib.parse.quote(app_conversation_id)}/agent_final_response',
|
||||
headers={'X-Session-API-Key': session_api_key},
|
||||
)
|
||||
if not isinstance(payload, dict):
|
||||
return ''
|
||||
return str(payload.get('response') or '').strip()
|
||||
|
||||
|
||||
def extract_agent_server_url(conversation_url: str) -> str | None:
|
||||
marker = '/api/conversations/'
|
||||
if marker not in conversation_url:
|
||||
return None
|
||||
return conversation_url.rsplit(marker, 1)[0]
|
||||
|
||||
|
||||
def extract_last_agent_text(events: list[dict[str, Any]]) -> str:
|
||||
agent_events = [
|
||||
event
|
||||
for event in events
|
||||
if event.get('kind') == 'MessageEvent' and event.get('source') == 'agent'
|
||||
]
|
||||
if not agent_events:
|
||||
raise RuntimeError(
|
||||
'No assistant text message was found in the conversation events'
|
||||
)
|
||||
|
||||
llm_message = agent_events[-1].get('llm_message')
|
||||
if not isinstance(llm_message, dict):
|
||||
raise RuntimeError('Last agent message has no llm_message field')
|
||||
content = llm_message.get('content')
|
||||
if not isinstance(content, list):
|
||||
raise RuntimeError('Last agent message content is not a list')
|
||||
|
||||
text_parts: list[str] = []
|
||||
for part in content:
|
||||
if not isinstance(part, dict):
|
||||
continue
|
||||
if part.get('type') == 'text' and part.get('text'):
|
||||
text_parts.append(str(part['text']))
|
||||
if not text_parts:
|
||||
raise RuntimeError('Last agent message contains no text content')
|
||||
return ''.join(text_parts).strip()
|
||||
|
||||
|
||||
def parse_agent_json(text: str) -> dict[str, Any]:
|
||||
cleaned = text.strip()
|
||||
try:
|
||||
return json.loads(cleaned)
|
||||
except json.JSONDecodeError:
|
||||
decoder = json.JSONDecoder()
|
||||
for start, character in enumerate(cleaned):
|
||||
if character != '{':
|
||||
continue
|
||||
try:
|
||||
candidate, end = decoder.raw_decode(cleaned[start:])
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
trailing = cleaned[start + end :].strip()
|
||||
if trailing not in {'', '```'}:
|
||||
continue
|
||||
if isinstance(candidate, dict):
|
||||
return candidate
|
||||
raise ValueError('No valid JSON object found in the agent response')
|
||||
|
||||
|
||||
def as_bool(value: Any) -> bool:
|
||||
if isinstance(value, bool):
|
||||
return value
|
||||
if isinstance(value, str):
|
||||
return value.strip().lower() in {'true', '1', 'yes'}
|
||||
if isinstance(value, (int, float)):
|
||||
return bool(value)
|
||||
return False
|
||||
|
||||
|
||||
def normalize_result(result: dict[str, Any]) -> dict[str, Any]:
|
||||
normalized = dict(result)
|
||||
normalized['should_comment'] = as_bool(normalized.get('should_comment'))
|
||||
normalized['is_duplicate'] = as_bool(normalized.get('is_duplicate'))
|
||||
normalized['auto_close_candidate'] = as_bool(normalized.get('auto_close_candidate'))
|
||||
|
||||
classification = str(normalized.get('classification') or 'no-match').strip().lower()
|
||||
if classification not in {
|
||||
'duplicate',
|
||||
'overlapping-scope',
|
||||
'related-but-distinct',
|
||||
'no-match',
|
||||
}:
|
||||
classification = 'no-match'
|
||||
normalized['classification'] = classification
|
||||
|
||||
confidence = str(normalized.get('confidence') or 'low').strip().lower()
|
||||
if confidence not in {'high', 'medium', 'low'}:
|
||||
confidence = 'low'
|
||||
normalized['confidence'] = confidence
|
||||
|
||||
try:
|
||||
canonical_issue_number = normalized.get('canonical_issue_number')
|
||||
if canonical_issue_number in {None, ''}:
|
||||
normalized['canonical_issue_number'] = None
|
||||
else:
|
||||
normalized['canonical_issue_number'] = int(str(canonical_issue_number))
|
||||
except (TypeError, ValueError):
|
||||
normalized['canonical_issue_number'] = None
|
||||
|
||||
candidate_issues = normalized.get('candidate_issues')
|
||||
if not isinstance(candidate_issues, list):
|
||||
candidate_issues = []
|
||||
normalized['candidate_issues'] = candidate_issues[:3]
|
||||
|
||||
if classification not in {'duplicate', 'overlapping-scope'}:
|
||||
normalized['should_comment'] = False
|
||||
if classification != 'duplicate':
|
||||
normalized['is_duplicate'] = False
|
||||
normalized['auto_close_candidate'] = False
|
||||
if (
|
||||
classification in {'duplicate', 'overlapping-scope'}
|
||||
and normalized['candidate_issues']
|
||||
and confidence in {'high', 'medium'}
|
||||
):
|
||||
normalized['should_comment'] = True
|
||||
if normalized['auto_close_candidate'] and confidence != 'high':
|
||||
normalized['auto_close_candidate'] = False
|
||||
if normalized['auto_close_candidate'] and not normalized['candidate_issues']:
|
||||
normalized['auto_close_candidate'] = False
|
||||
if (
|
||||
normalized['auto_close_candidate']
|
||||
and normalized['canonical_issue_number'] is None
|
||||
):
|
||||
first_candidate = (
|
||||
normalized['candidate_issues'][0] if normalized['candidate_issues'] else {}
|
||||
)
|
||||
candidate_number = first_candidate.get('number')
|
||||
try:
|
||||
if candidate_number is None:
|
||||
raise ValueError('candidate number is missing')
|
||||
normalized['canonical_issue_number'] = int(str(candidate_number))
|
||||
except (TypeError, ValueError, AttributeError):
|
||||
normalized['auto_close_candidate'] = False
|
||||
|
||||
normalized['summary'] = str(normalized.get('summary') or '').strip()
|
||||
return normalized
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
issue = fetch_issue(args.repository, args.issue_number)
|
||||
if issue.get('pull_request'):
|
||||
raise RuntimeError(f'#{args.issue_number} is a pull request, not an issue')
|
||||
|
||||
prompt = build_prompt(args.repository, issue)
|
||||
start_task = start_conversation(prompt, args.repository, args.issue_number)
|
||||
app_conversation_id = start_task.get('app_conversation_id')
|
||||
conversation_url = ''
|
||||
|
||||
if not app_conversation_id:
|
||||
task_id = start_task.get('id')
|
||||
if not task_id:
|
||||
raise RuntimeError(f'Missing id in start task response: {start_task}')
|
||||
ready_task = poll_start_task(
|
||||
task_id,
|
||||
args.poll_interval_seconds,
|
||||
args.max_wait_seconds,
|
||||
)
|
||||
app_conversation_id = ready_task.get('app_conversation_id')
|
||||
if not app_conversation_id:
|
||||
raise RuntimeError(f'Missing app_conversation_id in response: {ready_task}')
|
||||
|
||||
conversation = poll_conversation(
|
||||
app_conversation_id,
|
||||
args.poll_interval_seconds,
|
||||
args.max_wait_seconds,
|
||||
)
|
||||
conversation_url = (
|
||||
conversation.get('conversation_url')
|
||||
or f'{OPENHANDS_BASE_URL}/conversations/{app_conversation_id}'
|
||||
)
|
||||
session_api_key_value = conversation.get('session_api_key')
|
||||
if session_api_key_value and not isinstance(session_api_key_value, str):
|
||||
raise RuntimeError(
|
||||
'session_api_key had unexpected type in the OpenHands conversation: '
|
||||
f'{type(session_api_key_value).__name__}'
|
||||
)
|
||||
session_api_key = session_api_key_value or ''
|
||||
agent_server_url = extract_agent_server_url(conversation_url)
|
||||
|
||||
agent_text = ''
|
||||
if agent_server_url and session_api_key:
|
||||
try:
|
||||
agent_text = fetch_agent_server_final_response(
|
||||
app_conversation_id,
|
||||
agent_server_url,
|
||||
session_api_key,
|
||||
)
|
||||
except RuntimeError:
|
||||
agent_text = ''
|
||||
if not agent_text:
|
||||
events = fetch_app_server_events(app_conversation_id)
|
||||
try:
|
||||
agent_text = extract_last_agent_text(events)
|
||||
except RuntimeError as exc:
|
||||
if not session_api_key:
|
||||
raise RuntimeError(
|
||||
'App server events did not contain assistant text and '
|
||||
'session_api_key was missing from the OpenHands conversation'
|
||||
) from exc
|
||||
if not agent_server_url:
|
||||
raise RuntimeError(
|
||||
'App server events did not contain assistant text and cannot '
|
||||
'extract agent server URL from conversation URL: '
|
||||
f'{conversation_url}'
|
||||
) from exc
|
||||
events = fetch_agent_server_events(
|
||||
app_conversation_id,
|
||||
agent_server_url,
|
||||
session_api_key,
|
||||
)
|
||||
agent_text = extract_last_agent_text(events)
|
||||
result = normalize_result(parse_agent_json(agent_text))
|
||||
|
||||
result['issue_number'] = args.issue_number
|
||||
result['repository'] = args.repository
|
||||
result['app_conversation_id'] = app_conversation_id
|
||||
result['conversation_url'] = conversation_url
|
||||
result['agent_response'] = agent_text
|
||||
|
||||
output_path = Path(args.output)
|
||||
try:
|
||||
output_path.write_text(json.dumps(result, indent=2, ensure_ascii=False) + '\n')
|
||||
except OSError as exc:
|
||||
raise RuntimeError(f'Failed to write output to {output_path}: {exc}') from exc
|
||||
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
'issue_number': result.get('issue_number'),
|
||||
'should_comment': result.get('should_comment'),
|
||||
'is_duplicate': result.get('is_duplicate'),
|
||||
'auto_close_candidate': result.get('auto_close_candidate'),
|
||||
'classification': result.get('classification'),
|
||||
'confidence': result.get('confidence'),
|
||||
'conversation_url': result.get('conversation_url'),
|
||||
'output': str(output_path),
|
||||
},
|
||||
ensure_ascii=False,
|
||||
)
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
try:
|
||||
raise SystemExit(main())
|
||||
except Exception as exc: # noqa: BLE001
|
||||
print(f'error: {exc}', file=sys.stderr)
|
||||
raise
|
||||
Reference in New Issue
Block a user