Repository navigation
Expand file tree
/
Copy pathimport_github_bugfixes.py
More file actions
231 lines (200 loc) · 9.2 KB
/
Copy pathimport_github_bugfixes.py
File metadata and controls
231 lines (200 loc) · 9.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
"""Real-data importer for the CodeReview task battery — v2.
DATA SOURCE
Real bug-fix commits from popular open-source Python repos. The PRE-fix
version of the changed function is the buggy `code` example — real code
broken by real developers, not a model.
GOLD LABEL — the hard part, done honestly
CodeReview uses a CLOSED 5-class vocabulary (code_review.py BUG_TYPES:
logic_error, null_pointer, off_by_one, race_condition, resource_leak),
exact-match scored.
Conventional-commit prefixes ("fix:") and GitHub issue labels reliably
say "this is a bug fix" but NOT which of the 5 categories it is — issue
labels are almost never that fine-grained. So the category is derived
by inspecting the DIFF with rule-based patterns (e.g. a fix that adds a
`with`/`.close()` -> resource_leak; adds an `is None` guard ->
null_pointer; flips `<=`/`<` or tweaks an index by one -> off_by_one;
adds a lock -> race_condition). This is rule-based, inspectable, and
non-circular — but still a heuristic.
Every entry carries a `label_source`:
* diff-pattern — a specific code-change pattern matched (stronger)
* commit-keyword — only a commit-message keyword matched (weak)
Commits where neither resolves a category are SKIPPED. Review every
label regardless — confirm it against the diff at the commit URL.
DIFFICULTY
Heuristic by function length. Re-tier by hand — difficulty for
CodeReview is really bug subtlety, which the script cannot judge.
LIMITATIONS
* Only single-file, small-diff commits are kept (to isolate one
function) — low hit rate; run across many repos to reach 70.
* Needs GITHUB_TOKEN for volume (unauthenticated API is 60 req/hr).
OUTPUT
scripts/staging/code_review_candidates.py
USAGE
GITHUB_TOKEN=ghp_xxx python scripts/import_github_bugfixes.py [--pages N]
"""
import argparse
import os
import re
import sys
from datetime import date
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from _importer_common import ( # noqa: E402
GHError, gh_get, prefix_function_from_commit, tier, write_staging,
)
REPOS = [
# Large, actively-maintained repos produce the most small bug-fix
# commits. Yield per repo is low — single-file, small-diff fixes are
# genuinely rare — so the list is broad on purpose.
"psf/requests", "pallets/flask", "pallets/click", "tornadoweb/tornado",
"scrapy/scrapy", "sqlalchemy/sqlalchemy", "python-pillow/Pillow",
"boto/boto3", "psf/black", "encode/httpx", "aio-libs/aiohttp",
"pydantic/pydantic", "django/django", "pandas-dev/pandas",
"scikit-learn/scikit-learn", "numpy/numpy", "apache/airflow",
"celery/celery", "pytest-dev/pytest", "python/mypy", "ansible/ansible",
"home-assistant/core", "getsentry/sentry-python", "encode/django-rest-framework",
"sphinx-doc/sphinx", "matplotlib/matplotlib", "pypa/pip", "huggingface/transformers",
]
TARGET_TOTAL = 80
_CONV_FIX_RE = re.compile(r"^fix(\([^)]+\))?:", re.IGNORECASE)
_BUG_RE = re.compile(r"\b(bug|fix|fixes|fixed|incorrect|wrong|crash|"
r"regression|broken)\b", re.IGNORECASE)
# Commit-message keyword fallback -> BUG_TYPES (weak signal).
LABEL_KEYWORDS = [
("off_by_one", ["off-by-one", "off by one", "index out of range",
"indexerror", "boundary", "fencepost"]),
("null_pointer", ["nonetype", "attributeerror", "none check",
"null check", "missing check", "keyerror"]),
("race_condition", ["race condition", "deadlock", "concurrenc",
"thread-safe", "thread safety", "atomic"]),
("resource_leak", ["leak", "unclosed", "not closed", "file handle",
"fd leak"]),
("logic_error", ["incorrect", "wrong", "logic error", "typo"]),
]
_HERE = os.path.dirname(os.path.abspath(__file__))
def _added_removed(patch):
add, rem = [], []
for line in patch.splitlines():
if line.startswith("+") and not line.startswith("+++"):
add.append(line[1:])
elif line.startswith("-") and not line.startswith("---"):
rem.append(line[1:])
return "\n".join(add), "\n".join(rem)
def classify_by_diff(patch):
"""Rule-based bug-category from the diff. Returns label or None."""
added, removed = _added_removed(patch)
al, rl = added.lower(), removed.lower()
# resource_leak: the fix introduces a context manager / close / finally.
if (("with " in added and "with " not in removed)
or (".close()" in added and ".close()" not in removed)
or ("finally:" in al and "finally:" not in rl)):
return "resource_leak"
# race_condition: the fix introduces locking / concurrency guards.
lock_kw = ("lock(", ".acquire(", "rlock(", "semaphore(", "threading.")
if any(k in al for k in lock_kw) and not any(k in rl for k in lock_kw):
return "race_condition"
# null_pointer: the fix adds a None / existence guard.
if (("is none" in al or "is not none" in al) and "none" not in rl):
return "null_pointer"
if (".get(" in added and ".get(" not in removed
and ("[" in removed or "keyerror" in al)):
return "null_pointer"
# off_by_one: the fix flips a bound or tweaks an index by one.
if ("<=" in added) != ("<=" in removed) and ("<" in added or "<" in removed):
return "off_by_one"
if (">=" in added) != (">=" in removed):
return "off_by_one"
if (("- 1" in added) != ("- 1" in removed) or
("+ 1" in added) != ("+ 1" in removed)):
if "range(" in (added + removed) or "[" in removed:
return "off_by_one"
return None
def classify_by_keyword(message):
low = message.lower()
for label, kws in LABEL_KEYWORDS:
if any(k in low for k in kws):
return label
return None
def harvest_repo(repo, pages, token):
out = []
for page in range(1, pages + 1):
try:
commits = gh_get(
f"/repos/{repo}/commits?per_page=100&page={page}", token)
except GHError as exc:
print(f" {repo}: commit list failed ({exc})")
break
if not commits:
break
for c in commits:
message = (c.get("commit", {}) or {}).get("message", "")
first = message.split("\n")[0]
conventional = bool(_CONV_FIX_RE.match(first))
if not conventional and not _BUG_RE.search(first):
continue
try:
fn = prefix_function_from_commit(
repo.split("/")[0], repo.split("/")[1], c["sha"],
token, max_changes=25)
except GHError:
fn = None
if not fn:
continue
label = classify_by_diff(fn["patch"])
source = "diff-pattern"
if not label:
label = classify_by_keyword(message)
source = "commit-keyword"
if not label:
continue # category not reliably determinable -> skip
out.append({
"comments": [
f"{repo} | {fn['commit_url']}",
f"commit: {first[:150]}",
f"label_source: {source}"
f"{' (conventional fix:)' if conventional else ''}",
f"NEEDS LABEL REVIEW — proposed: {label}",
],
"complexity": tier(fn["body_lines"]),
"code": fn["code"],
"label": label,
"_source": source,
})
print(f" {repo} p{page}: {len(out)} total candidates")
return out
def main():
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("--pages", type=int, default=3)
ap.add_argument("--token", default=os.environ.get("GITHUB_TOKEN"))
args = ap.parse_args()
if not args.token:
print("WARNING: no GITHUB_TOKEN — capped at 60 req/hr.")
entries = []
for repo in REPOS:
if len(entries) >= TARGET_TOTAL:
break
entries.extend(harvest_repo(repo, args.pages, args.token))
out = os.path.join(_HERE, "staging", "code_review_candidates.py")
write_staging(out, [
"Auto-generated CodeReview candidates — REVIEW BEFORE MERGING into "
"tasks/code_review.py",
"Source: real pre-fix code from bug-fix commits in open-source "
"Python repos.",
"Labels are heuristic (see label_source per entry) — confirm EVERY "
"one against the diff.",
f"Generated: {date.today().isoformat()} by "
f"scripts/import_github_bugfixes.py v2 — {len(entries)} candidates.",
], entries)
by_label, by_src, by_tier = {}, {}, {}
for e in entries:
by_label[e["label"]] = by_label.get(e["label"], 0) + 1
by_src[e["_source"]] = by_src.get(e["_source"], 0) + 1
by_tier[e["complexity"]] = by_tier.get(e["complexity"], 0) + 1
print(f"\nWrote {len(entries)} candidates to {out}")
print(f" By label: {by_label}")
print(f" By source: {by_src} (diff-pattern is the stronger signal)")
print(f" By tier: {by_tier}")
if len(entries) < 70:
print(" NOTE: under 70 — raise --pages or add repos, then re-run.")
return 0
if __name__ == "__main__":
raise SystemExit(main())