-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathgithub_fetch.py
More file actions
132 lines (113 loc) · 4.78 KB
/
Copy pathgithub_fetch.py
File metadata and controls
132 lines (113 loc) · 4.78 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
"""GitHub metrics fetcher.
Two signals:
* star history (sampled, the way star-history.com works around the
40k-stargazer pagination cap),
* weekly commit activity (last 52w) and current repo stats.
Set a GITHUB_TOKEN env var to lift the rate limit from 60/hr to 5000/hr.
Everything is cached on disk so the notebook is re-runnable.
Repo note: Vue and Angular have split repos (Vue 2 `vuejs/vue` vs Vue 3
`vuejs/core`; modern `angular/angular` vs archived `angular/angular.js`).
We track the MODERN repo and flag the split.
"""
from __future__ import annotations
import json
import os
import subprocess
import time
from pathlib import Path
import pandas as pd
RAW_DIR = Path(__file__).resolve().parent.parent / "data" / "raw" / "github"
FRAMEWORK_REPOS = {
"Angular": "angular/angular",
"React": "facebook/react",
"Vue": "vuejs/core", # Vue 3 (older stars live in vuejs/vue)
"Svelte": "sveltejs/svelte",
}
def _curl_get(url: str, accept: str = "application/vnd.github+json"):
"""GET via curl (Python HTTPS hangs in this env). Returns (status, json)."""
args = ["curl", "-s", "-L", "-4", "--max-time", "30", "-w", "\n%{http_code}",
"-H", f"Accept: {accept}",
"-H", "User-Agent: angularstats-research"]
tok = os.environ.get("GITHUB_TOKEN")
if tok:
args += ["-H", f"Authorization: Bearer {tok}"]
args.append(url)
out = subprocess.run(args, capture_output=True, text=True, timeout=40)
body, _, status = out.stdout.rpartition("\n")
try:
return int(status), json.loads(body) if body.strip() else None
except Exception: # noqa: BLE001
return int(status or 0), None
def fetch_repo_stats(repo: str, use_cache: bool = True) -> dict:
RAW_DIR.mkdir(parents=True, exist_ok=True)
cache = RAW_DIR / f"{repo.replace('/', '__')}__stats.json"
if use_cache and cache.exists():
return json.loads(cache.read_text())
status, d = _curl_get(f"https://api.github.com/repos/{repo}")
if status != 200 or not d:
raise RuntimeError(f"GitHub repo stats {repo} -> HTTP {status}")
out = {
"repo": repo,
"stars": d["stargazers_count"],
"forks": d["forks_count"],
"open_issues": d["open_issues_count"],
"created_at": d["created_at"],
"pushed_at": d["pushed_at"],
}
cache.write_text(json.dumps(out))
return out
def fetch_star_history(repo: str, max_pages: int = 100,
use_cache: bool = True, pause: float = 0.3) -> pd.Series:
"""Sampled cumulative star history.
GitHub caps stargazer pagination at 400 pages (40k stars). For large
repos we SAMPLE sparse pages and interpolate -- the cumulative count at
each sampled page's last `starred_at` -- exactly how star-history.com
handles the cap. Returns a cumulative-stars Series indexed by date.
"""
RAW_DIR.mkdir(parents=True, exist_ok=True)
cache = RAW_DIR / f"{repo.replace('/', '__')}__stars.json"
if use_cache and cache.exists():
d = json.loads(cache.read_text())
s = pd.Series(d)
s.index = pd.to_datetime(s.index)
return s.sort_index()
stats = fetch_repo_stats(repo, use_cache=use_cache)
total = stats["stars"]
per_page = 100
# GitHub hard-caps stargazer pagination at page 400 (40k stars). Unauth,
# we cannot page past that, so the sampled history only spans a repo's
# first 40k stars (its early-growth years). Set GITHUB_TOKEN for more.
real_last = max(1, -(-total // per_page))
reachable_last = min(400, real_last)
if reachable_last <= max_pages:
pages = list(range(1, reachable_last + 1))
else:
step = reachable_last / max_pages
pages = sorted({max(1, int(round(i * step))) for i in range(1, max_pages + 1)})
points: dict[str, int] = {}
for p in pages:
url = (f"https://api.github.com/repos/{repo}/stargazers"
f"?per_page={per_page}&page={p}")
status, rows = _curl_get(url, accept="application/vnd.github.star+json")
if status != 200 or not rows:
break
cum = (p - 1) * per_page + len(rows)
last_star = rows[-1].get("starred_at")
if last_star:
points[last_star[:10]] = cum
time.sleep(pause)
cache.write_text(json.dumps(points))
s = pd.Series(points)
s.index = pd.to_datetime(s.index)
return s.sort_index()
def fetch_all_stars(repos: dict[str, str] = None, **kw) -> pd.DataFrame:
repos = repos or FRAMEWORK_REPOS
cols = {}
for name, repo in repos.items():
try:
cols[name] = fetch_star_history(repo, **kw)
except Exception as e: # noqa: BLE001
print(f" ! {name} ({repo}) star history failed: {e}")
# union of dates, forward-fill cumulative counts
df = pd.DataFrame(cols).sort_index()
return df.ffill()