-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathfetch_bwf.py
More file actions
147 lines (124 loc) · 5.33 KB
/
Copy pathfetch_bwf.py
File metadata and controls
147 lines (124 loc) · 5.33 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
#!/usr/bin/env python
"""
Pull match data straight from BWF's own API (the one bwfbadminton.com's
frontend talks to). Found by watching the network tab on a tournament page --
no auth, no cookies, just a browser user-agent.
Two steps, both cached and resumable:
1. tournament index -> data/bwf/index.json (one search call per year)
2. per-tournament matches -> data/bwf/matches/{id}.json.gz
We only pull the elite tier: Superseries / Grand Prix (2007-17), the HSBC
World Tour (2018+), and the Grade 1 majors (Worlds, Olympics). One call per
tournament returns every match with game scores, winner, and all four player
slots -- doubles partners intact, unique player ids, none of the name-spelling
mess the old CSVs had.
Run it, walk away (~40 min first time), rerun any time to top up new events:
python fetch_bwf.py
python fetch_bwf.py --index-only # just refresh the tournament list
"""
import argparse
import gzip
import json
import time
import urllib.request
from datetime import date
from pathlib import Path
API = "https://extranet-lv.bwfbadminton.com/api"
UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/125.0 Safari/537.36")
OUT = Path(__file__).resolve().parent / "data" / "bwf"
INDEX = OUT / "index.json"
MATCHES = OUT / "matches"
FIRST_YEAR = 2007 # Superseries starts here; before it the top tier is patchy
DELAY = 1.3 # seconds between requests -- don't be a nuisance
# category strings as they appear in the index
ELITE = {
"World Superseries", "World Superseries Premier", "World Superseries Finals",
"Grand Prix", "Grand Prix Gold",
"HSBC BWF World Tour Super 300", "HSBC BWF World Tour Super 500",
"HSBC BWF World Tour Super 750", "HSBC BWF World Tour Super 1000",
"HSBC BWF World Tour Finals", "BWF Tour Super 100",
}
# grade-1 individual events (Worlds, Olympics) carry this category; the same
# label is also used for junior worlds, so we screen names too
G1_INDIVIDUAL = "Individual Tournaments"
G1_NAME_SKIP = ("junior", "youth", "senior", "university")
def get(url, tries=4):
for i in range(tries):
try:
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=45) as r:
return json.load(r)
except Exception:
if i == tries - 1:
raise
time.sleep(3 * (i + 1))
def fetch_index():
"""One search call per year (deep pagination times out server-side)."""
seen, out = set(), []
for y in range(FIRST_YEAR, date.today().year + 1):
page, last = 1, 1
while page <= last:
d = get(f"{API}/vue-tournaments-search?startDate={y}-01-01"
f"&endDate={y}-12-31&page={page}&perPage=100"
f"&drawCount=1&activeTab=6")
res = d["results"]
last = res["last_page"]
for t in res["data"]:
if t["id"] not in seen:
seen.add(t["id"])
out.append(t)
page += 1
time.sleep(DELAY)
print(f" {y}: index has {len(out)} tournaments so far")
OUT.mkdir(parents=True, exist_ok=True)
json.dump(out, open(INDEX, "w", encoding="utf-8"))
print(f"index: {len(out)} tournaments -> {INDEX}")
return out
def is_elite(t):
cat = (t.get("category") or "").strip()
if cat in ELITE:
return True
if cat.startswith("Grade 1") and G1_INDIVIDUAL in cat:
name = (t.get("name") or "").lower()
return not any(s in name for s in G1_NAME_SKIP)
return False
def fetch_matches(tournaments):
MATCHES.mkdir(parents=True, exist_ok=True)
todo = [t for t in tournaments if is_elite(t)
and "(cancelled)" not in (t.get("name") or "").lower()
and not (MATCHES / f"{t['id']}.json.gz").exists()]
print(f"{len(todo)} tournaments to fetch")
for i, t in enumerate(todo, 1):
tid = t["id"]
try:
d = get(f"{API}/vue-tournament-matches?drawCount=0&searchKey="
f"&tmtId={tid}&tmtType=0&isPara=false")
from badminton.bwf import _matches
groups = _matches(d)
if not groups:
# not played yet (or no detail) -- don't cache, retry next run
print(f"[{i}/{len(todo)}] {t['start_date'][:4]} {t['name'].strip()[:55]} "
f"(no matches yet, skipped)", flush=True)
time.sleep(DELAY)
continue
with gzip.open(MATCHES / f"{tid}.json.gz", "wt", encoding="utf-8") as f:
json.dump(d, f)
print(f"[{i}/{len(todo)}] {t['start_date'][:4]} {t['name'].strip()[:55]} "
f"({len(groups)} matches)", flush=True)
except Exception as e:
print(f"[{i}/{len(todo)}] {tid} FAILED: {e}", flush=True)
time.sleep(DELAY)
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--index-only", action="store_true")
a = ap.parse_args()
if INDEX.exists() and not a.index_only:
tournaments = json.load(open(INDEX, encoding="utf-8"))
print(f"using cached index ({len(tournaments)} tournaments); "
f"--index-only to refresh")
else:
tournaments = fetch_index()
if not a.index_only:
fetch_matches(tournaments)
if __name__ == "__main__":
main()