-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathqualify_channels.py
More file actions
101 lines (83 loc) · 3.54 KB
/
Copy pathqualify_channels.py
File metadata and controls
101 lines (83 loc) · 3.54 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
"""
Qualify a list of YouTube channels by contact route.
The job most people arrive with: a list of channel handles (a sponsorship
shortlist, a partner pipeline, a competitor set) and a need for one row each with
the contact route filled in: website, socials, and a public email where the
creator published one.
pip install requests
export CHOCODATA_API_KEY="your_key"
python youtube_email_scraper_api_codes/qualify_channels.py
Reads channels.txt if present (one @handle or channel URL per line), otherwise
runs the sample list. Writes channels_contacts.csv and prints one row per channel.
Reads public brand, media and institutional channels only.
"""
import csv
import os
import sys
import time
from concurrent.futures import ThreadPoolExecutor
import requests
API = "https://api.chocodata.com/api/v1/youtube/email"
KEY = os.environ.get("CHOCODATA_API_KEY")
if not KEY:
sys.exit("Set CHOCODATA_API_KEY first. Free key (1,000 requests, one-time): https://chocodata.com")
SAMPLE = ["@NASA", "@TED", "@khanacademy", "@Microsoft", "@GitHub"]
def load_targets() -> list:
if os.path.exists("channels.txt"):
with open("channels.txt", encoding="utf-8") as fh:
return [ln.strip() for ln in fh if ln.strip() and not ln.startswith("#")]
return SAMPLE
def one(channel: str, retries: int = 1) -> dict | None:
for attempt in range(retries + 1):
r = requests.get(API, params={"api_key": KEY, "channel": channel}, timeout=90)
if r.status_code == 200:
return r.json()
if r.status_code == 401:
sys.exit("401 INVALID_API_KEY: check CHOCODATA_API_KEY. https://chocodata.com")
if r.status_code == 402:
sys.exit("402 INSUFFICIENT_CREDITS: top up or upgrade. https://chocodata.com/pricing")
if r.status_code == 429:
time.sleep(5)
continue
if r.status_code in (404, 502) and attempt < retries:
time.sleep(6)
continue
return None
return None
def website(links: list) -> str:
"""First non-social About link is usually the channel's own site."""
social = ("youtube.com", "instagram.", "tiktok.", "twitter.", "x.com",
"facebook.", "discord.", "patreon.", "linktr.ee")
for link in links:
url = (link.get("url") or "").lower()
if url and not any(s in url for s in social):
return link["url"]
return links[0]["url"] if links else ""
def main() -> None:
targets = load_targets()
# Four workers keep well inside the concurrency cap and the 120/60s window.
with ThreadPoolExecutor(max_workers=4) as pool:
results = list(pool.map(one, targets))
rows = []
for channel, data in zip(targets, results):
if not data:
print(f" skip {channel} (not resolved)")
continue
row = {
"channel": channel,
"channel_name": data.get("channel_name"),
"email": "; ".join(data.get("emails") or []),
"website": website(data.get("links") or []),
"links_count": data.get("links_count"),
}
rows.append(row)
mark = row["email"] if row["email"] else "-"
print(f' {channel:16s} email={mark:32s} links={row["links_count"]}')
if rows:
with open("channels_contacts.csv", "w", newline="", encoding="utf-8") as fh:
w = csv.DictWriter(fh, fieldnames=list(rows[0].keys()))
w.writeheader()
w.writerows(rows)
print(f"\nWrote channels_contacts.csv ({len(rows)} rows)")
if __name__ == "__main__":
main()