-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathprepdata.py
More file actions
17 lines (17 loc) · 848 Bytes
/
Copy pathprepdata.py
File metadata and controls
17 lines (17 loc) · 848 Bytes
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
# Taken from https://github.com/KellerJordan/modded-nanogpt/blob/master/train_gpt.py
import os
import sys
from huggingface_hub import hf_hub_download
# Download the GPT-2 tokens of FinewebEDU10B from huggingface. This
# saves about an hour of startup time compared to regenerating them.
def get(fname):
local_dir = os.path.join(os.path.dirname(__file__), 'finewebedu10B')
if not os.path.exists(os.path.join(local_dir, fname)):
hf_hub_download(repo_id="kjj0/finewebedu10B-gpt2", filename=fname,
repo_type="dataset", local_dir=local_dir)
get("finewebedu_val_%06d.bin" % 0)
num_chunks = 99 # full FinewebEDU10B. Each chunk is 100M tokens
if len(sys.argv) >= 2: # we can pass an argument to download less
num_chunks = int(sys.argv[1])
for i in range(1, num_chunks+1):
get("finewebedu_train_%06d.bin" % i)