-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathtf_idf.py
More file actions
58 lines (42 loc) · 1.37 KB
/
Copy pathtf_idf.py
File metadata and controls
58 lines (42 loc) · 1.37 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
import glob
import os
import re
import pandas as pd
from sklearn.feature_extraction.text import TfidfVectorizer
from nltk.corpus import stopwords
path = "INSERT PATH HERE"
# number of keywords to be extracted; initialised to 10
number_of_keywords = 10
# USE THIS TO CREATE YOUR CORPUS FROM .TXT FILES
def extract_corpus():
corpus = []
for file in glob.glob(os.path.join(path, '*.txt')):
f = open(file, 'r', encoding='latin-1')
text = f.read()
f.close()
# formatting text for pretty print
text = text.replace('.', ' ')
text = re.sub(r'\s+', ' ', re.sub(r'[^\w \s]', '', text)).lower()
corpus.append(text)
return corpus
# TF_IDF MODULE; RETURNS KEYWORDS FOR EACH DOCUMENT
def tf_idf(corpus):
# initializing vectorizer
vectorizer = TfidfVectorizer()
vectors = vectorizer.fit_transform(corpus)
names = vectorizer.get_feature_names()
data = vectors.todense().tolist()
# Create a dataframe with the results
df = pd.DataFrame(data, columns=names)
# filtering stopwords
st = set(stopwords.words('english'))
df = df[filter(lambda x: x not in list(st), df.columns)]
# printing top 10 keywords
N = number_of_keywords
for i in df.iterrows():
print(i[1].sort_values(ascending=False)[:N])
return df
def test():
corpus = extract_corpus()
tf_idf(corpus)
test()