-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathnlptopython.py
More file actions
118 lines (57 loc) · 2.2 KB
/
Copy pathnlptopython.py
File metadata and controls
118 lines (57 loc) · 2.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
# import spacy
# nlp = spacy.load("en_core_web_sm")
# doc=nlp("Apple is looking for buying uk startup for $1 billion")
# for token in doc:
# print(token.text)
#SPECIAL CASE TOKENIZATION
# import spacy
# from spacy.symbols import ORTH
# nlp=spacy.load("en_core_web_sm")
# doc=nlp("gimme that")
# special_case=[{ORTH:"gim"},{ORTH:"me"}]
# nlp.tokenizer.add_special_case("gimme",special_case)
# for token in nlp("gimme that"):
# print(token.text)
#POS (parts of speech) tagging in NLP;
# import spacy
# from spacy import displacy
# nlp=spacy.load("en_core_web_sm")
# doc=nlp("Python is a programming language. Current year is 2025 . Dollar Symbol is $")
# #print(doc)
# # for token in doc:
# # print(token)
# for token in doc:
# print(token,"--->",token.pos_)
# for token in doc:
# print(token,"--->",token.pos) #Now this will give the numbers because it gives the numbers
# displacy.serve(doc,style="dep") # this shows us the visualization part.
#Spacy stopwords removal.
#stopwords ----> these are the words that are often used in the sentence and create confusion to understand context of the document. ex: a an i me , my , myself, he , him,
# import spacy
# from spacy.lang.en.stop_words import STOP_WORDS
# #print(STOP_WORDS)
# #print("in" in STOP_WORDS) #first way to check.
# nlp=spacy.load("en_core_web_sm")
# # print(nlp.vocab["in"].is_stop) #second way to check.
# doc=nlp("Python is a programming language. I am learning Natural Language Processing")
# for token in doc:
# if(nlp.vocab[token.text].is_stop):
# print(token.text)
# for token in doc:
# if not nlp.vocab[token.text].is_stop:
# print(token.text)
#Named Entity Recongnition:
# import spacy
# from spacy import displacy
# nlp=spacy.load("en_core_web_sm")
# doc=nlp("Apple is looking at buying U.K startup for $1 billion")
# for token in doc:
# print(token.text,'-->', token.pos_)
# # displacy.serve(doc,style="ent")
# for entity in doc.ents:
# print(entity.text)
# # if entity.label_ == "ORG":
# # print(entity.text)
#Lemmatization -- goal is to reduce different forms of words to common forms.
# ex: am, are, is -> be
# car,cars,car's -> car