-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathutils.py
More file actions
48 lines (38 loc) · 1.51 KB
/
Copy pathutils.py
File metadata and controls
48 lines (38 loc) · 1.51 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
import re
import lxml.html
from lxml.html.clean import Cleaner
import nltk
from nltk.stem.wordnet import WordNetLemmatizer
import string
def clean_punctuation(text: string):
return text.translate(str.maketrans(string.punctuation, ' ' * len(string.punctuation)))
def clean_html(html_string: string, stopwords, lemmatize: bool):
# Parse the HTML
html_content = lxml.html.fromstring(html_string)
# Remove HTML tags to only keep text
cleaner = Cleaner()
cleaner.javascript = True
cleaner.style = True
text = cleaner.clean_html(html_content).text_content()
# Clean lowered text from punctuation and remove stopwords
no_punc = clean_punctuation(text.lower())
words = nltk.word_tokenize(no_punc)
pattern = re.compile(r'[^\W\d_]+', re.U)
final_words = []
if lemmatize:
lemmatiser = WordNetLemmatizer()
wordnet_tag = {'NN': 'n', 'JJ': 'a', 'VB': 'v', 'RB': 'r'}
tagged = nltk.pos_tag(words)
for token in tagged:
word = token[0]
if len(word) >= 2 and pattern.match(word) is not None and word not in stopwords:
try:
lemma = lemmatiser.lemmatize(word, wordnet_tag[token[1][:2]])
except:
lemma = lemmatiser.lemmatize(word)
final_words.append(lemma)
else:
for word in words:
if len(word) >= 2 and pattern.match(word) is not None and word not in stopwords:
final_words.append(word)
return ' '.join(final_words)