-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathclean.py
More file actions
150 lines (111 loc) · 6.42 KB
/
Copy pathclean.py
File metadata and controls
150 lines (111 loc) · 6.42 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
import pandas as pd
'''
to add in the future:
progress bar
'''
def main():
df = pd.read_csv('data/articles_2014-2018.csv')
df = clean(df)
return df
def clean(df):
###First clean
#remove rows based on nan (these columns must have values, otherwise remove the row)
df = df.dropna(subset=['byline_original'])
df = df.dropna(subset=['headline_main'])
df = df.dropna(subset=['byline_person_0_lastname'])
df = df.dropna(subset=['snippet'])
#rows to delete based on conditions
delete_cols = df[df['type_of_material'] == 'Video'].index
df.drop(delete_cols , inplace=True)
delete_cols = df[df['byline_person_1_lastname'].notnull() == True].index
df.drop(delete_cols , inplace=True)
#remove unneeded columns
cols_to_keep = ['byline_person_0_firstname',
'byline_person_0_lastname',
'byline_person_0_middlename',
'headline_main',
'keywords_0_value',
'keywords_1_value',
'keywords_2_value',
'pub_date',
'section_name',
'snippet',
'source',
'type_of_material',
'web_url',
'word_count'
]
df = df[cols_to_keep]
df = df.reset_index()
# use first, middel, and last names to create an author column
df['byline_person_0_middlename'].fillna('None', inplace=True)
df = df.reset_index()
df['author'] = df[['byline_person_0_firstname', 'byline_person_0_middlename', 'byline_person_0_lastname']].apply(lambda x: ' '.join(x), axis=1)
df['author'] = df['author'].map(lambda x: x.replace(' None ', ' ').lower())
#fill all remaining nan values - only additional keywords are missing values at this point
df = df.fillna('None')
#clean and lower all text columns - need to find a cleaner way to write the following code (i should atleast make each section a one liner)
df['byline_person_0_firstname'] = df['byline_person_0_firstname'].str.replace('[^\w\s]','')
df['byline_person_0_firstname'] = df['byline_person_0_firstname'].str.lower()
df['byline_person_0_firstname'] = df['byline_person_0_firstname'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
df['byline_person_0_lastname'] = df['byline_person_0_lastname'].str.replace('[^\w\s]','')
df['byline_person_0_lastname'] = df['byline_person_0_lastname'].str.lower()
df['byline_person_0_lastname'] = df['byline_person_0_lastname'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
df['byline_person_0_middlename'] = df['byline_person_0_middlename'].str.replace('[^\w\s]','')
df['byline_person_0_middlename'] = df['byline_person_0_middlename'].str.lower()
df['byline_person_0_middlename'] = df['byline_person_0_middlename'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
df['headline_main'] = df['headline_main'].str.replace('[^\w\s]','')
df['headline_main'] = df['headline_main'].str.lower()
df['headline_main'] = df['headline_main'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
df['keywords_0_value'] = df['keywords_0_value'].str.replace('[^\w\s]','')
df['keywords_0_value'] = df['keywords_0_value'].str.lower()
df['keywords_0_value'] = df['keywords_0_value'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
df['keywords_1_value'] = df['keywords_1_value'].str.replace('[^\w\s]','')
df['keywords_1_value'] = df['keywords_1_value'].str.lower()
df['keywords_1_value'] = df['keywords_1_value'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
df['keywords_2_value'] = df['keywords_2_value'].str.replace('[^\w\s]','')
df['keywords_2_value'] = df['keywords_2_value'].str.lower()
df['keywords_2_value'] = df['keywords_2_value'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
df['section_name'] = df['section_name'].str.replace('[^\w\s]','')
df['section_name'] = df['section_name'].str.lower()
df['section_name'] = df['section_name'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
df['snippet'] = df['snippet'].str.replace('[^\w\s]','')
df['snippet'] = df['snippet'].str.lower()
df['snippet'] = df['snippet'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
df['source'] = df['source'].str.replace('[^\w\s]','')
df['source'] = df['source'].str.lower()
df['source'] = df['source'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
df['type_of_material'] = df['type_of_material'].str.replace('[^\w\s]','')
df['type_of_material'] = df['type_of_material'].str.lower()
df['type_of_material'] = df['type_of_material'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
#clean date column
df['pub_date'] = df['pub_date'].str.replace('T', ' ').str.replace('0000', '').str.replace('+', '')
#create Author ID - i need to come back and generate numerical IDs
df['author_id'] = df[['byline_person_0_firstname', 'byline_person_0_middlename', 'byline_person_0_lastname']].apply(lambda x: ''.join(x), axis=1)
#create article ID - i need to come back and generate numerical IDs
df['article_id'] = df['web_url']
df['article_id'] = df['article_id'].str.replace('[^\w\s]','')
df['article_id'] = df['article_id'].str.lower()
df['article_id'] = df['article_id'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
df['article_id'] = df['article_id'].str.replace('httpswwwnytimescom','')
### NLP Preprocesing - enriching data with added columns specifically for the NLP
#all columns will live in a postgres db, but the folowwing columns will be used for NLP
#convert headline and leads to list of a string, after removing punctuation, lowercaseing, removing accented letters,
# and removing duplicate rows based on text and author columns
df['text'] = df['headline_main'] + ' ' + df['snippet']
df['text'] = df['text'].str.replace('[^\w\s]','')
df['text'] = df['text'].str.lower()
df['text'] = df['text'].str.normalize('NFKD').str.encode('ascii', errors='ignore').str.decode('utf-8')
#remove duplicates
df = df[~df.duplicated(['text', 'author'])]
#turn text column in to a list of the string for LDA, etc.
df['text'] = df['text'].map(lambda x: [x])
#remove junk columns
df = df.drop('index', axis=1)
df = df.drop('level_0', axis=1)
#save result as csv and pkl (not sure which is best right now, need to come back and choose)
df.to_csv('data/articles_2014-2018_clean.csv')
df.to_pickle('data/articles_2014-2018_clean.pkl')
return df
if __name__== "__main__":
main()