-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathscrape.py
More file actions
124 lines (90 loc) · 4.18 KB
/
Copy pathscrape.py
File metadata and controls
124 lines (90 loc) · 4.18 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
import requests
import pandas as pd
import json
from flatten_json import flatten
def main():
make_many_year_dfs(5)
combine()
return None
def make_many_year_dfs(num_of_years):
'''returns a dataframe of all articles each month going back the specified number of years'''
articles_all_years_df = make_year_df(2019)
filename = str('data/articles_' + str(2019) + '.csv')
articles_all_years_df.to_csv(filename)
for i in range(1, num_of_years):
filename = str('data/articles_' + str(2019 - i) + '.csv')
articles_year_df = make_year_df(2019 - i)
articles_year_df.to_csv(filename)
articles_all_years_df = articles_all_years_df.append(articles_year_df, ignore_index=True)
filename = str('data/articles_going_back_' + str(num_of_years) + '_years.csv')
articles_all_years_df.to_csv(filename)
return articles_all_years_df
def make_year_df(year):
'''returns a dataframe of the year, and creates a csv.'''
articles_year_df = pd.DataFrame()
if year == 2019:
# hard code month of 2019 in for now in the end of first for loop range, will change this in final tweaks
for i in range(1,3):
month_df = get_article_month_year(i, year)
articles_year_df = articles_year_df.append(month_df, ignore_index=True)
# save csv file from dataframe
filename = str('data/articles_' + str(year) + '.csv')
articles_year_df.to_csv(filename)
return articles_year_df
else:
for i in range(1,13):
month_df = get_article_month_year(i, year)
articles_year_df = articles_year_df.append(month_df, ignore_index=True)
# save csv file from dataframe
filename = str('data/articles_' + str(year) + '.csv')
articles_year_df.to_csv(filename)
return articles_year_df
def get_article_month_year(month, year):
'''returns a dataframe of all meta data of all articles published in the given month and year'''
#set url base and end
url_base = 'https://api.nytimes.com/svc/archive/v1/'
url_end = '.json?api-key=[API KEY]'
url = str(url_base + str(year) + '/' + str(month) + url_end)
#get month's archive from nytimes api, and save as json file
request = requests.get(url)
data = request.json()
with open('data.json', 'w') as f:
json.dump(data, f)
#get data from json file to dictionary
with open('data.json') as json_file:
data = json_file.readlines()
#convert all strings in list to actual json objects.
data = list(map(json.loads, data))
#create staging dataframe for next step
df = pd.DataFrame(data)
# create empty dataframe for all month's articles
articles_all_df = pd.DataFrame()
#loop through each article
for i in range(len(df['response'][0]['docs'])): # df['response'][0]['docs'] is the location to dig in to get to the article dictionary level
try:
article_dic = (df['response'][0]['docs'][i])
article_dic_flat = flatten(article_dic)
#next four lines of code are to remove keys that are making df creation error
article_dic_flat.pop('blog', None)
article_dic_flat.pop('multimedia_2_legacy', None)
article_dic_flat.pop('multimedia_3_legacy', None)
article_dic_flat.pop('multimedia_4_legacy', None)
print(((len(df['response'][0]['docs'])) - i), month, year)
#create dataframe from flattened and cleaned article dictionary
article_df = pd.DataFrame(article_dic_flat, index=[0])
#append article df to all articles df
articles_all_df = articles_all_df.append(article_df, ignore_index=True)
except:
None
return articles_all_df
def combine():
df1 = pd.read_csv('data/articles_2018.csv')
df2 = pd.read_csv('data/articles_2017.csv')
df3 = pd.read_csv('data/articles_2016.csv')
df4 = pd.read_csv('data/articles_2015.csv')
df5 = pd.read_csv('data/articles_2014.csv')
df = pd.concat([df1, df2, df3, df4, df5])
df.to_csv('data/articles_2014-2018.csv')
return None
if __name__== "__main__":
main()