Repository navigation
Expand file tree
/
Copy pathURL.py
More file actions
102 lines (83 loc) · 2.44 KB
/
Copy pathURL.py
File metadata and controls
102 lines (83 loc) · 2.44 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
import pickle
import pprint
import requests
import tldextract
pp = pprint.PrettyPrinter(indent = 4)
spamUrlSet = set([])
#lol plain text API key
# client = SafebrowsinglookupClient('ABQIAAAACwkq5pvuWZ3fWsO072_EyRSv8I3jTLfAjRBzDFENLuchTwK1Zw')
# print client.lookup(*['google.com'])
def PopulateSpamUrlSet():
with open('dom-bl-base.txt') as f:
for line in f:
url = line.strip()
if ';' in url:
index = url.index(';')
url = url[:index]
spamUrlSet.add(url)
"""if contains ';' character, remove it and
everything after """
# with open('spam.csv') as f:
# for line in f:
# line = line.split(',')
# print line[1]
print "loaded spam urls"
"""
Input: tweeter
abstract:
goes through all urls of every tweet from a user
if any of the urls are in well known spam list, returns true
returns: true iff tweeter has sent a message containing a spam URL
"""
def isSpamTweeter(tweeter):
numSpam = 0
for tweet in tweeter[1]:
for url in tweet.urls:
urlTrail = RedirectHistory(str(url.url))
for hist in urlTrail[1]:
if IsSpam(hist):
numSpam+=1
print "found spam: " + hist
else:
print "not spam: " + hist
print "found " + str(numSpam) + " spam messages"
def main():
tweeter = pickle.load(open("USER-UnaChicaHapppy.pkl", "rb"))
isSpamTweeter(tweeter)
"""
Input:
url - string
Returns
tuple(numRedirects, allUrlsList)
"""
def RedirectHistory(url):
r = requests.get(url)
numRedirects = len(r.history)
allUrls = []
for hist in r.history:
allUrls.append(hist.url)
IsSpam(hist.url)
allUrls.append(r.url)
IsSpam(r.url)
return (numRedirects, allUrls)
"""
Input:
url - string
Returns
true iff url's domain is a spam domain
"""
def IsSpam(url):
ext = tldextract.extract(url)
domain = ext.domain + '.' + ext.suffix
# print domain
isSpam = False
if domain in spamUrlSet:
return True
else:
#make API call to google safebrowsing
#make call to other spam service
#if either of those return true add it to the spam set
return False
PopulateSpamUrlSet()
# if __name__ == "__main__":
# main()