From 1be4c327464ff3c4efdb36ac06cecc351d488021 Mon Sep 17 00:00:00 2001 From: Alice Wang Date: Thu, 22 Jun 2017 12:01:52 +0200 Subject: [PATCH 1/6] add support to ETF --- morningscraper/__init__.py | 65 +++++++++++++++++++++++++++----------- 1 file changed, 47 insertions(+), 18 deletions(-) diff --git a/morningscraper/__init__.py b/morningscraper/__init__.py index 8674e84..86e89cc 100644 --- a/morningscraper/__init__.py +++ b/morningscraper/__init__.py @@ -33,6 +33,12 @@ def fix_url(url): return url +def make_soup(url, parser="html.parser"): + response = urlopen(url) + soup = BeautifulSoup(response, parser) + return soup + + def search(ref, verbose=False): ''' Search morningstar.co.uk for ref @@ -75,8 +81,7 @@ def search(ref, verbose=False): if verbose: print('Search for: %s' % ref) - data = urlopen(SEARCH_BASE % quote(ref)).read() - parsed_html = BeautifulSoup(data) + parsed_html = make_soup(SEARCH_BASE % quote(ref)) results = [] stocks = parsed_html.find_all( 'table', id='ctl00_MainContent_stockTable' @@ -93,25 +98,29 @@ def search(ref, verbose=False): 'td', class_='searchCurrency' )[0].text, }) - funds = parsed_html.find_all( - 'table', id='ctl00_MainContent_fundTable' - ) - if funds: - funds = funds[0].find_all('tr')[1:] - for fund in funds: - data = fund.find_all('td') - results.append({ - 'name': data[0].text, - 'url': fix_url(data[0].a.get('href')), - 'type': 'Fund', - 'ISIN': data[1].text, - }) + + funds = [] + for instrument_type in ["fund", "etf"]: + funds = parsed_html.find_all( + 'table', id='ctl00_MainContent_{}Table'.format(instrument_type) + ) + if funds: + funds = funds[0].find_all('tr')[1:] + for fund in funds: + data = fund.find_all('td') + results.append({ + 'name': data[0].text, + 'url': fix_url(data[0].a.get('href')), + 'type': instrument_type, + 'ISIN': data[1].text, + }) + break if verbose: if results: print('%s item(s) found.' % len(results)) for item in results: - print('\t%s\t%s' % (item['type'], item['name'])) + print(item) else: print('No items found.') return results @@ -190,6 +199,11 @@ def get_url(url, verbose=False): result = _get_stock(url) except: result = None + elif '/uk/etf/' in url: + try: + result = _get_etf(url) + except: + result = None else: raise Exception('Unrecognised url %r' % url) if verbose: @@ -201,8 +215,7 @@ def _get_funds(url): ''' Get and parse returned html for fund pages e.g. http://www.morningstar.co.uk/uk/funds/snapshot/snapshot.aspx?id=F00000NGEH ''' - data = urlopen(url).read() - parsed_html = BeautifulSoup(data) + parsed_html = make_soup(url) title = parsed_html.find_all('div', class_='snapshotTitleBox')[0].h1.text table = parsed_html.find_all('table', class_='overviewKeyStatsTable')[0] for tr in table.find_all('tr'): @@ -228,6 +241,20 @@ def _get_funds(url): } +def _get_etf(url): + soup = make_soup(url) + text = soup.find_all('div', class_='snapshotTitleBox')[0].h1.text + result = {"name": text.split('|')[0].strip(), + "ticker": text.split('|')[1].strip()} + for keyword in ["Exchange", "ISIN"]: + line = soup.find(text=keyword) + if line is None: + continue + text = line.parent.nextSibling.nextSibling.text + result[keyword] = str(text) + return result + + def _get_stock(url): ''' Get and parse returned html for stock pages e.g. http://tools.morningstar.co.uk/uk/stockreport/default.aspx?SecurityToken=0P000090RG]3]0]E0WWE$$ALL @@ -255,6 +282,8 @@ def _get_stock(url): if __name__ == '__main__': + search('EWJ', verbose=True) + get_data('EWJ', verbose=True) get_data('GB00B54RK123', verbose=True) get_data('LLOY LSE', verbose=True) get_data('GOOG NASDAQ', verbose=True) From 57bba5cee9d5fca81631428ad1f942e516b87a2a Mon Sep 17 00:00:00 2001 From: Alice Wang Date: Thu, 22 Jun 2017 15:03:08 +0200 Subject: [PATCH 2/6] update scraper page structure --- morningscraper/__init__.py | 101 ++------------------------------- morningscraper/security.py | 113 +++++++++++++++++++++++++++++++++++++ setup.py | 2 +- 3 files changed, 119 insertions(+), 97 deletions(-) create mode 100644 morningscraper/security.py diff --git a/morningscraper/__init__.py b/morningscraper/__init__.py index 86e89cc..ffc4615 100644 --- a/morningscraper/__init__.py +++ b/morningscraper/__init__.py @@ -6,6 +6,8 @@ from bs4 import BeautifulSoup +from security import make_soup, SecurityPage + if sys.version_info[0] == 3: from urllib.request import urlopen @@ -33,12 +35,6 @@ def fix_url(url): return url -def make_soup(url, parser="html.parser"): - response = urlopen(url) - soup = BeautifulSoup(response, parser) - return soup - - def search(ref, verbose=False): ''' Search morningstar.co.uk for ref @@ -99,7 +95,6 @@ def search(ref, verbose=False): )[0].text, }) - funds = [] for instrument_type in ["fund", "etf"]: funds = parsed_html.find_all( 'table', id='ctl00_MainContent_{}Table'.format(instrument_type) @@ -188,101 +183,15 @@ def get_url(url, verbose=False): print('\nOpening %s' % url) if not urlsplit(url).netloc.endswith(SITE): raise Exception('Non morningstar.co.uk url %r' % url) - result = None - if '/uk/funds/snapshot/snapshot' in url: - try: - result = _get_funds(url) - except: - result = None - elif '/uk/stockreport/' in url: - try: - result = _get_stock(url) - except: - result = None - elif '/uk/etf/' in url: - try: - result = _get_etf(url) - except: - result = None - else: - raise Exception('Unrecognised url %r' % url) + page = SecurityPage.from_url(url) + result = page.get_data() if verbose: print(result) return result -def _get_funds(url): - ''' Get and parse returned html for fund pages e.g. - http://www.morningstar.co.uk/uk/funds/snapshot/snapshot.aspx?id=F00000NGEH - ''' - parsed_html = make_soup(url) - title = parsed_html.find_all('div', class_='snapshotTitleBox')[0].h1.text - table = parsed_html.find_all('table', class_='overviewKeyStatsTable')[0] - for tr in table.find_all('tr'): - tds = tr.find_all('td') - if len(tds) != 3: - continue - if tds[0].text.startswith('NAV'): - date = tds[0].span.text - (currency, value) = tds[2].text.split() - if tds[0].text.startswith('Day Change'): - change = tds[2].text.strip() - if tds[0].text.startswith('ISIN'): - isin = tds[2].text.strip() - return { - 'title': title, - 'value': Decimal(value), - 'currency': currency, - 'change': change, - 'date': dmy_2_date(date), - 'url': url, - 'ISIN': isin, - 'type': 'Fund', - } - - -def _get_etf(url): - soup = make_soup(url) - text = soup.find_all('div', class_='snapshotTitleBox')[0].h1.text - result = {"name": text.split('|')[0].strip(), - "ticker": text.split('|')[1].strip()} - for keyword in ["Exchange", "ISIN"]: - line = soup.find(text=keyword) - if line is None: - continue - text = line.parent.nextSibling.nextSibling.text - result[keyword] = str(text) - return result - - -def _get_stock(url): - ''' Get and parse returned html for stock pages e.g. - http://tools.morningstar.co.uk/uk/stockreport/default.aspx?SecurityToken=0P000090RG]3]0]E0WWE$$ALL - ''' - data = urlopen(url).read() - parsed_html = BeautifulSoup(data) - title = parsed_html.find_all('span', class_='securityName')[0].text - value = parsed_html.find_all('span', id='Col0Price')[0].text - change = parsed_html.find_all('span', id='Col0PriceDetail')[0].text - change = change.split('|')[1].strip() - date = parsed_html.find_all('p', id='Col0PriceTime')[0].text[6:16] - currency = parsed_html.find_all('p', id='Col0PriceTime')[0].text - currency = re.search(r'\|\s([A-Z]{3,4})\b', currency).group(1) - isin = parsed_html.find_all('td', id='Col0Isin')[0].text - return { - 'title': title, - 'value': Decimal(value), - 'currency': currency, - 'change': change, - 'date': dmy_2_date(date), - 'url': url, - 'ISIN': isin, - 'type': 'Stock', - } - - if __name__ == '__main__': - search('EWJ', verbose=True) + # search('EWJ', verbose=True) get_data('EWJ', verbose=True) get_data('GB00B54RK123', verbose=True) get_data('LLOY LSE', verbose=True) diff --git a/morningscraper/security.py b/morningscraper/security.py new file mode 100644 index 0000000..020807d --- /dev/null +++ b/morningscraper/security.py @@ -0,0 +1,113 @@ +import sys +import re +import abc +import six +from decimal import Decimal +from datetime import datetime + +from bs4 import BeautifulSoup + +if sys.version_info[0] == 3: + from urllib.request import urlopen +elif sys.version_info[0] == 2: + from urllib import urlopen +else: + raise Exception('Python version 2 or 3 required') + + +def make_soup(url, parser="html.parser"): + response = urlopen(url) + soup = BeautifulSoup(response, parser) + return soup + + +@six.add_metaclass(abc.ABCMeta) +class SecurityPage(object): + + @classmethod + def from_url(cls, url): + if '/uk/funds/snapshot/snapshot' in url: + return FundsPage(url) + elif '/uk/stockreport/' in url: + return StockPage(url) + elif '/uk/etf/' in url: + return ETFPage(url) + + def __init__(self, url): + self.url = url + cls_name = self.__class__.__name__ + security_type = cls_name[:cls_name.find("Page")] + self.data_ = {"type": security_type, "url": self.url} + + def get_data(self): + soup = make_soup(self.url) + self._update_data(soup) + return self.data_ + + @abc.abstractmethod + def _update_data(self, soup): + """""" + + +class FundsPage(SecurityPage): + """ + http://www.morningstar.co.uk/uk/funds/snapshot/snapshot.aspx?id=F00000NGEH + """ + def _update_data(self, soup): + text = soup.find_all('div', class_='snapshotTitleBox')[0].h1.text + self.data_["name"] = str(text) + table = soup.find_all('table', class_='overviewKeyStatsTable')[0] + for tr in table.find_all('tr'): + tds = tr.find_all('td') + if len(tds) != 3: + continue + if tds[0].text.startswith('NAV'): + date = tds[0].span.text + (currency, value) = tds[2].text.split() + if tds[0].text.startswith('Day Change'): + change = tds[2].text.strip() + if tds[0].text.startswith('ISIN'): + isin = tds[2].text.strip() + result = { + 'value': Decimal(value), + 'currency': currency, + 'change': change, + 'date': datetime.strptime(date, '%d/%m/%Y').date(), + 'ISIN': isin + } + self.data_.update(result) + + +class StockPage(SecurityPage): + def _update_data(self, soup): + title = soup.find_all('span', class_='securityName')[0].text + value = soup.find_all('span', id='Col0Price')[0].text + change = soup.find_all('span', id='Col0PriceDetail')[0].text + change = change.split('|')[1].strip() + date = soup.find_all('p', id='Col0PriceTime')[0].text[6:16] + currency = soup.find_all('p', id='Col0PriceTime')[0].text + currency = re.search(r'\|\s([A-Z]{3,4})\b', currency).group(1) + isin = soup.find_all('td', id='Col0Isin')[0].text + return { + 'name': title, + 'value': Decimal(value), + 'currency': currency, + 'change': change, + 'date': datetime.strptime(date, '%d/%m/%Y').date(), + 'ISIN': isin + } + + +class ETFPage(SecurityPage): + def _update_data(self, soup): + text = soup.find_all('div', class_='snapshotTitleBox')[0].h1.text + self.data_["name"] = text.split('|')[0].strip(), + self.data_["ticker"] = text.split('|')[1].strip() + for keyword in ["Exchange", "ISIN"]: + line = soup.find(text=keyword) + if line is None: + continue + text = line.parent.nextSibling.nextSibling.text + self.data_[keyword] = str(text) + line = soup.find(text="Closing Price") + self.data_["currency"] = line.parent.nextSibling.nextSibling.text[:3] diff --git a/setup.py b/setup.py index e8abd88..aa9a1da 100644 --- a/setup.py +++ b/setup.py @@ -14,7 +14,7 @@ url="https://github.com/tobes/MorningScraper", packages=find_packages(), long_description=long_desc, - install_requires=['beautifulsoup4'], + install_requires=['beautifulsoup4 six'], classifiers=[ "Development Status :: 3 - Alpha", "Topic :: Utilities", From 994c60059660a53527f5f73917d15ba6ebe267e0 Mon Sep 17 00:00:00 2001 From: Alice Wang Date: Thu, 22 Jun 2017 15:25:04 +0200 Subject: [PATCH 3/6] fix ETF name --- morningscraper/security.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/morningscraper/security.py b/morningscraper/security.py index 020807d..d244322 100644 --- a/morningscraper/security.py +++ b/morningscraper/security.py @@ -101,7 +101,7 @@ def _update_data(self, soup): class ETFPage(SecurityPage): def _update_data(self, soup): text = soup.find_all('div', class_='snapshotTitleBox')[0].h1.text - self.data_["name"] = text.split('|')[0].strip(), + self.data_["name"] = text.split('|')[0].strip() self.data_["ticker"] = text.split('|')[1].strip() for keyword in ["Exchange", "ISIN"]: line = soup.find(text=keyword) From c3cbe91efc8c600e384c34f5f7ac0aef8d720541 Mon Sep 17 00:00:00 2001 From: Alice Wang Date: Thu, 22 Jun 2017 17:28:33 +0200 Subject: [PATCH 4/6] put currency to safe fetch --- morningscraper/security.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/morningscraper/security.py b/morningscraper/security.py index d244322..830a0be 100644 --- a/morningscraper/security.py +++ b/morningscraper/security.py @@ -110,4 +110,6 @@ def _update_data(self, soup): text = line.parent.nextSibling.nextSibling.text self.data_[keyword] = str(text) line = soup.find(text="Closing Price") - self.data_["currency"] = line.parent.nextSibling.nextSibling.text[:3] + if line is not None: + self.data_["currency"] = \ + line.parent.nextSibling.nextSibling.text[:3] From 0f28e78bc953875ccad509c36fa13e5f4730ea50 Mon Sep 17 00:00:00 2001 From: Alice Wang Date: Wed, 2 Aug 2017 10:27:36 +0200 Subject: [PATCH 5/6] remove redundant import --- .gitignore | 1 + morningscraper/__init__.py | 22 ++++++++-------------- 2 files changed, 9 insertions(+), 14 deletions(-) diff --git a/.gitignore b/.gitignore index ba74660..e4e18ae 100644 --- a/.gitignore +++ b/.gitignore @@ -1,4 +1,5 @@ # Byte-compiled / optimized / DLL files +.idea __pycache__/ *.py[cod] diff --git a/morningscraper/__init__.py b/morningscraper/__init__.py index ffc4615..4f22bf2 100644 --- a/morningscraper/__init__.py +++ b/morningscraper/__init__.py @@ -1,19 +1,12 @@ import sys -import re - -from decimal import Decimal from datetime import datetime - -from bs4 import BeautifulSoup - from security import make_soup, SecurityPage if sys.version_info[0] == 3: - from urllib.request import urlopen from urllib.parse import quote, urlsplit elif sys.version_info[0] == 2: - from urllib import urlopen, quote + from urllib import quote from urlparse import urlsplit else: raise Exception('Python version 2 or 3 required') @@ -191,9 +184,10 @@ def get_url(url, verbose=False): if __name__ == '__main__': - # search('EWJ', verbose=True) - get_data('EWJ', verbose=True) - get_data('GB00B54RK123', verbose=True) - get_data('LLOY LSE', verbose=True) - get_data('GOOG NASDAQ', verbose=True) - get_data('LU1023728089', verbose=True) + search('EWJ', verbose=True) + get_data('ASHR', verbose=True) + # get_data('GLD ETF', verbose=True) + # get_data('GB00B54RK123', verbose=True) + # get_data('LLOY LSE', verbose=True) + # get_data('GOOG NASDAQ', verbose=True) + # get_data('LU1023728089', verbose=True) From 01d282ba746372601b4f2d6526e8b7971a149db3 Mon Sep 17 00:00:00 2001 From: Eric Schrijver Date: Sun, 13 May 2018 14:44:21 +0200 Subject: [PATCH 6/6] Fix `install_requires` --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index aa9a1da..cdf6901 100644 --- a/setup.py +++ b/setup.py @@ -14,7 +14,7 @@ url="https://github.com/tobes/MorningScraper", packages=find_packages(), long_description=long_desc, - install_requires=['beautifulsoup4 six'], + install_requires=['beautifulsoup4', 'six'], classifiers=[ "Development Status :: 3 - Alpha", "Topic :: Utilities",