diff --git a/scripts/download.py b/scripts/download.py index c8b7b9e..f5eb974 100644 --- a/scripts/download.py +++ b/scripts/download.py @@ -1,31 +1,17 @@ #! /usr/bin/env python3 +import argparse +import configparser +import copy +import logging import os +import pickle import re import time -import logging -import argparse import html2text from bs4 import BeautifulSoup -from selenium import webdriver +from selenium.webdriver import Firefox from selenium.webdriver.firefox.options import Options -from selenium.common.exceptions import WebDriverException - -cat_counts = {'a': 155, 'b': 95, 't': 27} - -# the number of problems per page -number_per_page = 100 - -urlidx = { - 'b': "994805260223102976", - 'a': "994805342720868352", - 't': "994805148990160896" -} - -# setting logging -logging.basicConfig( - format='[%(levelname)s] %(filename)s:%(lineno)d: %(message)s', - level=logging.INFO) def process_katex(soup): @@ -59,50 +45,44 @@ def extract_sample_IO(soup): return samplt_input_text, samplt_output_text -class PATDownloader: +class PATDownloader(Firefox): """ Automate browser to visit pintia.cn and download problem texts. The reason for the tool is the content on the website is dynamically generated, so simply get the webpage with requests is not working. """ def __init__(self, force): - self._phantom_is_setup = False self._force = force - self._base_url = "https://pintia.cn" - self._problem_sets_url = self._base_url + "/problem-sets" + self.base_url = "https://pintia.cn" + self.problem_sets_url = self.base_url + "/problem-sets" + self.options = Options() + self.options.headless = True + self.options.add_argument("--safe-mode") - # phantom_setup - options = Options() - options.headless = True - try: - self._phantom_browser = webdriver.Firefox( - options=options, - service_log_path=os.devnull - ) - logging.debug("Starting phantom firefox driver") - except WebDriverException as e: - print(e) - logging.error("Starting phantom firefox driver failed") - exit(1) + super().__init__(options=self.options, service_log_path=os.devnull) + self.implicitly_wait(10) - def __del__(self): - try: - self._phantom_browser.quit() - except AttributeError: - pass + self.get(self.base_url) + for cookie in self.get_cookies(): + self.add_cookie(cookie) - def _phantom_parse_soup(self, url): - self._phantom_browser.get(url) - try: - html = self._phantom_browser.page_source - except WebDriverException as e: - print(e) - logging.warning("you just have to run the script again, " - "something unfortunate happened.") - exit(1) - # TODO: check return here - soup = BeautifulSoup(html, 'html.parser') - return soup + def get_default_profile(self): + parser = configparser.ConfigParser() + parser.read(os.path.join(os.getenv("HOME"), ".mozilla", "firefox", "profiles.ini")) + profile = parser.get(parser.sections()[0], "Default") + return os.path.join(os.getenv("HOME"), ".mozilla", "firefox", profile) + + def get_cookies(self): + if os.path.exists("cookies.pkl"): + return pickle.load(open("cookies.pkl", "rb")) + opts = copy.deepcopy(self.options) + opts.profile = self.get_default_profile() + driver = Firefox(options=opts, service_log_path=os.devnull) + driver.get(self.base_url) + driver.get_cookies() + cookies = driver.get_cookies() + pickle.dump(cookies, open("cookies.pkl", "wb")) + return cookies def _parse_catatory(self, cat): problem_list = [] @@ -110,48 +90,37 @@ class PATDownloader: logging.info("retrieving infomation for category {c}".format(c=cat)) for page in range(cat_counts[cat] // number_per_page + 1): category_url = "{baseurl}/{ID}/problems/type/7?page={page}".format( - baseurl=self._problem_sets_url, + baseurl=self.problem_sets_url, ID=urlidx[cat], - page=page) + page=page + ) - logging.debug('requesting page \'%s\'', category_url) - soup = self._phantom_parse_soup(category_url) - table = soup.find('tbody') - if table is None: - logging.warning( - 'requesting page \'%s\' failed, will retry ' - '(table returned None)', - category_url - ) - return None - rows = table.find_all('tr') + logging.info('requesting page \'%s\'', category_url) + self.get(category_url) + table = self.find_element_by_tag_name('tbody') + rows = table.find_elements_by_tag_name('tr') for row in rows: - tdlist = row.find_all('td') - link = tdlist[2].find('a') + check, label, title, score, rate = row.find_elements_by_tag_name('td') + link = title.find_element_by_tag_name('a') problem_list.append({ - 'index': tdlist[1].contents[0], - 'title': "{content} ({score})".format( - content=link.contents[0], - score=tdlist[3].contents[0]), - 'link': self._base_url + link['href'] + 'index': label.text, + 'title': f"{link.text} ({score.text})", + 'link': link.get_property('href') }) return problem_list def _parse_problem(self, url): - logging.debug('requesting page \'%s\'', url) - - soup = self._phantom_parse_soup(url) - - pc_divs = soup.find_all('div', 'rendered-markdown') - if len(pc_divs) < 2: - logging.error("Not enough div with class \'rendered-markdown\'" + - "for url: {}, got {}".format(url, len(pc_divs))) - return None, None, None + logging.info('requesting page \'%s\'', url) + self.get(url) + # the first find is only to implicit wait until loaded + self.find_elements_by_id('input-specification') + pc_divs = self.find_elements_by_class_name('rendered-markdown') pc_div = pc_divs[1] - content_soup = process_katex(pc_div) + soup = BeautifulSoup(pc_div.get_attribute("innerHTML"), 'html.parser') + content_soup = process_katex(soup) for tag in content_soup.find_all('code'): if tag.parent.name == 'pre': tag.parent.replace_with(tag.wrap(soup.new_tag("pre"))) @@ -189,7 +158,6 @@ class PATDownloader: url_list = self._parse_catatory(c) if url_list is None: logging.info("retrying") - time.sleep(1) # find the corresponding url url_index = next((url for url in url_list @@ -198,12 +166,7 @@ class PATDownloader: # download if url_index: logging.info("downloading %s", textfile) - pc = None - while pc is None: - pc, si, so = self._parse_problem(url_index['link']) - if pc is None: - logging.info("retrying") - time.sleep(1) + pc, si, so = self._parse_problem(url_index['link']) logging.debug("saving %s", textfile) with open(textfile, 'w') as f: @@ -211,10 +174,9 @@ class PATDownloader: url_index['title'], pc)) # There might be more than one samples for i in range(len(si)): - with open(si_file.format(i + 1), 'w') as f: - f.write(si[i]) - with open(so_file.format(i + 1), 'w') as f: - f.write(so[i]) + open(si_file.format(i + 1), 'w').write(si[i]) + open(so_file.format(i + 1), 'w').write(so[i]) + time.sleep(2) else: logging.error("Index %s%s not available", c, i) @@ -233,6 +195,20 @@ def get_parser(): return parser +cat_counts = {'a': 155, 'b': 95, 't': 27} + +# the number of problems per page +number_per_page = 100 + +urlidx = { + 'b': "994805260223102976", + 'a': "994805342720868352", + 't': "994805148990160896" +} + +# setting logging +logging.basicConfig(level=logging.INFO) + if __name__ == "__main__": # parse arguments parser = get_parser() @@ -263,3 +239,6 @@ if __name__ == "__main__": dl.download(dlIndexes) except KeyboardInterrupt: logging.info("exiting...") + finally: + dl.close() + dl.quit()