-
-
Save goeroeku/941106e289338ca335cfc1721b0e9a88 to your computer and use it in GitHub Desktop.
Useful functions for web scraping with Selenium and Chromedriver in Python
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # -*- coding: utf-8 -*- | |
| ## Collection of useful functions for working with Selenium and Chromedriver | |
| ## in Python. These functions are geared toward web scraping. | |
| import time | |
| from selenium import webdriver | |
| from selenium.webdriver.common.by import By | |
| from selenium.webdriver.common.alert import Alert | |
| from selenium.webdriver.support.ui import WebDriverWait | |
| from selenium.webdriver.support import expected_conditions as EC | |
| from selenium.common.exceptions import TimeoutException | |
| from selenium.common.exceptions import NoSuchElementException | |
| from selenium.common.exceptions import NoAlertPresentException | |
| def init_driver(file_path, start_maximized = False, headless = False, | |
| verbose = True, proxy = None, ua = None, | |
| page_load_timeout = None): | |
| ''' | |
| Function to establish instance of chromedriver. | |
| NOTE: The "proxy" arg only works with proxy IP's that do NOT | |
| require a user name and pw. | |
| file_path: str, file path to local chromedriver exe. | |
| start_maximized: logical, launch browser at max size or not. Default False. | |
| headless: logical, launch as headless browser instance or not. Default False. | |
| verbose: logical, display standard chromedriver logging messages or not. | |
| Default True. If False, "--log-level=3" gets passed as an option. | |
| proxy: str or None, proxy IP address to use. Default None. | |
| ua: str or None, user agent to use. Default None. | |
| page_load_timeout: int or None, number of seconds to wait before throwing | |
| TimeoutException when loading a new page. Default None. | |
| ''' | |
| # Input validation. | |
| assert isinstance(file_path, str), "'file_path' must be a str" | |
| assert isinstance(start_maximized, bool), "'start_maximized' must be a bool obj" | |
| assert isinstance(headless, bool), "'headless' must be a bool obj" | |
| assert isinstance(verbose, bool), "'verbose' must be a bool obj" | |
| # Assign options. | |
| opts = [] | |
| if start_maximized: | |
| opts.append("--start-maximized") | |
| if proxy: | |
| assert isinstance(proxy, str), "'proxy' must be a str" | |
| opts.append("--proxy-server=%s" % proxy) | |
| if ua: | |
| assert isinstance(ua, str), "'ua' must be a str" | |
| opts.append("user-agent=%s" % ua) | |
| if headless: | |
| opts.append("--headless") | |
| if len(opts) > 0: | |
| options = webdriver.ChromeOptions() | |
| for opt in opts: | |
| options.add_argument(opt) | |
| # Create driver instance (with options). | |
| driver = webdriver.Chrome(executable_path = file_path, | |
| chrome_options = options) | |
| else: | |
| # Create driver instance (without options). | |
| driver = webdriver.Chrome(executable_path = file_path) | |
| if page_load_timeout: | |
| assert isinstance(page_load_timeout, int), "'page_load_timeout' must be an int" | |
| driver.set_page_load_timeout(page_load_timeout) | |
| driver.wait = WebDriverWait(driver, 10) | |
| return(driver) | |
| def get_page_source(driver): | |
| ''' | |
| # Function for scraping the page_source from the top level of a page, and | |
| # getting the page_source from an inner iframe (if one exists), and getting | |
| # the page source of all frames within the inner iframe (if any exist). | |
| # Returns a dict containing raw page_source html from each of these frame | |
| # levels. | |
| # Related: https://stackoverflow.com/questions/23223018/selenium-get-all-iframes-in-a-page-even-nested-ones | |
| ''' | |
| # Initialize variables. | |
| curr_url = driver.current_url | |
| all_html = {"source_1" : driver.page_source} | |
| all_frames = get_frame_tag_elements(driver) | |
| counter = 2 | |
| if len(all_frames["iframe"]) > 0: | |
| driver.switch_to.frame(all_frames["iframe"][0]) | |
| all_html["source_%s" % counter] = driver.page_source | |
| counter += 1 | |
| all_frames = get_frame_tag_elements(driver) | |
| if len(all_frames["frame"]) > 0: | |
| for idx in range(len(all_frames["frame"])): | |
| driver.switch_to.frame(all_frames["frame"][idx]) | |
| all_html["source_%s" % counter] = driver.page_source | |
| counter += 1 | |
| refocus_to_iframe(driver) | |
| all_frames = get_frame_tag_elements(driver) | |
| elif len(all_frames["frame"]) > 0: | |
| for child_frame in all_frames["frame"]: | |
| driver.switch_to.frame(child_frame) | |
| all_html["source_%s" % counter] = driver.page_source | |
| counter += 1 | |
| all_frames = get_frame_tag_elements(driver) | |
| # If the current url isn't what it was when function began, | |
| # navigate to curr_url. | |
| if driver.current_url != curr_url: | |
| driver.get(curr_url) | |
| return(all_html) | |
| def refocus_to_iframe(driver): | |
| ''' | |
| Put webdriver focus on the top-level iframe within a webpage. | |
| Returns True if refocusing on the top-level iframe is successful, otherwise | |
| returns False. | |
| ''' | |
| # Switch to default content (one level above the inner iframe). | |
| driver.switch_to.default_content() | |
| # Try switching to the inner iframe. | |
| try: | |
| out = driver.wait.until(EC.frame_to_be_available_and_switch_to_it( | |
| (By.TAG_NAME, "iframe"))) | |
| except TimeoutException: | |
| out = False | |
| return(out) | |
| def get_frame_tag_elements(driver): | |
| ''' | |
| Get all visible elements with tag name "iframe" and "frame". Returns a | |
| dict object containing all web elements related to these two tags. | |
| ''' | |
| out = {"iframe" : [], "frame" : []} | |
| out["iframe"] = driver.find_elements_by_tag_name("iframe") | |
| out["frame"] = driver.find_elements_by_tag_name("frame") | |
| return(out) | |
| def clear_pop_up_alert(driver): | |
| ''' | |
| Check for (and clear) an Alert pop-up box at the top of the page. To be | |
| used on alerts that look like this: | |
| http://ptgmedia.pearsoncmg.com/imprint_downloads/informit/learninglabs/9780134173719/graphics/01fig04.jpg | |
| Useful to use this function in conjunction with UnexpectedAlertPresentException. | |
| ''' | |
| try: | |
| Alert(driver).accept() | |
| except NoAlertPresentException: | |
| pass | |
| def close_pop_up_browsers(driver, curr_win_handle): | |
| ''' | |
| Check for (and close) any and all pop-up browsers. This function will | |
| close all browser windows EXCEPT for the window with handle ID | |
| curr_win_handle. | |
| Example use: | |
| curr_win_handle = driver.current_window_handle | |
| close_pop_up_browsers(driver, curr_win_handle) | |
| ''' | |
| if len(driver.window_handles) > 1: | |
| open_handles = driver.window_handles | |
| while len(open_handles) > 1: | |
| focus = [n for n in open_handles if n != curr_win_handle] | |
| driver.switch_to_window(focus[0]) | |
| driver.close() | |
| open_handles = driver.window_handles | |
| driver.switch_to_window(curr_win_handle) | |
| def pause_if_captcha(driver, captcha_class_name): | |
| ''' | |
| Pause scraping if captcha page is up, print a message to console | |
| instructing the user to manually satisfy the captcha. Function | |
| will pause forever in an infinite loop until the captcha is gone. It checks | |
| for the presence of the captcha every 30 seconds while in the loop. | |
| ''' | |
| if not check_for_captcha(driver, captcha_class_name): | |
| return(None) | |
| else: | |
| print("captcha!\nManually satisfy the captcha requirements.\n"\ | |
| "Code will resume once the captcha is gone.") | |
| captcha = True | |
| while captcha: | |
| time.sleep(30) | |
| captcha = check_for_captcha(driver, captcha_class_name) | |
| def check_for_captcha(driver, captcha_class_name): | |
| ''' | |
| Check to see if the page is currently stuck on a captcha page. | |
| ''' | |
| try: | |
| driver.find_element_by_class_name(captcha_class_name) | |
| except NoSuchElementException: | |
| return(False) | |
| return(True) |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment