Last active
February 8, 2019 07:00
-
-
Save level120/fddf96c0dde7beafc4810231a5f29808 to your computer and use it in GitHub Desktop.
python3 웹 크롤링 결과를 json 저장, selenium + webdriver(크롬-headless, 파폭-headless) 사용 -> excel 저장
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| from selenium import webdriver | |
| def crawling(url, browser, wl, fn): | |
| print("set url : " + url + ", browser : " + browser) | |
| if browser is "Chrome": | |
| options = webdriver.ChromeOptions() | |
| options.add_argument('headless') | |
| options.add_argument('window-size=1920x1080') | |
| options.add_argument("disable-gpu") | |
| chrome_driver = wl | |
| driver = webdriver.Chrome(executable_path=chrome_driver, chrome_options=options) | |
| elif browser is "Firefox": | |
| from selenium.webdriver.firefox.options import Options | |
| options = Options() | |
| options.headless = True | |
| firefox_driver = wl | |
| driver = webdriver.Firefox(executable_path=firefox_driver, options=options) | |
| else: | |
| return False | |
| print("webdriver's ready") | |
| driver.implicitly_wait(3) | |
| # 크롤링할 사이트 호출 | |
| print("getting url site") | |
| driver.get(url) | |
| print("parsing data") | |
| # 크롤링 위치 변경하려면 여기수정 | |
| img = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_img > img") | |
| brand = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_info > div.brand") | |
| name = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_info > div.name") | |
| ref_no = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_info > div.ref_no > span") | |
| price_origin = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_info > div.price > span.sale") | |
| price_final = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_info > div.price > span.won") | |
| # Json 영역 | |
| import json | |
| from collections import OrderedDict | |
| LENGTH = min(len(img), len(brand), len(name), len(ref_no), len(price_origin), len(price_final)) | |
| products = OrderedDict() | |
| print("processing data to json") | |
| # json 형태 저장하는걸 바꾸려면 여기수정 | |
| for idx in range(0, LENGTH): | |
| products['no_' + str(idx + 1)] = { | |
| 'img': img[idx].get_attribute("data-original"), | |
| 'brand': brand[idx].text, | |
| 'name': name[idx].text, | |
| 'ref_no': ref_no[idx].text, | |
| 'price': { | |
| 'sale': price_origin[idx].text, | |
| 'won': price_final[idx].text | |
| } | |
| } | |
| # json 저장 | |
| with open(fn, 'w', encoding='utf-8') as f: | |
| json.dump(products, f, ensure_ascii=False, indent="\t") | |
| # 브라우저 종료, 웹 드라이버 종료 | |
| print("This job is finished and close the web browser") | |
| driver.quit() | |
| return True | |
| def setup(): | |
| browser = "Firefox" # Chrome 혹은 Firefox 선택 입력 | |
| url = r"http://www.shilladfs.com/estore/kr/ko/Skin-Care/Basic-Skin-Care/Skin-Toner/c/79" | |
| json_filename = "product.json" # 저장할 json 파일명 | |
| webdriver_location = r"D:/temp/phantomjs/geckodriver.exe" # 웹드라이버 위치(exe 파일까지 작성) | |
| if not crawling(url, browser, webdriver_location, json_filename): | |
| print("Failed to crawling.") | |
| def exportToExcel(): | |
| import pandas | |
| pandas.read_json('product.json', encoding='UTF8', orient='index').to_excel('product.xlsx', encoding='UTF8') | |
| if __name__ == "__main__": | |
| setup() | |
| exportToExcel() |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| from selenium import webdriver | |
| url = r"http://www.shilladfs.com/estore/kr/ko/Skin-Care/Basic-Skin-Care/Skin-Toner/c/79" | |
| print("set url : " + url) | |
| # Chrome Headless 전용 옵션, Firefox 사용시 모두 주석 | |
| options = webdriver.ChromeOptions() | |
| options.add_argument('headless') | |
| options.add_argument('window-size=1920x1080') | |
| options.add_argument("disable-gpu") | |
| # Chrome 드라이버 생성(둘 중 하나만 켤것) | |
| chrome_driver = r"D:/temp/phantomjs/chromedriver.exe" | |
| driver = webdriver.Chrome(executable_path=chrome_driver, chrome_options=options) | |
| # Firefox 드라이버 생성(둘 중 하나만 켤것) | |
| #firefox_driver = r"D:/temp/phantomjs/geckodriver.exe" | |
| #driver = webdriver.Firefox(executable_path=firefox_driver) | |
| driver.implicitly_wait(3) | |
| print("webdriver's ready") | |
| # 크롤링할 사이트 호출 | |
| print("getting url site") | |
| driver.get(url) | |
| print("parsing data") | |
| #context = driver.find_elements_by_class_name("facet-product-list") | |
| img = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_img > img") | |
| brand = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_info > div.brand") | |
| name = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_info > div.name") | |
| ref_no = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_info > div.ref_no > span") | |
| price_origin = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_info > div.price > span.sale") | |
| price_final = driver.find_elements_by_css_selector("div.product_off_wrap > div.product_off > div.pr_info > div.price > span.won") | |
| # Json | |
| import json | |
| from collections import OrderedDict | |
| LENGTH = min(len(img), len(brand), len(name), len(ref_no), len(price_origin), len(price_final)) | |
| products = OrderedDict() | |
| print("processing data to json") | |
| for idx in range(0, LENGTH): | |
| products['no_' + str(idx + 1)] = { | |
| 'img': img[idx].get_attribute("data-original"), | |
| 'brand': brand[idx].text, | |
| 'name': name[idx].text, | |
| 'ref_no': ref_no[idx].text, | |
| 'price': { | |
| 'sale': price_origin[idx].text, | |
| 'won': price_final[idx].text | |
| } | |
| } | |
| # json 저장 | |
| with open('product.json', 'w', encoding='utf-8') as f: | |
| json.dump(products, f, ensure_ascii=False, indent="\t") | |
| # 브라우저 종료, 웹 드라이버 종료 | |
| print("This job is finished and close the web browser") | |
| driver.quit() |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment