Skip to content

Instantly share code, notes, and snippets.

View jonathanoheix's full-sized avatar

jonathanoheix

  • Macif-Mutualité
  • France
View GitHub Profile
pages_urls = []
new_page = "http://books.toscrape.com/catalogue/page-1.html"
while requests.get(new_page).status_code == 200:
pages_urls.append(new_page)
new_page = pages_urls[-1].split("-")[0] + "-" + str(int(pages_urls[-1].split("-")[1].split(".")[0]) + 1) + ".html"
print(str(len(pages_urls)) + " fetched URLs")
print("Some examples:")
result = requests.get("http://books.toscrape.com/catalogue/page-50.html")
print("status code for page 50: " + str(result.status_code))
result = requests.get("http://books.toscrape.com/catalogue/page-51.html")
print("status code for page 51: " + str(result.status_code))
# store all the results into a list
pages_urls = [main_url]
soup = getAndParseURL(pages_urls[0])
# while we get two matches, this means that the webpage contains a 'previous' and a 'next' button
# if there is only one button, this means that we are either on the first page or on the last page
# we stop when we get to the last page
while len(soup.findAll("a", href=re.compile("page"))) == 2 or len(pages_urls) == 1:
import re
categories_urls = [main_url + x.get('href') for x in soup.find_all("a", href=re.compile("catalogue/category/books"))]
categories_urls = categories_urls[1:] # we remove the first one because it corresponds to all the books
print(str(len(categories_urls)) + " fetched categories URLs")
print("Some examples:")
categories_urls[:5]
def getBooksURLs(url):
soup = getAndParseURL(url)
# remove the index.html part of the base url before returning the results
return(["/".join(url.split("/")[:-1]) + "/" + x.div.a.get('href') for x in soup.findAll("article", class_ = "product_pod")])
main_page_products_urls = [x.div.a.get('href') for x in soup.findAll("article", class_ = "product_pod")]
print(str(len(main_page_products_urls)) + " fetched products URLs")
print("One example:")
main_page_products_urls[0]
soup.find("article", class_ = "product_pod").div.a.get('href')
soup.find("article", class_ = "product_pod").div.a
soup.find("article", class_ = "product_pod")
def getAndParseURL(url):
result = requests.get(url)
soup = BeautifulSoup(result.text, 'html.parser')
return(soup)