Last active
May 9, 2026 13:20
-
-
Save the-bokya/53ee2efce93e3db96682f2ad4e38fed8 to your computer and use it in GitHub Desktop.
Test the "you'll always eventually reach 'Philosophy' if you keep on clicking the first link" hypothesis
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| import requests | |
| from bs4 import BeautifulSoup as bs4 | |
| import time | |
| from itertools import pairwise | |
| import json | |
| import re | |
| FILE_NAME = "links.gv" | |
| ITERATIONS = 50 | |
| class Wiki: | |
| def __init__(self, topic): | |
| self.headers = { | |
| 'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64; rv:150.0) Gecko/20100101 Firefox/150.0', | |
| 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8', | |
| 'Accept-Language': 'en-US,en;q=0.9', | |
| } | |
| self.url = f"https://en.wikipedia.org/wiki/{topic}" | |
| def get(self): | |
| resp = requests.get(self.url, headers=self.headers) | |
| self.resp = resp | |
| return resp.content | |
| def get_content(self): | |
| soup = bs4(self.get(), "html.parser") | |
| # Main article content | |
| content = soup.find("div", id="bodyContent") | |
| for table in content.find_all("table"): | |
| table.decompose() | |
| content = str(content) | |
| re.sub(r'\([^()]*:[^()]*\)', '', content) | |
| content = re.sub( | |
| r'\([^()]*:[^()]*(?:\([^()]*\)[^()]*)*\)', | |
| '', | |
| content | |
| ) | |
| return bs4(content, "html.parser") | |
| def get_first_link(self): | |
| for p in self.get_content().find_all("p"): | |
| first_link_with_id = p.find("a", href=True, title=True, attrs={"role": False}) | |
| if first_link_with_id: | |
| if "Help:" in first_link_with_id.get("href"): | |
| continue | |
| if "Wikipedia:" in first_link_with_id.get("href"): | |
| continue | |
| if "File:" in first_link_with_id.get("href"): | |
| continue | |
| return first_link_with_id | |
| with open(FILE_NAME, "a") as f: | |
| f.write("digraph {\n") | |
| prevs = {'"Philosophy"'} | |
| for i in range(ITERATIONS): | |
| title = "Special:Random" | |
| # title = "Ujari" | |
| # title = "Timeline_of_Italian_history" | |
| links = [] | |
| skip = False | |
| while True: | |
| if json.dumps(title) in prevs or json.dumps(title) in links: | |
| links.append(json.dumps(title)) | |
| break | |
| x = Wiki(title) | |
| # print(x.get_content()) | |
| try: | |
| url = x.get_first_link().get("href") | |
| except: | |
| try: | |
| url = x.get_first_link().get("href") | |
| except: | |
| skip = True | |
| break | |
| link = x.resp.url.split("/wiki/")[-1] | |
| link = link.split("#")[0] | |
| link = json.dumps(link) | |
| print(link) | |
| links.append(link) | |
| if not url: | |
| break | |
| url = url.split("/wiki/") | |
| url = url[-1] | |
| title = url | |
| time.sleep(1) | |
| if skip: | |
| continue | |
| prevs.update(set(links)) | |
| print(prevs) | |
| pairs = list(pairwise(links)) | |
| s = "\n".join(["->".join(i) for i in pairs]) | |
| s += "\n" | |
| with open(FILE_NAME, "a") as f: | |
| f.write(s) | |
| with open(FILE_NAME, "a") as f: | |
| f.write("}") |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment