Created
March 16, 2010 11:27
-
-
Save simonw/333856 to your computer and use it in GitHub Desktop.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| """ | |
| Generator that crawls an Apache directory listing (automatically using the | |
| ?F=0 querystring parameter which causes Apache to return simpler HTML) and | |
| yields file URLs one by one. | |
| """ | |
| import urllib, urlparse, re | |
| always = lambda x: True | |
| # http://stackoverflow.com/questions/1732348#1732454 | |
| link_re = re.compile('<li><a href="([^"]+)">([^<]+)</a></li>') | |
| def append_f0(url): | |
| scheme, netloc, path, query, fragment = urlparse.urlsplit(url) | |
| return urlparse.urlunsplit((scheme, netloc, path, 'F=0', '')) | |
| def crawl_apache_dir(url, crawl_dir=always, yield_link=always): | |
| if url.endswith('/'): | |
| if crawl_dir(url): | |
| html = urllib.urlopen(append_f0(url)).read() | |
| paths = link_re.findall(html) | |
| for path, title in paths: | |
| if title.strip() == 'Parent Directory': | |
| continue | |
| new_url = urlparse.urljoin(url, path) | |
| for item in crawl_apache_dir(new_url): | |
| yield item | |
| else: | |
| if yield_link(url): | |
| yield url |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment