Skip to content

Instantly share code, notes, and snippets.

@simonw
Created March 16, 2010 11:27
Show Gist options
  • Select an option

  • Save simonw/333856 to your computer and use it in GitHub Desktop.

Select an option

Save simonw/333856 to your computer and use it in GitHub Desktop.
"""
Generator that crawls an Apache directory listing (automatically using the
?F=0 querystring parameter which causes Apache to return simpler HTML) and
yields file URLs one by one.
"""
import urllib, urlparse, re
always = lambda x: True
# http://stackoverflow.com/questions/1732348#1732454
link_re = re.compile('<li><a href="([^"]+)">([^<]+)</a></li>')
def append_f0(url):
scheme, netloc, path, query, fragment = urlparse.urlsplit(url)
return urlparse.urlunsplit((scheme, netloc, path, 'F=0', ''))
def crawl_apache_dir(url, crawl_dir=always, yield_link=always):
if url.endswith('/'):
if crawl_dir(url):
html = urllib.urlopen(append_f0(url)).read()
paths = link_re.findall(html)
for path, title in paths:
if title.strip() == 'Parent Directory':
continue
new_url = urlparse.urljoin(url, path)
for item in crawl_apache_dir(new_url):
yield item
else:
if yield_link(url):
yield url
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment