Skip to content

Instantly share code, notes, and snippets.

View simonw's full-sized avatar

Simon Willison simonw

View GitHub Profile
import httplib2, simplejson
h = httplib2.Http()
h.request("http://localhost:9173/jobs", "POST", urllib.urlencode({
'job': simplejson.dumps({
'action': 'word_count',
'inputs': [
'http://www.gutenberg.org/dirs/etext97/1ws3010.txt',
'http://www.gutenberg.org/dirs/etext99/1ws3511.txt',
'http://www.gutenberg.org/dirs/etext97/1ws2510.txt',
# Python logging handler for stashing most recent 100 log items in a redis list
import logging
import redis
r = redis.Redis()
class RedisLogHandler(logging.Handler):
def emit(self, record):
r.push('python-log', unicode(record).encode("utf8"), tail=True)
@simonw
simonw / is_hard.py
Created November 8, 2009 09:06
Python script for telling if two files are hard links to the same thing
#!/usr/bin/env python
"is_hard.py - tell if two files are hard links to the same thing"
import os, sys, stat
def is_hard_link(filename, other):
s1 = os.stat(filename)
s2 = os.stat(other)
return (s1[stat.ST_INO], s1[stat.ST_DEV]) == \
(s2[stat.ST_INO], s2[stat.ST_DEV])
174.129.50.201 - - [28/Nov/2009:18:20:45 -0600] "HEAD / HTTP/1.0" 200 - "-" "@hourlypress"
70.32.121.67 - - [29/Nov/2009:00:20:46 +0000] "GET / HTTP/1.0" 200 8826 "-" "Java/1.6.0"
66.249.71.186 - - [28/Nov/2009:18:20:48 -0600] "GET / HTTP/1.0" 200 3110 "-" "Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)"
208.74.66.39 - - [29/Nov/2009:00:20:49 +0000] "GET / HTTP/1.0" 200 8832 "-" "Mozilla/5.0 (compatible; Butterfly/1.0; +http://labs.topsy.com/butterfly.html) Gecko/2009032608 Firefox/3.0.8"
174.129.107.161 - - [28/Nov/2009:18:20:50 -0600] "HEAD / HTTP/1.0" 200 - "-" "PycURL/7.18.2"
174.129.89.199 - - [29/Nov/2009:00:20:51 +0000] "GET / HTTP/1.0" 200 8797 "-" "Python-urllib/2.5"
174.129.58.57 - - [28/Nov/2009:18:20:53 -0600] "HEAD / HTTP/1.0" 200 - "-" "Mozilla/4.0 (compatible; MSIE 5.01; Windows NT 5.0)"
89.151.84.35 - - [29/Nov/2009:00:20:53 +0000] "GET / HTTP/1.0" 200 8912 "-" "Mozilla/5.0 (compatible; MSIE 6.0b; Windows NT 5.0) Gecko/2009011913 Firefox/3.0.6 TweetmemeBot"
89.151.84.
def all_from_view(db, view_name, batch_size=100):
kwargs = {
'limit': batch_size,
'include_docs': True,
}
startkey_docid = None
last_served_id = None
while True:
if startkey_docid is not None:
kwargs['startkey_docid'] = startkey_docid
@simonw
simonw / natread.py
Created December 7, 2009 21:03
A cow that can read Nat's mind
#!/usr/bin/python
import json, urllib, os, textwrap
user = 'natbat'
url = "http://twitter.com/statuses/user_timeline/%s.json"
pipe = os.popen("cowthink -n", 'w')
pipe.write('\n'.join(textwrap.wrap(
json.load(urllib.urlopen(url % user))[0]['text'])
))
# -*- coding: utf-8 -*-
# Copy of http://dpaste.com/136418/
# Based on http://github.com/simonw/django-signed
# All cookie related code is untested.
import base64
import hmac
import struct
import time
from django.conf import settings
# Depends on the OS X "say" command
import time, datetime, subprocess, math, sys
def say(s):
subprocess.call(['say', str(s)])
def seconds_until(dt):
return time.mktime(dt.timetuple()) - time.time()
import MySQLdb;
from datetime import datetime
import re
import mailbox
import sys
name = "enation"
threading=False
if len(sys.argv) > 1:
from xml.etree import ElementTree as ET
import urllib
dbpedia_endpoint = "http://dbpedia.org/sparql?"
ns = {'ns': '{http://www.w3.org/2005/sparql-results#}'}
result_xpath ='%(ns)sresults/%(ns)sresult/%(ns)sbinding/%(ns)sliteral' % ns
def wikipedia_abstract_from_dbpedia(dbpedia_key):
sparql = """
SELECT ?abstract