Skip to content

Instantly share code, notes, and snippets.

@rtanglao
Created April 19, 2010 20:41
Show Gist options
  • Save rtanglao/371573 to your computer and use it in GitHub Desktop.
Save rtanglao/371573 to your computer and use it in GitHub Desktop.
search.py
# forked version of http://github.com/davedash/SUMO-issues/blob/master/search.py
# Will put this file in a real git repo instead of a gist, "real soon now"
STOPWORDS = ('a', 'about', 'above', 'across', 'after', 'afterwards', 'again',
'against', 'all', 'almost', 'alone', 'along', 'already', 'also', 'although',
'always', 'am', 'among', 'amongst', 'amoungst', 'amount', 'an', 'and',
'another', 'any', 'anyhow', 'anyone', 'anything', 'anyway', 'anywhere', 'are',
'around', 'as', 'at', 'back', 'be', 'became', 'because', 'become', 'becomes',
'becoming', 'been', 'before', 'beforehand', 'behind', 'being', 'below',
'beside', 'besides', 'between', 'beyond', 'bill', 'both', 'bottom', 'but',
'by', 'call', 'can', 'cannot', 'cant', 'co', 'computer', 'con', 'could',
'couldnt', 'cry', 'de', 'describe', 'detail', 'do', 'done', 'down', 'due',
'during', 'each', 'eg', 'eight', 'either', 'eleven', 'else', 'elsewhere',
'empty', 'enough', 'etc', 'even', 'ever', 'every', 'everyone', 'everything',
'everywhere', 'except', 'few', 'fifteen', 'fify', 'fill', 'find', 'fire',
'firefox', 'first', 'five', 'for', 'former', 'formerly', 'forty', 'found',
'four', 'from', 'front', 'full', 'further', 'get', 'give', 'go', 'had', 'has',
'hasnt', 'have', 'he', 'hence', 'her', 'here', 'hereafter', 'hereby', 'herein',
'hereupon', 'hers', 'herse', 'him', 'himse', 'his', 'how', 'however',
'hundred', 'i', 'ie', 'if', 'in', 'inc', 'indeed', 'interest', 'into', 'is',
'it', 'its', 'itse', 'keep', 'last', 'latter', 'latterly', 'least', 'less',
'ltd', 'made', 'many', 'may', 'me', 'meanwhile', 'might', 'mill', 'mine',
'more', 'moreover', 'most', 'mostly', 'move', 'much', 'must', 'my', 'myse',
'name', 'namely', 'neither', 'never', 'nevertheless', 'next', 'nine', 'no',
'nobody', 'none', 'noone', 'nor', 'not', 'nothing', 'now', 'nowhere', 'of',
'off', 'often', 'on', 'once', 'one', 'only', 'onto', 'or', 'other', 'others',
'otherwise', 'our', 'ours', 'ourselves', 'out', 'over', 'own', 'part', 'per',
'perhaps', 'please', 'put', 'rather', 're', 'same', 'see', 'seem', 'seemed',
'seeming', 'seems', 'serious', 'several', 'she', 'should', 'show', 'side',
'since', 'sincere', 'six', 'sixty', 'so', 'some', 'somehow', 'someone',
'something', 'sometime', 'sometimes', 'somewhere', 'still', 'such', 'system',
'take', 'ten', 'than', 'that', 'the', 'their', 'them', 'themselves', 'then',
'thence', 'there', 'thereafter', 'thereby', 'therefore', 'therein',
'thereupon', 'these', 'they', 'thick', 'thin', 'third', 'this', 'those',
'though', 'three', 'through', 'throughout', 'thru', 'thus', 'to', 'together',
'too', 'top', 'toward', 'towards', 'twelve', 'twenty', 'two', 'un', 'under',
'until', 'up', 'upon', 'us', 'very', 'via', 'was', 'we', 'well', 'were',
'what', 'whatever', 'when', 'whence', 'whenever', 'where', 'whereafter',
'whereas', 'whereby', 'wherein', 'whereupon', 'wherever', 'whether', 'which',
'while', 'whither', 'who', 'whoever', 'whole', 'whom', 'whose', 'why', 'will',
'with', 'within', 'without', 'would', 'yet', 'you', 'your', 'yours',
'yourself', 'yourselves',
"thunderbird", "email", "e-mail", "mail", "thunderbird3", "tbird", "tbird3", "tb", "emails", "mails", "e-mails", "tb3", "tb2", "support", "help", "error", "support", "please", "new", "ok", "message", "messages", "thanks", "got", "page", "two", "etc", "etc", "e.g.", "i.e", "fix", "computer", "seems", "right", "like", "fine", "also", "first", "fix", "worked", "something", "trying", "even", "much", "every", 'client', "different", "may", "since", "default", "problem", "many", "hi", "mozilla", "bug", "feature", "already", "unable", "using", "use", "one", "anyone", "however", "anything", "wrong", "now", "think", "found", "see", "still", "want", "might", "answer", "going", "question", "else", "used", "user", "appears", "line", "problems", "questions", "works", "thank", "works", "really", "great", "good", "well", "everything", "mac", "lot", "nothing", "nothing", "correct", "firefox", "people", "just", "get", "set" )
STOPWORDS = dict(zip(STOPWORDS, STOPWORDS))
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment