Last active
June 28, 2017 18:29
-
-
Save simonw/f67ca9b0a71d68be6700ad56875e5d7f to your computer and use it in GitHub Desktop.
I use Atom and the flake8 plugin, and I was getting a barrage of flake8 UnicodeDecode errors. I couldn't figure out why, so I wrote this script to recursively scan all of the files in a directory and show me which ones had unexpected non-ascii characters in them based on a whitelist. This helped me track down the file causing the problem.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| import os | |
| import re | |
| skip_paths = ( | |
| '.git', | |
| '.DS_Store', | |
| '.bz2', | |
| ) | |
| bad_chars_re = re.compile( | |
| r'[^a-zA-Z0-9_\(\),\.\s"%\':;=@\[\]`\{\}\-!#\*\+/<>\?\$&\\\^\|~]' | |
| ) | |
| def recursively_find_bad_characters(entrypath): | |
| allbaduns = set() | |
| for (path, dirs, files) in os.walk(entrypath): | |
| for filename in files: | |
| filepath = os.path.join(path, filename) | |
| skipmatches = [skip for skip in skip_paths if skip in filepath] | |
| if skipmatches: | |
| continue | |
| s = open(filepath).read() | |
| baduns = set(bad_chars_re.findall(s)) | |
| allbaduns.update(baduns) | |
| if baduns: | |
| print filepath, baduns | |
| for badun in baduns: | |
| print badun, repr(badun) | |
| index = s.index(badun) | |
| print index, s[max(0, index - 10):index + 10] | |
| print "on line %d" % (s[:index].count('\n') + 1) | |
| if allbaduns: | |
| print 'allbaduns = %r' % allbaduns | |
| if __name__ == '__main__': | |
| import sys | |
| for arg in sys.argv[1:]: | |
| recursively_find_bad_characters(arg) |
Author
Author
Usage:
python recursively_find_bad_characters.py ~/path/to/my/project
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment
The problem turned out to be in the .flake8 configuration file itself, which had a rogue curly apostrophe in a comment.