Last active
April 8, 2023 17:39
-
-
Save zed/9616954 to your computer and use it in GitHub Desktop.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/usr/bin/env python | |
| """Remove consecutive duplicate characters unless inside a known word. | |
| Usage: | |
| $ ./replace-elongated-words input_file | |
| http://stackoverflow.com/questions/22471255/replace-mutiple-occurance-of-characters-as-well-as-punctuation-in-strings-of-fil | |
| """ | |
| import fileinput | |
| import re | |
| import sys | |
| from itertools import groupby, product | |
| import enchant # $ pip install pyenchant | |
| def remove_consecutive_dups(s): | |
| return re.sub(r'(?i)(.)\1+', r'\1', s) | |
| def all_consecutive_duplicates_edits(word, max_repeat=float('inf')): | |
| chars = [[c*i for i in range(min(len(list(dups)), max_repeat), 0, -1)] | |
| for c, dups in groupby(word)] | |
| return map(''.join, product(*chars)) | |
| words = enchant.Dict("en") | |
| is_known_word = words.check | |
| for line in fileinput.input(inplace=False): | |
| #NOTE: unnecessary work, optimize if needed | |
| output = [next((e for e in all_consecutive_duplicates_edits(s) | |
| if e and is_known_word(e)), remove_consecutive_dups(s)) | |
| for s in re.split(r'(\W+)', line)] | |
| sys.stdout.write(''.join(output)) |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment