Created
March 5, 2017 01:46
-
-
Save zed/6a10c598c378de7d7f568868c95dad4e to your computer and use it in GitHub Desktop.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/usr/bin/env python3 | |
| r"""Parse space-separated key=value pairs. | |
| Format from http://ru.stackoverflow.com/a/635295/23044 | |
| >>> from pprint import pprint | |
| >>> pprint(parse(r'a=1 b="2" c=3.123 d=[1, 2, 3] e={"Ke\n":["h=4"],"Key2":{" ":1}}')) | |
| {'a': 1, | |
| 'b': '2', | |
| 'c': 3.123, | |
| 'd': [1, 2, 3], | |
| 'e': {'Ke\n': ['h=4'], 'Key2': {' ': 1}}} | |
| """ | |
| import codecs | |
| import inspect | |
| from io import BytesIO | |
| import tokenize as T | |
| def error(msg='error'): | |
| token = inspect.currentframe().f_back.f_locals['token'] | |
| raise SyntaxError(msg, ("<string>", | |
| token.start[0], token.start[1], token.line)) | |
| def parse_list(tokens): | |
| """list = '[' ']' | '[' ( value ',' )* value ','? ']'. | |
| ] | |
| +---------------------+ | |
| | v | |
| IN #===# [ #===# V #===# ] #===# | |
| * ----> H 1 H ---> H 2 H ---> H 3 H ---> H 4 H | |
| #===# #===# #===# #===# | |
| ^ , | | |
| +----------+ | |
| """ | |
| L = [] | |
| expect_value, expect_comma = range(2) | |
| state = expect_value | |
| for token in tokens: | |
| if token.type == T.OP and token.string == ']': | |
| return L | |
| elif state == expect_value: | |
| L.append(parse_value(token, tokens)) | |
| state = expect_comma | |
| elif state == expect_comma and token.type == T.OP and token.string == ',': | |
| state = expect_value | |
| else: | |
| error() | |
| error() | |
| def parse_pair(token, tokens): | |
| """pair = key ':' value. | |
| IN #===# K #===# : #===# V #===# | |
| * ----> H 1 H ---> H 2 H ---> H 3 H ---> H 4 H | |
| #===# #===# #===# #===# | |
| """ | |
| expect_colon, expect_value = range(2) | |
| key = parse_value(token, tokens) # NOTE: it accepts unhashable keys | |
| state = expect_colon | |
| for token in tokens: | |
| if state == expect_value: | |
| value = parse_value(token, tokens) | |
| return key, value | |
| elif state == expect_colon and token.type == T.OP and token.string == ':': | |
| state = expect_value | |
| else: | |
| error() | |
| error() | |
| def parse_dict(tokens): | |
| """dict = '{' '}' | '{' ( pair ',' )* pair ','? '}'. | |
| pair = key ':' value; | |
| } | |
| +-------------------------------------------+ | |
| | v | |
| IN #===# { #===# K #===# : #===# V #===# } #===# | |
| * ----> H 1 H ---> H 2 H ---> H 3 H ---> H 4 H ---> H 5 H ---> H 6 H | |
| #===# #===# #===# #===# #===# #===# | |
| ^ , | | |
| +--------------------------------+ | |
| """ | |
| d = {} | |
| expect_pair, expect_comma = range(2) | |
| state = expect_pair | |
| for token in tokens: | |
| if token.type == T.OP and token.string == '}': | |
| return d | |
| elif state == expect_pair: | |
| key, value = parse_pair(token, tokens) | |
| d[key] = value | |
| state = expect_comma | |
| elif state == expect_comma and token.type == T.OP and token.string == ',': | |
| state = expect_pair | |
| else: | |
| error() | |
| error() | |
| def parse_value(token, tokens): | |
| """value = scalar | list | dict.""" | |
| if token.type == T.NUMBER: | |
| try: | |
| return int(token.string) | |
| except ValueError: | |
| return float(token.string) | |
| elif token.type == T.STRING: | |
| return codecs.decode(token.string[1:-1], 'unicode-escape') | |
| elif token.type == T.OP: | |
| if token.string == '[': | |
| return parse_list(tokens) | |
| elif token.string == '{': | |
| return parse_dict(tokens) | |
| error() | |
| def parse_assignments(tokens): | |
| """kv_fsm := (name '=' value)**. | |
| V | |
| +---------------------+ | |
| v | | |
| IN #===# N #===# = #===# | |
| * ----> H 3 H ---> H 1 H ---> H 2 H | |
| #===# #===# #===# | |
| """ | |
| result = {} | |
| expect_name, expect_assignment, expect_value, expect_eof = range(4) | |
| state = expect_name | |
| for token in tokens: | |
| if state == expect_name: | |
| if token.type == T.NAME: | |
| name = token.string | |
| state = expect_assignment | |
| elif token.type == T.ENCODING: | |
| pass | |
| elif token.type == T.ENDMARKER: | |
| state = expect_eof | |
| else: | |
| error() | |
| elif state == expect_value: | |
| result[name] = parse_value(token, tokens) | |
| state = expect_name | |
| elif state == expect_assignment and token.type == T.OP and token.string == '=': | |
| state = expect_value | |
| elif state == expect_eof: | |
| error('unexpected token after EOF') | |
| else: | |
| error() | |
| return result | |
| def parse(data): | |
| if isinstance(data, str): | |
| data = data.encode() | |
| return parse_assignments(T.tokenize(BytesIO(data).readline)) | |
| if __name__ == "__main__": | |
| import doctest | |
| doctest.testmod() |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment