Created
February 9, 2026 19:47
-
-
Save Vizonex/fb39bb9439df050fddeb06edd345e184 to your computer and use it in GitHub Desktop.
MultiDict Optimization workaround that focuses on making it possible to optimize iterations without hindering the performance of other types of objects or the multidict python backup This idea is still under concept so just a warning!
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # here's how we could use our custom module we made to eliminate some slowness with other things. | |
| # example show is aiohttp/_http_writer.pyx with a few upgrades but also our own module being injected on in. | |
| from cpython.bytes cimport PyBytes_FromStringAndSize | |
| from cpython.exc cimport PyErr_NoMemory, PyErr_SetString, PyErr_SetObject | |
| from cpython.mem cimport PyMem_Free, PyMem_Malloc, PyMem_Realloc | |
| from cpython.object cimport PyObject_Str | |
| from libc.stdint cimport uint8_t, uint64_t | |
| from libc.string cimport memcpy | |
| from .mdcs cimport AnyDict_items | |
| from multidict import istr | |
| DEF BUF_SIZE = 16 * 1024 # 16KiB | |
| cdef object _istr = istr | |
| # ----------------- writer --------------------------- | |
| cdef struct Writer: | |
| char *buf | |
| Py_ssize_t size | |
| Py_ssize_t pos | |
| bint heap_allocated | |
| cdef inline void _init_writer(Writer* writer, char *buf): | |
| writer.buf = buf | |
| writer.size = BUF_SIZE | |
| writer.pos = 0 | |
| writer.heap_allocated = 0 | |
| cdef inline void _release_writer(Writer* writer): | |
| if writer.heap_allocated: | |
| PyMem_Free(writer.buf) | |
| cdef inline int _write_byte(Writer* writer, uint8_t ch): | |
| cdef char * buf | |
| cdef Py_ssize_t size | |
| if writer.pos == writer.size: | |
| # reallocate | |
| size = writer.size + BUF_SIZE | |
| if not writer.heap_allocated: | |
| buf = <char*>PyMem_Malloc(size) | |
| if buf == NULL: | |
| PyErr_NoMemory() | |
| return -1 | |
| memcpy(buf, writer.buf, writer.size) | |
| else: | |
| buf = <char*>PyMem_Realloc(writer.buf, size) | |
| if buf == NULL: | |
| PyErr_NoMemory() | |
| return -1 | |
| writer.buf = buf | |
| writer.size = size | |
| writer.heap_allocated = 1 | |
| writer.buf[writer.pos] = <char>ch | |
| writer.pos += 1 | |
| return 0 | |
| cdef inline int _write_utf8(Writer* writer, Py_UCS4 symbol): | |
| cdef uint64_t utf = <uint64_t> symbol | |
| if utf < 0x80: | |
| return _write_byte(writer, <uint8_t>utf) | |
| elif utf < 0x800: | |
| if _write_byte(writer, <uint8_t>(0xc0 | (utf >> 6))) < 0: | |
| return -1 | |
| return _write_byte(writer, <uint8_t>(0x80 | (utf & 0x3f))) | |
| elif 0xD800 <= utf <= 0xDFFF: | |
| # surogate pair, ignored | |
| return 0 | |
| elif utf < 0x10000: | |
| if _write_byte(writer, <uint8_t>(0xe0 | (utf >> 12))) < 0: | |
| return -1 | |
| if _write_byte(writer, <uint8_t>(0x80 | ((utf >> 6) & 0x3f))) < 0: | |
| return -1 | |
| return _write_byte(writer, <uint8_t>(0x80 | (utf & 0x3f))) | |
| elif utf > 0x10FFFF: | |
| # symbol is too large | |
| return 0 | |
| else: | |
| if _write_byte(writer, <uint8_t>(0xf0 | (utf >> 18))) < 0: | |
| return -1 | |
| if _write_byte(writer, | |
| <uint8_t>(0x80 | ((utf >> 12) & 0x3f))) < 0: | |
| return -1 | |
| if _write_byte(writer, | |
| <uint8_t>(0x80 | ((utf >> 6) & 0x3f))) < 0: | |
| return -1 | |
| return _write_byte(writer, <uint8_t>(0x80 | (utf & 0x3f))) | |
| cdef inline int _write_str(Writer* writer, str s): | |
| cdef Py_UCS4 ch | |
| for ch in s: | |
| if _write_utf8(writer, ch) < 0: | |
| return -1 | |
| cdef inline int _write_str_raise_on_nlcr(Writer* writer, object s): | |
| cdef Py_UCS4 ch | |
| cdef str out_str | |
| if type(s) is str: | |
| out_str = <str>s | |
| elif type(s) is _istr: | |
| out_str = PyObject_Str(s) | |
| elif not isinstance(s, str): | |
| PyErr_SetObject(TypeError, "Cannot serialize non-str key {!r}".format(s)) | |
| return -1 | |
| else: | |
| out_str = str(s) | |
| for ch in out_str: | |
| if ch == 0x0D or ch == 0x0A: | |
| PyErr_SetString(ValueError, | |
| "Newline or carriage return detected in headers. " | |
| "Potential header injection attack." | |
| ) | |
| return -1 | |
| if _write_utf8(writer, ch) < 0: | |
| return -1 | |
| return 0 | |
| # --------------- _serialize_headers ---------------------- | |
| cdef int _serlize_header(void* hook, object key, object value) except -1: | |
| cdef Writer* writer = <Writer*>hook | |
| if _write_str_raise_on_nlcr(writer, key) < 0: | |
| return -1 | |
| if _write_byte(writer, b':') < 0: | |
| return -1 | |
| if _write_byte(writer, b' ') < 0: | |
| return -1 | |
| if _write_str_raise_on_nlcr(writer, value) < 0: | |
| return -1 | |
| if _write_byte(writer, b'\r') < 0: | |
| return -1 | |
| if _write_byte(writer, b'\n') < 0: | |
| return -1 | |
| return 0 | |
| def _serialize_headers(str status_line, object headers): | |
| cdef Writer writer | |
| cdef object key | |
| cdef object val | |
| cdef char buf[BUF_SIZE] | |
| _init_writer(&writer, buf) | |
| try: | |
| if _write_str(&writer, status_line) < 0: | |
| raise | |
| if _write_byte(&writer, b'\r') < 0: | |
| raise | |
| if _write_byte(&writer, b'\n') < 0: | |
| raise | |
| if AnyDict_items(<void*>(&writer), headers, _serlize_header) < 0: | |
| raise | |
| if _write_byte(&writer, b'\r') < 0: | |
| raise | |
| if _write_byte(&writer, b'\n') < 0: | |
| raise | |
| return PyBytes_FromStringAndSize(writer.buf, writer.pos) | |
| finally: | |
| _release_writer(&writer) |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # cython: language_level = 3, freethreading_compatible = True | |
| # Here we have a pure example of how the previous script shown above could be utilized to help us. | |
| # Used for speeding up multidict whenever it's possible to speedup, | |
| # this also makes up for it's python backend by providing a workaround | |
| # when its being provided. | |
| # mdcs stands for multidict-compatable-speedups for being python compatable. | |
| # it's mostly utilized for wrapping dict/ multidict-py and multidict-c | |
| # iterables but it can also be used without being slow but this can be utilized | |
| # in other ways. | |
| from cpython.object cimport PyTypeObject, PyObject | |
| from cpython.dict cimport PyDict_Next | |
| from cpython.mapping cimport PyMapping_Items, PyMapping_Check | |
| from cpython.ref cimport Py_CLEAR | |
| from cpython.exc cimport PyErr_SetObject | |
| from libc.stdint cimport uint64_t # version | |
| # Used to detect all 4 MultiDict Types in the Python backend | |
| # incase someone randomly mixes them somehow (This may have been the reason) | |
| # the C-API Capsule Multidict was supposed to be getting never came true. | |
| # mod_state if anybody needs it (even though were only intrested in iterables) | |
| cdef extern from "_multilib/state.h": | |
| # We mostly want this for checking IStrType when doing iterations | |
| ctypedef struct mod_state: | |
| PyTypeObject *IStrType | |
| PyTypeObject *MultiDictType | |
| PyTypeObject *CIMultiDictType | |
| PyTypeObject *MultiDictProxyType | |
| PyTypeObject *CIMultiDictProxyType | |
| PyTypeObject *KeysViewType | |
| PyTypeObject *ItemsViewType | |
| PyTypeObject *ValuesViewType | |
| PyTypeObject *KeysIterType | |
| PyTypeObject *ItemsIterType | |
| PyTypeObject *ValuesIterType | |
| # https://cython.readthedocs.io/en/stable/src/userguide/extension_types.html#external-extension-types | |
| cdef extern from "_multilib/dict.h": | |
| """ | |
| /* part of mdcs.pyx made for speeding up multidict and other mapping objects in iteration */ | |
| typedef MultiDictObject CIMultiDictObject; | |
| typedef MultiDictProxyObject CIMultiDictProxyObject; | |
| """ | |
| ctypedef struct MultiDictObject: | |
| pass | |
| # It's the same picture but cython doesn't understand this. | |
| ctypedef struct CIMultiDictObject "MultiDictObject": # type: ignore (Stupid Cyright) | |
| pass | |
| ctypedef struct MultiDictProxyObject: | |
| MultiDictObject* md | |
| ctypedef struct CIMultiDictProxyObject "MultiDictProxyObject": # type: ignore | |
| MultiDictObject* md | |
| # NOTE: A Safe-importable way to skip around ._multidict_py would be a nice touch. | |
| # I'm currently having trouble getting ._multidict to cooperate with me at the moment but a change in how | |
| # these two module work would be a nice touch/workaround to this problem. | |
| # We only need to cherrypick the stuff out that we would want from the capi other things can be ignored | |
| ctypedef class multidict.MultiDict [object MultiDictObject, check_size ignore]: # type: ignore | |
| pass | |
| ctypedef class multidict.CIMultiDict [object CIMultiDictObject, check_size ignore]: # type: ignore | |
| pass | |
| ctypedef class multidict.MultiDictProxy [object MultiDictProxyObject, check_size ignore]: # type: ignore | |
| pass | |
| ctypedef class multidict.CIMultiDictProxy [object CIMultiDictProxyObject, check_size ignore]: # type: ignore | |
| pass | |
| cdef extern from "_multilib/hashtable.h": | |
| cdef struct _md_pos: | |
| Py_ssize_t pos | |
| uint64_t version | |
| ctypedef _md_pos md_pos_t | |
| void md_init_pos(MultiDictObject* md, md_pos_t *pos) # type: ignore | |
| int md_next(MultiDictObject* md, md_pos_t *pos, PyObject **pidentity, PyObject **pkey, PyObject **pvalue) except -1 # type: ignore | |
| # we use a datacallback to simulate iterators over multiple different | |
| # dictionary types so that we're never slow while cutting out as many steps | |
| # as possible for each different case scenario... | |
| ctypedef int (*md_iter_next)(void* hook, object key, object value) except -1 | |
| cdef inline int _MultiDict_items(void* hook, MultiDictObject* md, md_iter_next cb) except -1: | |
| cdef md_pos_t pos | |
| cdef PyObject* key = NULL | |
| cdef PyObject* value = NULL | |
| cdef int res | |
| md_init_pos(md, &pos) | |
| while True: | |
| try: | |
| res = md_next(md, &pos, NULL, &key, &value) | |
| if res < 0: | |
| return -1 | |
| if res == 0: | |
| # Finished | |
| return 0 | |
| if cb(hook, <object>key, <object>value) < 0: | |
| return -1 | |
| finally: | |
| Py_CLEAR(key) | |
| Py_CLEAR(value) | |
| cdef inline int _PyMapping_items(void* hook, object md, md_iter_next cb) except -1: | |
| cdef object k, v | |
| try: | |
| for k, v in PyMapping_Items(md): | |
| if cb(hook, k, v) < 0: | |
| return -1 | |
| return 0 | |
| except BaseException as e: | |
| PyErr_SetObject(type(e), e) | |
| return -1 | |
| cdef inline int _PyDict_items(void* hook, object pydict, md_iter_next cb) except -1: | |
| cdef PyObject* key = NULL | |
| cdef PyObject* value = NULL | |
| cdef Py_ssize_t pos = 0 | |
| while PyDict_Next(pydict, &pos, &key, &value): | |
| if cb(hook, <object>key, <object>value) < 0: | |
| return -1 | |
| return 0 | |
| cdef inline int AnyDict_items(void* hook , object any_dict, md_iter_next cb) except -1: | |
| if isinstance(any_dict, (MultiDict, CIMultiDict)): # type: ignore | |
| return _MultiDict_items(hook, <MultiDictObject*>any_dict, cb) | |
| elif isinstance(any_dict, (MultiDictProxy, CIMultiDictProxy)): # type: ignore | |
| return _MultiDict_items(hook, (<MultiDictProxyObject*>any_dict).md, cb) | |
| elif isinstance(any_dict, dict): | |
| return _PyDict_items(hook, any_dict, cb) | |
| elif PyMapping_Check(any_dict): | |
| return _PyMapping_items(hook, any_dict, cb) | |
| else: | |
| PyErr_SetObject(TypeError, f"Invalid type, expected a dictionary or multidict-like object got {type(any_dict).__name__!r}") | |
| return -1 | |
| # I used this for testing purposes... | |
| # cdef int test_iter_cb(void* hook, object key, object item) except -1: | |
| # try: | |
| # (<object>hook)(key, item) | |
| # return 0 | |
| # except BaseException as e: | |
| # PyErr_SetObject(type(e), e) | |
| # return -1 | |
| # def test_iterable(object cb, object any_dict): | |
| # if AnyDict_items(<void*>cb, any_dict, test_iter_cb) < 0: | |
| # raise | |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # because we may never see a real C-API see the light of day I made a new solution to extract the current multidict headers | |
| # based off how python_capicompat works. This is less costly for the maintainers and means that the problem | |
| # now relies soley off of us instead. | |
| import argparse | |
| import os | |
| import urllib.request | |
| import sys | |
| from pathlib import Path | |
| from functools import lru_cache | |
| @lru_cache | |
| def _import_toml(): | |
| """Inspired by msgspec's version but has another backup | |
| called rtoml which takes priority if found as it is the fastest | |
| version and has readable typehints that tools like vs-code can | |
| understand""" | |
| try: | |
| import rtoml | |
| return rtoml | |
| except ImportError: | |
| pass | |
| try: | |
| import tomllib # type: ignore | |
| return tomllib | |
| except ImportError: | |
| pass | |
| try: | |
| import tomli # type: ignore | |
| return tomli | |
| except ImportError: | |
| raise ImportError( | |
| "`update_multidict.py` requires `tomli` or `rtoml` be installed.\n\n" | |
| "Please either `pip` or `conda` install it as follows:\n\n" | |
| " $ python -m pip install rtoml # using pip\n" | |
| " $ conda install rtoml # or using conda\n" | |
| " $ uv pip install rtoml # using uv" | |
| ) from None | |
| URL_PATH = ( | |
| "https://raw.githubusercontent.com/aio-libs/multidict/refs/heads/master/multidict/_multilib/{}" | |
| ) | |
| @lru_cache() | |
| def project_path() -> Path: | |
| """Assuming this item is in `<project-path>/tools/<me>` it will use this info to find pyproject.toml | |
| and other items""" | |
| return Path(__file__).parent.parent | |
| @lru_cache() | |
| def project_name() -> str: | |
| toml = _import_toml() | |
| data = toml.load(project_path() / "pyproject.toml") | |
| return data['project']['name'] | |
| @lru_cache() | |
| def code_path() -> Path: | |
| path = project_path() | |
| name = project_name() | |
| if os.path.exists(path / "src"): | |
| return path / "src" / name | |
| else: | |
| # second best guess | |
| assert os.path.exists(path / name), f"could not find code folder for project {name!r}" | |
| return path / name | |
| class MultiDictUpdater: | |
| FILES = [ | |
| "dict.h", | |
| "hashtable.h", | |
| "htkeys.h", | |
| "istr.h", | |
| "iter.h", | |
| "parser.h", | |
| "pythoncapi_compat.h", | |
| "state.h", | |
| "views.h", | |
| ] | |
| def __init__(self, path: str | Path): | |
| self._path = Path(path) / "_multilib" | |
| if not self._path.exists(): | |
| os.mkdir(self._path) | |
| def log(self, msg=""): | |
| print(msg, file=sys.stderr, flush=True) | |
| def get_header(self, file: str): | |
| header_url = URL_PATH.format(file) | |
| target = os.path.join(self._path, file) | |
| self.log(f"Download the file from _multilib/{file} to {target}.") | |
| urllib.request.urlretrieve(str(header_url), target) | |
| def update(self): | |
| for f in self.FILES: | |
| self.get_header(f) | |
| @staticmethod | |
| def find_project_name() -> str: | |
| return project_name() | |
| @staticmethod | |
| def get_possible_project_path() -> Path: | |
| return code_path() | |
| @classmethod | |
| def from_args(cls) -> "MultiDictUpdater": | |
| parser = argparse.ArgumentParser( | |
| description="Downloads Multidict C Library as a cython speedup" | |
| " this tool can be used to safely update multidict's backend to speedup cython" | |
| ) | |
| parser.add_argument( | |
| "--path", default=cls.get_possible_project_path(), required=False, help="defaults to: %(default)s" | |
| ) | |
| return cls(parser.parse_args().path) | |
| def main(): | |
| MultiDictUpdater.from_args().update() | |
| if __name__ == "__main__": | |
| main() |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment