Last active
July 1, 2025 14:52
-
-
Save Vizonex/01ec0c9d3e49069a5711e004163368c3 to your computer and use it in GitHub Desktop.
This is a new concept for generating the New Multidict CAPI to cython using libclang, Couldn't be mypy friendly enough yet so I'm shoving it here.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| """ | |
| Generates A Multidict Cython API from multidict_api.h | |
| by parsing the functions to generate the Cython API | |
| Automatically. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import os | |
| from dataclasses import dataclass, field | |
| from datetime import datetime, timezone | |
| from functools import cached_property | |
| from pathlib import Path | |
| from typing import Any, Callable, Sequence | |
| from clang.cindex import Cursor, CursorKind, Index, Token, TranslationUnit, Type | |
| # This uses cython by default because it's assumed you have cython installed | |
| # if not throw an issue about it, will program a workaround | |
| from Cython.CodeWriter import LinesResult | |
| from Cython.Tempita import Template | |
| __author__ = "Vizonex" | |
| __license__ = "MIT" | |
| # Inspired by rust's Bindgen tool they use for making C bindings | |
| # Originally I was going to make my own new version of pxdgen that | |
| # retained argument names but I guess it had a different use... | |
| class CursorVisitor: | |
| """Inspired by Cython's Visitor Approch made for use with Clang""" | |
| dispatch_table: dict[CursorKind, Callable[[Cursor], None | Any]] | |
| def __init__(self) -> None: | |
| self.dispatch_table = {} | |
| def visit(self, cursor: Cursor) -> Any: | |
| return self._visit(cursor) | |
| def _visit(self, cursor: Cursor) -> Any: | |
| try: | |
| handler_method = self.dispatch_table[cursor.kind] | |
| except KeyError: | |
| handler_method = self.find_handler(cursor) # type: ignore[no-untyped-def] | |
| self.dispatch_table[cursor.kind] = handler_method | |
| if handler_method is not None: | |
| return handler_method(cursor) | |
| raise RuntimeError( | |
| "Visitor %r does not accept %s" % (self, cursor.kind.name.lower()) | |
| ) | |
| def find_handler(self, cursor: Cursor) -> Callable[[Cursor], None] | None: | |
| return getattr(self, "visit_" + cursor.kind.name.lower(), None) | |
| def visitchildren( | |
| self, | |
| parent: Cursor, | |
| attrs: Sequence[CursorKind] | None = None, | |
| exclude: Sequence[CursorKind] | None = None, | |
| ) -> list[Any]: | |
| return self._visitchildren(parent, attrs, exclude) | |
| def _visitchildren( | |
| self, | |
| parent: Cursor, | |
| attrs: Sequence[CursorKind] | None, | |
| exclude: Sequence[CursorKind] | None, | |
| ) -> list[Any]: | |
| items = [] | |
| for child in parent.get_children(): | |
| if attrs is not None and child.kind not in attrs: | |
| continue | |
| if exclude is not None and child.kind in exclude: | |
| continue | |
| items.append(self.visit(child)) | |
| return items | |
| class RootVisitor(CursorVisitor): | |
| """Basic BaseType for working with different clang visitors""" | |
| def __init__(self, translation_unit: TranslationUnit, header: str): | |
| super().__init__() | |
| self.tu = translation_unit | |
| self.header = header | |
| self.data = {} | |
| def start(self): | |
| return [r for r in self.visitchildren(self.tu.cursor) if r is not None] | |
| def visit(self, cursor: Cursor): | |
| # Filter anything this isn't us... | |
| if Path(str(cursor.location.file)).parts[-1] == self.header: | |
| return super().visit(cursor) | |
| # === C Datatypes === | |
| @dataclass | |
| class EnumType: | |
| name: str | |
| fields: list[tuple[str, int]] | |
| @dataclass | |
| class TypeDef: | |
| name: str | |
| obj: list | |
| @dataclass | |
| class StructType: | |
| name: str | |
| fields: list | |
| @dataclass | |
| class FieldType: | |
| name: str | |
| type_name: str | |
| @property | |
| def real_typename(self): | |
| """Determines and translates typenames""" | |
| if self.type_name == "PyObject *": | |
| return "object" | |
| return self.type_name | |
| @dataclass | |
| class PrototypeFunction: | |
| name: str | |
| fields: list[FieldType] | |
| ret_name: str | |
| def define(self): | |
| code = f"{self.ret_name} (*{self.name})(" | |
| code += ", ".join([f"{f.real_typename} {f.name}" for f in self.fields]) + ")" | |
| return code | |
| @dataclass | |
| class TypeRef: | |
| name: str | |
| @dataclass | |
| class FunctionType(PrototypeFunction): | |
| @cached_property | |
| def exception_check(self): | |
| if "*" in self.ret_name: | |
| return " except NULL" | |
| if "int" == self.ret_name: | |
| return " except -1" | |
| return '' | |
| @cached_property | |
| def cy_return_name(self): | |
| """defines the Cythonic return name of a variable""" | |
| if self.name.endswith("New"): | |
| return self.name[:self.name.find("_")] | |
| elif self.name.endswith("PopItem"): | |
| return "tuple" | |
| elif self.name.endswith("GetType"): | |
| return "type" | |
| elif self.name.endswith(("_CheckExact", "_Check")): | |
| return "int" | |
| elif self.name.startswith("IStr"): | |
| return "istr" | |
| else: | |
| return self.ret_name | |
| def define(self): | |
| # we have to underscore the definitions in order to ensure we can | |
| # still define the others the way we wish to define them... | |
| # I'll give it a C at the beginning for users who want to | |
| # use the CAPI directly instead of with the global __cython_multidict_api block. | |
| code = f"{self.ret_name} C_{self.name} \"{self.name}\"(" | |
| code += ", ".join([f"{f.real_typename} {f.name}" for f in self.fields]) + ")" | |
| return code + self.exception_check | |
| def cython_definition(self, w:CythonBodyWriter): | |
| w.startline(f"cdef inline {self.cy_return_name} {self.name}(") | |
| w.endline(", ".join([f"{f.real_typename} {f.name}" for f in self.fields][1:]) + "):") | |
| w.indent() | |
| w.startline("return ") | |
| if self.ret_name != self.cy_return_name: | |
| w.put(f"<{self.cy_return_name}>") | |
| w.put(f"C_{self.name}(") | |
| if self.fields and self.fields[0].name == "api": | |
| w.put("__cython_multidict_api") | |
| if len(self.fields) > 1: | |
| w.put(", ") | |
| w.put(", ".join([f.name for f in self.fields][1:])) | |
| else: | |
| w.put(", ".join([f.name for f in self.fields])) | |
| w.endline(")") | |
| w.dedent() | |
| # Give a newline | |
| w.putline("") | |
| class CythonBodyWriter: | |
| """Inspired by Cython's DeclarationWriter""" | |
| indent_string = " " | |
| def __init__(self, result=None): | |
| super().__init__() | |
| if result is None: | |
| result = LinesResult() | |
| self.result = result | |
| self.numindents = 0 | |
| def indent(self): | |
| self.numindents += 1 | |
| def dedent(self): | |
| self.numindents -= 1 | |
| def startline(self, s=""): | |
| self.result.put(self.indent_string * self.numindents + s) | |
| def put(self, s): | |
| self.result.put(s) | |
| def putline(self, s): | |
| self.result.putline(self.indent_string * self.numindents + s) | |
| def endline(self, s=""): | |
| self.result.putline(s) | |
| def line(self, s): | |
| self.startline(s) | |
| self.endline() | |
| def finish(self): | |
| return "\n".join(self.result.lines) | |
| class Library: | |
| def __init__(self, attrs: list[Any]): | |
| self.attrs = attrs | |
| @cached_property | |
| def capi(self): | |
| for s in self.attrs: | |
| if isinstance(s, StructType): | |
| if s.name == "MultiDict_CAPI": | |
| return s | |
| raise RuntimeError("MultiDict_CAPI Not found") | |
| def write_cython_body(self): | |
| w = CythonBodyWriter() | |
| # indent once becuase were inside of an extern | |
| w.indent() | |
| for t in self.attrs: | |
| w.putline("") | |
| if isinstance(t, EnumType): | |
| # === Enums === | |
| w.putline(f"enum {t.name}:") | |
| w.indent() | |
| for f in t.fields: | |
| w.putline(f"{f[0]} = {f[1]}") | |
| w.dedent() | |
| elif isinstance(t, StructType): | |
| # === Structs === | |
| w.putline(f"struct {t.name}:") | |
| w.indent() | |
| for f in t.fields: | |
| if isinstance(f, PrototypeFunction): | |
| w.putline(f.define()) | |
| w.dedent() | |
| elif isinstance(t, TypeDef): | |
| # === TypeDefines === | |
| # Cython compiler hates same-named things so ignore these... | |
| if t.name != t.obj[0].name: | |
| w.putline(f"ctypedef {t.obj[0].name} {t.name}") | |
| elif isinstance(t, FunctionType): | |
| # === Functions === | |
| w.putline(t.define()) | |
| # Finally the Cherry On top... | |
| w.putline("MultiDict_CAPI* __cython_multidict_api") | |
| w.dedent() | |
| w.putline("") | |
| w.putline("# === Cython API ===") | |
| for t in self.attrs: | |
| # Filter Non-Function-Types Were only intrested in the C-API Hooks | |
| if not isinstance(t, FunctionType): | |
| continue | |
| t.cython_definition(w) | |
| return w.finish() | |
| class CAPI_Visitor(RootVisitor): | |
| def start(self): | |
| return Library(list(super().start())) | |
| def visit_enum_constant_decl(self, cursor: Cursor): | |
| return (cursor.spelling, cursor.enum_value) | |
| def visit_enum_decl(self, cursor: Cursor): | |
| return EnumType(cursor.spelling, self.visitchildren(cursor)) | |
| def visit_typedef_decl(self, cursor: Cursor): | |
| return TypeDef(cursor.spelling, self.visitchildren(cursor)) | |
| def visit_struct_decl(self, cursor: Cursor): | |
| children = self.visitchildren(cursor) | |
| return StructType(cursor.spelling, children) | |
| def visit_type_ref(self, cursor: Cursor): | |
| return TypeRef(cursor.spelling) | |
| def visit_parm_decl(self, cursor: Cursor): | |
| return FieldType(cursor.spelling, cursor.type.spelling) # type: ignore[arg-type] | |
| def visit_field_decl(self, cursor: Cursor): | |
| if "(*)" in cursor.type.spelling: | |
| # Prototype Function | |
| rettype = cursor.type.spelling.split("(*)", 1)[0].strip() | |
| # Not done yet give me the parameter names, We can do far better than pxdgen... | |
| return PrototypeFunction( | |
| cursor.spelling, | |
| self.visitchildren(cursor, exclude=[CursorKind.TYPE_REF]), | |
| rettype, | |
| ) | |
| else: | |
| return FieldType(cursor.spelling, cursor.type.spelling) | |
| def visit_function_decl(self, cursor: Cursor): | |
| return FunctionType( | |
| cursor.spelling, | |
| self.visitchildren(cursor, attrs=[CursorKind.PARM_DECL]), | |
| ret_name=str(cursor.result_type.spelling), | |
| ) | |
| def visit_unexposed_decl(self, cursor: Cursor): | |
| # Unknown so pass, visitor will filter this as something to ignore | |
| pass | |
| @dataclass | |
| class MiniBindgen: | |
| clang_args: list[str] = field(default_factory=list) | |
| def clang_arg(self, arg: str | Sequence[str]) -> None: | |
| """Adds in a single clang argument a list of arguments that should be joined""" | |
| if not isinstance(arg, str): | |
| arg = "".join(arg) | |
| self.clang_args.append(arg) | |
| def include(self, path: str | Path) -> None: | |
| r"""Used for defining a path to your header files | |
| this will include clang arguments under the hood | |
| or you. This will also transform windows paths to forward | |
| slashes on the fly. | |
| :: | |
| bindings = MiniBindgen() | |
| bindings.include("path/to/header/files") | |
| pathlib is also supported if multiple operating systems need supporting | |
| :: | |
| from pathlib import Path | |
| bindings = MiniBindgen() | |
| bindings.header("header.h") | |
| bindings.include(Path("path") / "to" / "header" / "files") | |
| """ | |
| # Translate Windows Paths to forward slashes if possible. | |
| return self.clang_arg(("-I", Path(path).as_posix())) | |
| def includes(self, paths: list[str | Path]) -> None: | |
| """ | |
| Same idea as `include` but it made for iterators, | |
| sequence types and lists \n | |
| See: `include` function for details | |
| :: | |
| bindings = MiniBindings() | |
| # to demonstrate how it can be used... | |
| bindings.includes(["path1", "path2", Path("custom") / "path3"]) | |
| """ | |
| for p in paths: | |
| self.include(p) | |
| def generate( # type:ignore[no-untyped-def] | |
| self, | |
| header: Path | str, | |
| excludeDecls=False, | |
| options: int = 0, | |
| unsaved_files: list[tuple[str, str]] = [], | |
| ) -> CAPI_Visitor: | |
| """Generates a clang Translation unit to utilize""" | |
| index = Index.create(excludeDecls=excludeDecls) | |
| return CAPI_Visitor( | |
| index.parse( | |
| Path(header).as_posix(), | |
| self.clang_args, | |
| unsaved_files=unsaved_files, | |
| options=options, | |
| ), | |
| Path(header).parts[-1], | |
| ) | |
| CYTHON_API_HEAD = Template( | |
| """# cython: language_level = 3, freethreading_compatible=True | |
| from cpython.object cimport PyObject, PyTypeObject | |
| from libc.stdint cimport uint64_t | |
| # WARNING: THIS FILE IS AUTOGENERATED DO NOT EDIT!!! | |
| # YOUR CHANGES WILL GET OVERWRITTEN AND YOUR POOR EDITS WILL BE GONE!!! | |
| # GENERATED ON: {{date}} | |
| # COMPILIED BY: {{author}} | |
| cdef extern from "multidict_api.h": | |
| \"\"\" | |
| /* Extra Data comes from multidict.__init__.pxd */ | |
| /* Ensure we can obtain the Functions we wish to utilize */ | |
| #define MULTIDICT_IMPL | |
| MultiDict_CAPI* __cython_multidict_api; | |
| // Redefinitions incase required by cython... | |
| // Don't want size calculations to get screwed up... | |
| #if PY_VERSION_HEX >= 0x030c00f0 | |
| #define __MANAGED_WEAKREFS | |
| #endif | |
| typedef struct { | |
| PyObject_HEAD | |
| #ifndef __MANAGED_WEAKREFS | |
| PyObject *weaklist; | |
| #endif | |
| // we can ignore state however we already know it's | |
| // a size of 4/8 depending on 32/64 bit... | |
| void *state; | |
| Py_ssize_t used; | |
| uint64_t version; | |
| bool is_ci; | |
| htkeys_t *keys; | |
| } MultiDictObject; | |
| typedef struct { | |
| PyObject_HEAD | |
| #ifndef __MANAGED_WEAKREFS | |
| PyObject *weaklist; | |
| #endif | |
| MultiDictObject *md; | |
| } MultiDictProxyObject; | |
| typedef struct { | |
| PyUnicodeObject str; | |
| PyObject *canonical; | |
| void *state; | |
| } istrobject; | |
| int multidict_import(){ | |
| __cython_multidict_api = MultiDict_Import(); | |
| return __cython_multidict_api != NULL; | |
| } | |
| \"\"\" | |
| # NOTE: Important that you import this | |
| # After you've c-imported multidict | |
| int multidict_import() except 0 | |
| # Predefined objects from istr & multidict | |
| ctypedef struct MultiDictObject: | |
| pass | |
| ctypedef struct MultiDictProxyObject: | |
| pass | |
| ctypedef struct istrobject: | |
| pass | |
| ctypedef class _multidict.istr [object istrobject, check_size ignore]: | |
| pass | |
| ctypedef class _multidict.MultiDict [object MultiDictObject, check_size ignore]: | |
| pass | |
| ctypedef class _multidict.CIMultiDict [object MultiDictObject, check_size ignore]: | |
| pass | |
| ctypedef class _multidict.MultiDictProxy [object MultiDictProxyObject, check_size ignore]: | |
| pass | |
| ctypedef class _multidict.CIMultiDictProxy [object MultiDictProxyObject, check_size ignore]: | |
| pass | |
| """ | |
| ) # type: ignore[no-untyped-call] | |
| def scan_header_file(header_file: str, author_name: str = "anonymous") -> str: | |
| bg = MiniBindgen() | |
| # clang needs to be able to identify PyObjects and PyTypeObjects | |
| # this should be able to find a user's Python Path to where Python.h | |
| # is stored. If this is not you I am in the middle of making a workaround for that... | |
| bg.include(Path(os.environ["PYTHONPATH"]) / "include") | |
| # print(bg) | |
| data: Library = bg.generate(header_file).start() # type: ignore[no-untyped-call] | |
| file: str = CYTHON_API_HEAD.substitute( | |
| author=author_name, date=datetime.now(timezone.utc) | |
| ) # type: ignore[no-untyped-call] | |
| file += data.write_cython_body() # type: ignore[no-untyped-call] | |
| return file | |
| def cli() -> None: | |
| parser = argparse.ArgumentParser( | |
| description="Compiles Multidict Cython pxd file using clang" | |
| ) | |
| parser.add_argument( | |
| "-a", "--author", help="who's compiling the code", default="anonymous" | |
| ) | |
| parser.add_argument("-o", "--output", default="multidict/__init__.pxd") | |
| parser.add_argument( | |
| "-i", "--input", help="input header file", default="multidict/multidict_api.h" | |
| ) | |
| args = parser.parse_args() | |
| data = scan_header_file(args.input, args.author) | |
| with open(args.output, "w") as w: | |
| w.write(data) | |
| if __name__ == "__main__": | |
| cli() |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment