Skip to content

Instantly share code, notes, and snippets.

@Vizonex
Last active July 1, 2025 14:52
Show Gist options
  • Select an option

  • Save Vizonex/01ec0c9d3e49069a5711e004163368c3 to your computer and use it in GitHub Desktop.

Select an option

Save Vizonex/01ec0c9d3e49069a5711e004163368c3 to your computer and use it in GitHub Desktop.
This is a new concept for generating the New Multidict CAPI to cython using libclang, Couldn't be mypy friendly enough yet so I'm shoving it here.
"""
Generates A Multidict Cython API from multidict_api.h
by parsing the functions to generate the Cython API
Automatically.
"""
from __future__ import annotations
import argparse
import os
from dataclasses import dataclass, field
from datetime import datetime, timezone
from functools import cached_property
from pathlib import Path
from typing import Any, Callable, Sequence
from clang.cindex import Cursor, CursorKind, Index, Token, TranslationUnit, Type
# This uses cython by default because it's assumed you have cython installed
# if not throw an issue about it, will program a workaround
from Cython.CodeWriter import LinesResult
from Cython.Tempita import Template
__author__ = "Vizonex"
__license__ = "MIT"
# Inspired by rust's Bindgen tool they use for making C bindings
# Originally I was going to make my own new version of pxdgen that
# retained argument names but I guess it had a different use...
class CursorVisitor:
"""Inspired by Cython's Visitor Approch made for use with Clang"""
dispatch_table: dict[CursorKind, Callable[[Cursor], None | Any]]
def __init__(self) -> None:
self.dispatch_table = {}
def visit(self, cursor: Cursor) -> Any:
return self._visit(cursor)
def _visit(self, cursor: Cursor) -> Any:
try:
handler_method = self.dispatch_table[cursor.kind]
except KeyError:
handler_method = self.find_handler(cursor) # type: ignore[no-untyped-def]
self.dispatch_table[cursor.kind] = handler_method
if handler_method is not None:
return handler_method(cursor)
raise RuntimeError(
"Visitor %r does not accept %s" % (self, cursor.kind.name.lower())
)
def find_handler(self, cursor: Cursor) -> Callable[[Cursor], None] | None:
return getattr(self, "visit_" + cursor.kind.name.lower(), None)
def visitchildren(
self,
parent: Cursor,
attrs: Sequence[CursorKind] | None = None,
exclude: Sequence[CursorKind] | None = None,
) -> list[Any]:
return self._visitchildren(parent, attrs, exclude)
def _visitchildren(
self,
parent: Cursor,
attrs: Sequence[CursorKind] | None,
exclude: Sequence[CursorKind] | None,
) -> list[Any]:
items = []
for child in parent.get_children():
if attrs is not None and child.kind not in attrs:
continue
if exclude is not None and child.kind in exclude:
continue
items.append(self.visit(child))
return items
class RootVisitor(CursorVisitor):
"""Basic BaseType for working with different clang visitors"""
def __init__(self, translation_unit: TranslationUnit, header: str):
super().__init__()
self.tu = translation_unit
self.header = header
self.data = {}
def start(self):
return [r for r in self.visitchildren(self.tu.cursor) if r is not None]
def visit(self, cursor: Cursor):
# Filter anything this isn't us...
if Path(str(cursor.location.file)).parts[-1] == self.header:
return super().visit(cursor)
# === C Datatypes ===
@dataclass
class EnumType:
name: str
fields: list[tuple[str, int]]
@dataclass
class TypeDef:
name: str
obj: list
@dataclass
class StructType:
name: str
fields: list
@dataclass
class FieldType:
name: str
type_name: str
@property
def real_typename(self):
"""Determines and translates typenames"""
if self.type_name == "PyObject *":
return "object"
return self.type_name
@dataclass
class PrototypeFunction:
name: str
fields: list[FieldType]
ret_name: str
def define(self):
code = f"{self.ret_name} (*{self.name})("
code += ", ".join([f"{f.real_typename} {f.name}" for f in self.fields]) + ")"
return code
@dataclass
class TypeRef:
name: str
@dataclass
class FunctionType(PrototypeFunction):
@cached_property
def exception_check(self):
if "*" in self.ret_name:
return " except NULL"
if "int" == self.ret_name:
return " except -1"
return ''
@cached_property
def cy_return_name(self):
"""defines the Cythonic return name of a variable"""
if self.name.endswith("New"):
return self.name[:self.name.find("_")]
elif self.name.endswith("PopItem"):
return "tuple"
elif self.name.endswith("GetType"):
return "type"
elif self.name.endswith(("_CheckExact", "_Check")):
return "int"
elif self.name.startswith("IStr"):
return "istr"
else:
return self.ret_name
def define(self):
# we have to underscore the definitions in order to ensure we can
# still define the others the way we wish to define them...
# I'll give it a C at the beginning for users who want to
# use the CAPI directly instead of with the global __cython_multidict_api block.
code = f"{self.ret_name} C_{self.name} \"{self.name}\"("
code += ", ".join([f"{f.real_typename} {f.name}" for f in self.fields]) + ")"
return code + self.exception_check
def cython_definition(self, w:CythonBodyWriter):
w.startline(f"cdef inline {self.cy_return_name} {self.name}(")
w.endline(", ".join([f"{f.real_typename} {f.name}" for f in self.fields][1:]) + "):")
w.indent()
w.startline("return ")
if self.ret_name != self.cy_return_name:
w.put(f"<{self.cy_return_name}>")
w.put(f"C_{self.name}(")
if self.fields and self.fields[0].name == "api":
w.put("__cython_multidict_api")
if len(self.fields) > 1:
w.put(", ")
w.put(", ".join([f.name for f in self.fields][1:]))
else:
w.put(", ".join([f.name for f in self.fields]))
w.endline(")")
w.dedent()
# Give a newline
w.putline("")
class CythonBodyWriter:
"""Inspired by Cython's DeclarationWriter"""
indent_string = " "
def __init__(self, result=None):
super().__init__()
if result is None:
result = LinesResult()
self.result = result
self.numindents = 0
def indent(self):
self.numindents += 1
def dedent(self):
self.numindents -= 1
def startline(self, s=""):
self.result.put(self.indent_string * self.numindents + s)
def put(self, s):
self.result.put(s)
def putline(self, s):
self.result.putline(self.indent_string * self.numindents + s)
def endline(self, s=""):
self.result.putline(s)
def line(self, s):
self.startline(s)
self.endline()
def finish(self):
return "\n".join(self.result.lines)
class Library:
def __init__(self, attrs: list[Any]):
self.attrs = attrs
@cached_property
def capi(self):
for s in self.attrs:
if isinstance(s, StructType):
if s.name == "MultiDict_CAPI":
return s
raise RuntimeError("MultiDict_CAPI Not found")
def write_cython_body(self):
w = CythonBodyWriter()
# indent once becuase were inside of an extern
w.indent()
for t in self.attrs:
w.putline("")
if isinstance(t, EnumType):
# === Enums ===
w.putline(f"enum {t.name}:")
w.indent()
for f in t.fields:
w.putline(f"{f[0]} = {f[1]}")
w.dedent()
elif isinstance(t, StructType):
# === Structs ===
w.putline(f"struct {t.name}:")
w.indent()
for f in t.fields:
if isinstance(f, PrototypeFunction):
w.putline(f.define())
w.dedent()
elif isinstance(t, TypeDef):
# === TypeDefines ===
# Cython compiler hates same-named things so ignore these...
if t.name != t.obj[0].name:
w.putline(f"ctypedef {t.obj[0].name} {t.name}")
elif isinstance(t, FunctionType):
# === Functions ===
w.putline(t.define())
# Finally the Cherry On top...
w.putline("MultiDict_CAPI* __cython_multidict_api")
w.dedent()
w.putline("")
w.putline("# === Cython API ===")
for t in self.attrs:
# Filter Non-Function-Types Were only intrested in the C-API Hooks
if not isinstance(t, FunctionType):
continue
t.cython_definition(w)
return w.finish()
class CAPI_Visitor(RootVisitor):
def start(self):
return Library(list(super().start()))
def visit_enum_constant_decl(self, cursor: Cursor):
return (cursor.spelling, cursor.enum_value)
def visit_enum_decl(self, cursor: Cursor):
return EnumType(cursor.spelling, self.visitchildren(cursor))
def visit_typedef_decl(self, cursor: Cursor):
return TypeDef(cursor.spelling, self.visitchildren(cursor))
def visit_struct_decl(self, cursor: Cursor):
children = self.visitchildren(cursor)
return StructType(cursor.spelling, children)
def visit_type_ref(self, cursor: Cursor):
return TypeRef(cursor.spelling)
def visit_parm_decl(self, cursor: Cursor):
return FieldType(cursor.spelling, cursor.type.spelling) # type: ignore[arg-type]
def visit_field_decl(self, cursor: Cursor):
if "(*)" in cursor.type.spelling:
# Prototype Function
rettype = cursor.type.spelling.split("(*)", 1)[0].strip()
# Not done yet give me the parameter names, We can do far better than pxdgen...
return PrototypeFunction(
cursor.spelling,
self.visitchildren(cursor, exclude=[CursorKind.TYPE_REF]),
rettype,
)
else:
return FieldType(cursor.spelling, cursor.type.spelling)
def visit_function_decl(self, cursor: Cursor):
return FunctionType(
cursor.spelling,
self.visitchildren(cursor, attrs=[CursorKind.PARM_DECL]),
ret_name=str(cursor.result_type.spelling),
)
def visit_unexposed_decl(self, cursor: Cursor):
# Unknown so pass, visitor will filter this as something to ignore
pass
@dataclass
class MiniBindgen:
clang_args: list[str] = field(default_factory=list)
def clang_arg(self, arg: str | Sequence[str]) -> None:
"""Adds in a single clang argument a list of arguments that should be joined"""
if not isinstance(arg, str):
arg = "".join(arg)
self.clang_args.append(arg)
def include(self, path: str | Path) -> None:
r"""Used for defining a path to your header files
this will include clang arguments under the hood
or you. This will also transform windows paths to forward
slashes on the fly.
::
bindings = MiniBindgen()
bindings.include("path/to/header/files")
pathlib is also supported if multiple operating systems need supporting
::
from pathlib import Path
bindings = MiniBindgen()
bindings.header("header.h")
bindings.include(Path("path") / "to" / "header" / "files")
"""
# Translate Windows Paths to forward slashes if possible.
return self.clang_arg(("-I", Path(path).as_posix()))
def includes(self, paths: list[str | Path]) -> None:
"""
Same idea as `include` but it made for iterators,
sequence types and lists \n
See: `include` function for details
::
bindings = MiniBindings()
# to demonstrate how it can be used...
bindings.includes(["path1", "path2", Path("custom") / "path3"])
"""
for p in paths:
self.include(p)
def generate( # type:ignore[no-untyped-def]
self,
header: Path | str,
excludeDecls=False,
options: int = 0,
unsaved_files: list[tuple[str, str]] = [],
) -> CAPI_Visitor:
"""Generates a clang Translation unit to utilize"""
index = Index.create(excludeDecls=excludeDecls)
return CAPI_Visitor(
index.parse(
Path(header).as_posix(),
self.clang_args,
unsaved_files=unsaved_files,
options=options,
),
Path(header).parts[-1],
)
CYTHON_API_HEAD = Template(
"""# cython: language_level = 3, freethreading_compatible=True
from cpython.object cimport PyObject, PyTypeObject
from libc.stdint cimport uint64_t
# WARNING: THIS FILE IS AUTOGENERATED DO NOT EDIT!!!
# YOUR CHANGES WILL GET OVERWRITTEN AND YOUR POOR EDITS WILL BE GONE!!!
# GENERATED ON: {{date}}
# COMPILIED BY: {{author}}
cdef extern from "multidict_api.h":
\"\"\"
/* Extra Data comes from multidict.__init__.pxd */
/* Ensure we can obtain the Functions we wish to utilize */
#define MULTIDICT_IMPL
MultiDict_CAPI* __cython_multidict_api;
// Redefinitions incase required by cython...
// Don't want size calculations to get screwed up...
#if PY_VERSION_HEX >= 0x030c00f0
#define __MANAGED_WEAKREFS
#endif
typedef struct {
PyObject_HEAD
#ifndef __MANAGED_WEAKREFS
PyObject *weaklist;
#endif
// we can ignore state however we already know it's
// a size of 4/8 depending on 32/64 bit...
void *state;
Py_ssize_t used;
uint64_t version;
bool is_ci;
htkeys_t *keys;
} MultiDictObject;
typedef struct {
PyObject_HEAD
#ifndef __MANAGED_WEAKREFS
PyObject *weaklist;
#endif
MultiDictObject *md;
} MultiDictProxyObject;
typedef struct {
PyUnicodeObject str;
PyObject *canonical;
void *state;
} istrobject;
int multidict_import(){
__cython_multidict_api = MultiDict_Import();
return __cython_multidict_api != NULL;
}
\"\"\"
# NOTE: Important that you import this
# After you've c-imported multidict
int multidict_import() except 0
# Predefined objects from istr & multidict
ctypedef struct MultiDictObject:
pass
ctypedef struct MultiDictProxyObject:
pass
ctypedef struct istrobject:
pass
ctypedef class _multidict.istr [object istrobject, check_size ignore]:
pass
ctypedef class _multidict.MultiDict [object MultiDictObject, check_size ignore]:
pass
ctypedef class _multidict.CIMultiDict [object MultiDictObject, check_size ignore]:
pass
ctypedef class _multidict.MultiDictProxy [object MultiDictProxyObject, check_size ignore]:
pass
ctypedef class _multidict.CIMultiDictProxy [object MultiDictProxyObject, check_size ignore]:
pass
"""
) # type: ignore[no-untyped-call]
def scan_header_file(header_file: str, author_name: str = "anonymous") -> str:
bg = MiniBindgen()
# clang needs to be able to identify PyObjects and PyTypeObjects
# this should be able to find a user's Python Path to where Python.h
# is stored. If this is not you I am in the middle of making a workaround for that...
bg.include(Path(os.environ["PYTHONPATH"]) / "include")
# print(bg)
data: Library = bg.generate(header_file).start() # type: ignore[no-untyped-call]
file: str = CYTHON_API_HEAD.substitute(
author=author_name, date=datetime.now(timezone.utc)
) # type: ignore[no-untyped-call]
file += data.write_cython_body() # type: ignore[no-untyped-call]
return file
def cli() -> None:
parser = argparse.ArgumentParser(
description="Compiles Multidict Cython pxd file using clang"
)
parser.add_argument(
"-a", "--author", help="who's compiling the code", default="anonymous"
)
parser.add_argument("-o", "--output", default="multidict/__init__.pxd")
parser.add_argument(
"-i", "--input", help="input header file", default="multidict/multidict_api.h"
)
args = parser.parse_args()
data = scan_header_file(args.input, args.author)
with open(args.output, "w") as w:
w.write(data)
if __name__ == "__main__":
cli()
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment