Last active
September 15, 2026 14:21
-
-
Save unrays/fb528b74d0bbcc6e6d53300fafab11de to your computer and use it in GitHub Desktop.
COMPILE-TIME LEXER ASSEMBLY AND DFA CONFIGURATION - This code comes from an unfinished project of mine. I found certain parts quite interesting, so I extracted a few of them and turned them into gists. This code is for educational purposes and is not functional, as it requires the rest of the architecture to work.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| // Copyright (c) July 2026 Félix-Olivier Dumas. All rights reserved. | |
| // Licensed under the terms described in the LICENSE file | |
| #if !defined(__INTELLISENSE__) | |
| using LexingAutomaton__final = StaticDFA< | |
| generate_expanded_dfa_config_t< | |
| ENABLED <nttp_to_type<LexState::Start>, charset_alpha, nttp_to_type<LexState::Identifier>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset_digits, nttp_to_type<LexState::Number>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<':'>, nttp_to_type<LexState::DelimiterColon>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<';'>, nttp_to_type<LexState::DelimiterSemi>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<','>, nttp_to_type<LexState::DelimiterComma>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<'('>, nttp_to_type<LexState::DelimiterLParen>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<')'>, nttp_to_type<LexState::DelimiterRParen>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<'{'>, nttp_to_type<LexState::DelimiterLCurly>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<'}'>, nttp_to_type<LexState::DelimiterRCurly>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<'['>, nttp_to_type<LexState::DelimiterLSquare>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<']'>, nttp_to_type<LexState::DelimiterRSquare>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<'<'>, nttp_to_type<LexState::DelimiterLAngle>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<'>'>, nttp_to_type<LexState::DelimiterRAngle>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset<'\n'>, nttp_to_type<LexState::Newline>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset_isdelimiter_pure, nttp_to_type<LexState::DelimiterOpaque>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset_isoperator_pure, nttp_to_type<LexState::Operator>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset_iswhitespace, nttp_to_type<LexState::Whitespace>>, | |
| ENABLED <nttp_to_type<LexState::Start>, charset_ispreprocessor, nttp_to_type<LexState::Preprocessor>>, | |
| ENABLED <nttp_to_type<LexState::Preprocessor>, charset_alpha, nttp_to_type<LexState::Preprocessor>>, | |
| ENABLED <nttp_to_type<LexState::Identifier>, charset_alphanumeric, nttp_to_type<LexState::Identifier>>, | |
| ENABLED <nttp_to_type<LexState::Number>, charset_digits, nttp_to_type<LexState::Number>>, | |
| ENABLED <nttp_to_type<LexState::Operator>, charset_isoperator, nttp_to_type<LexState::Operator>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterColon>, charset<':'>, nttp_to_type<LexState::DelimiterColon>>, | |
| DISABLED <nttp_to_type<LexState::Whitespace>, charset_iswhitespace, nttp_to_type<LexState::Whitespace>> | |
| > | |
| >; | |
| #else | |
| using LexingAutomaton__final = StaticDFA< | |
| StaticDfaTransitions< | |
| ENABLED <nttp_to_type<LexState::Start>, nttp_to_type<'\n'>, nttp_to_type<LexState::Newline>> | |
| > | |
| >; | |
| #endif | |
| using LexStateToTokenMapper__final = EnumMapper< | |
| EnumMapperConfiguration< | |
| ENABLED <nttp_to_type<LexState::Identifier>, nttp_to_type<TokenKind::Identifier>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterOpaque>, nttp_to_type<TokenKind::DelimiterOpaque>>, | |
| ENABLED <nttp_to_type<LexState::Operator>, nttp_to_type<TokenKind::Operator>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterColon>, nttp_to_type<TokenKind::DelimiterColon>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterSemi>, nttp_to_type<TokenKind::DelimiterSemicolon>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterComma>, nttp_to_type<TokenKind::DelimiterComma>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterLParen>, nttp_to_type<TokenKind::DelimiterLParen>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterRParen>, nttp_to_type<TokenKind::DelimiterRParen>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterLCurly>, nttp_to_type<TokenKind::DelimiterLCurly>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterRCurly>, nttp_to_type<TokenKind::DelimiterRCurly>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterLSquare>, nttp_to_type<TokenKind::DelimiterLSquare>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterRSquare>, nttp_to_type<TokenKind::DelimiterRSquare>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterLAngle>, nttp_to_type<TokenKind::DelimiterLAngle>>, | |
| ENABLED <nttp_to_type<LexState::DelimiterRAngle>, nttp_to_type<TokenKind::DelimiterRAngle>>, | |
| ENABLED <nttp_to_type<LexState::Preprocessor>, nttp_to_type<TokenKind::Preprocessor>>, | |
| ENABLED <nttp_to_type<LexState::Newline>, nttp_to_type<TokenKind::Newline>>, | |
| ENABLED <nttp_to_type<LexState::Number>, nttp_to_type<TokenKind::Number>>, | |
| CONDITIONAL <(0 == 0), nttp_to_type<LexState::Invalid>, nttp_to_type<TokenKind::Unknown>> | |
| > | |
| >; | |
| using TokenKwrdCategorizer__final = TokenKeywordCategorizer< | |
| TokenKeywordCategorizerConfiguration< | |
| ENABLED <AccessKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordAccess>>, | |
| ENABLED <AlignmentKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordAlignment>>, | |
| ENABLED <ControlKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordControl>>, | |
| ENABLED <ModifierKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordModifier>>, | |
| DISABLED <QualifierKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordQualifier>>, | |
| DISABLED <SpecifierKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordSpecifier>>, | |
| DISABLED <TypeKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordType>> | |
| > | |
| >; | |
| using LexicalAnalyzer = | |
| Lexer< | |
| LexingAutomaton__final, | |
| LexStateToTokenMapper__final, | |
| TokenKwrdCategorizer__final | |
| >; | |
| LexStateToTokenMapper__final mapper; | |
| std::cout | |
| << "mapping result: " | |
| << static_cast<int>(mapper.find_target(LexState::Identifier)) | |
| << "\n"; | |
| std::string content = generate_lexer_stress_test(); | |
| LexicalAnalyzer lexer; | |
| auto start = std::chrono::steady_clock::now(); | |
| std::vector<Token> tokens = | |
| lexer.tokenize<CharReader, PositionTracker>(content); | |
| auto endA = std::chrono::steady_clock::now(); | |
| auto durationA = | |
| std::chrono::duration_cast<std::chrono::nanoseconds>(endA - start); | |
| std::cout | |
| << "[Tokenization] " | |
| << durationA.count() | |
| << " ns\n"; | |
| using test_configuration_static_dfa = | |
| generate_expanded_dfa_config_t< | |
| GenericConfigurationEntry< | |
| true, | |
| nttp_to_type<LexState::Start>, | |
| charset<'a', 'b', 'c'>, | |
| nttp_to_type<LexState::Newline> | |
| > | |
| >; | |
| LexingAutomaton__final dfaf{}; | |
| std::cout | |
| << "[dfaf -> before] result: " | |
| << static_cast<int>(dfaf.get_current_state()) | |
| << "\n"; | |
| dfaf.step('d'); | |
| std::cout | |
| << "[dfaf -> after] result: " | |
| << static_cast<int>(dfaf.get_current_state()) | |
| << "\n"; | |
| std::cout | |
| << "Generation result: " | |
| << type_name<test_configuration_static_dfa>() | |
| << "\n"; |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment