Skip to content

Instantly share code, notes, and snippets.

@unrays
Last active September 15, 2026 14:21
Show Gist options
  • Select an option

  • Save unrays/fb528b74d0bbcc6e6d53300fafab11de to your computer and use it in GitHub Desktop.

Select an option

Save unrays/fb528b74d0bbcc6e6d53300fafab11de to your computer and use it in GitHub Desktop.
COMPILE-TIME LEXER ASSEMBLY AND DFA CONFIGURATION - This code comes from an unfinished project of mine. I found certain parts quite interesting, so I extracted a few of them and turned them into gists. This code is for educational purposes and is not functional, as it requires the rest of the architecture to work.
// Copyright (c) July 2026 Félix-Olivier Dumas. All rights reserved.
// Licensed under the terms described in the LICENSE file
#if !defined(__INTELLISENSE__)
using LexingAutomaton__final = StaticDFA<
generate_expanded_dfa_config_t<
ENABLED <nttp_to_type<LexState::Start>, charset_alpha, nttp_to_type<LexState::Identifier>>,
ENABLED <nttp_to_type<LexState::Start>, charset_digits, nttp_to_type<LexState::Number>>,
ENABLED <nttp_to_type<LexState::Start>, charset<':'>, nttp_to_type<LexState::DelimiterColon>>,
ENABLED <nttp_to_type<LexState::Start>, charset<';'>, nttp_to_type<LexState::DelimiterSemi>>,
ENABLED <nttp_to_type<LexState::Start>, charset<','>, nttp_to_type<LexState::DelimiterComma>>,
ENABLED <nttp_to_type<LexState::Start>, charset<'('>, nttp_to_type<LexState::DelimiterLParen>>,
ENABLED <nttp_to_type<LexState::Start>, charset<')'>, nttp_to_type<LexState::DelimiterRParen>>,
ENABLED <nttp_to_type<LexState::Start>, charset<'{'>, nttp_to_type<LexState::DelimiterLCurly>>,
ENABLED <nttp_to_type<LexState::Start>, charset<'}'>, nttp_to_type<LexState::DelimiterRCurly>>,
ENABLED <nttp_to_type<LexState::Start>, charset<'['>, nttp_to_type<LexState::DelimiterLSquare>>,
ENABLED <nttp_to_type<LexState::Start>, charset<']'>, nttp_to_type<LexState::DelimiterRSquare>>,
ENABLED <nttp_to_type<LexState::Start>, charset<'<'>, nttp_to_type<LexState::DelimiterLAngle>>,
ENABLED <nttp_to_type<LexState::Start>, charset<'>'>, nttp_to_type<LexState::DelimiterRAngle>>,
ENABLED <nttp_to_type<LexState::Start>, charset<'\n'>, nttp_to_type<LexState::Newline>>,
ENABLED <nttp_to_type<LexState::Start>, charset_isdelimiter_pure, nttp_to_type<LexState::DelimiterOpaque>>,
ENABLED <nttp_to_type<LexState::Start>, charset_isoperator_pure, nttp_to_type<LexState::Operator>>,
ENABLED <nttp_to_type<LexState::Start>, charset_iswhitespace, nttp_to_type<LexState::Whitespace>>,
ENABLED <nttp_to_type<LexState::Start>, charset_ispreprocessor, nttp_to_type<LexState::Preprocessor>>,
ENABLED <nttp_to_type<LexState::Preprocessor>, charset_alpha, nttp_to_type<LexState::Preprocessor>>,
ENABLED <nttp_to_type<LexState::Identifier>, charset_alphanumeric, nttp_to_type<LexState::Identifier>>,
ENABLED <nttp_to_type<LexState::Number>, charset_digits, nttp_to_type<LexState::Number>>,
ENABLED <nttp_to_type<LexState::Operator>, charset_isoperator, nttp_to_type<LexState::Operator>>,
ENABLED <nttp_to_type<LexState::DelimiterColon>, charset<':'>, nttp_to_type<LexState::DelimiterColon>>,
DISABLED <nttp_to_type<LexState::Whitespace>, charset_iswhitespace, nttp_to_type<LexState::Whitespace>>
>
>;
#else
using LexingAutomaton__final = StaticDFA<
StaticDfaTransitions<
ENABLED <nttp_to_type<LexState::Start>, nttp_to_type<'\n'>, nttp_to_type<LexState::Newline>>
>
>;
#endif
using LexStateToTokenMapper__final = EnumMapper<
EnumMapperConfiguration<
ENABLED <nttp_to_type<LexState::Identifier>, nttp_to_type<TokenKind::Identifier>>,
ENABLED <nttp_to_type<LexState::DelimiterOpaque>, nttp_to_type<TokenKind::DelimiterOpaque>>,
ENABLED <nttp_to_type<LexState::Operator>, nttp_to_type<TokenKind::Operator>>,
ENABLED <nttp_to_type<LexState::DelimiterColon>, nttp_to_type<TokenKind::DelimiterColon>>,
ENABLED <nttp_to_type<LexState::DelimiterSemi>, nttp_to_type<TokenKind::DelimiterSemicolon>>,
ENABLED <nttp_to_type<LexState::DelimiterComma>, nttp_to_type<TokenKind::DelimiterComma>>,
ENABLED <nttp_to_type<LexState::DelimiterLParen>, nttp_to_type<TokenKind::DelimiterLParen>>,
ENABLED <nttp_to_type<LexState::DelimiterRParen>, nttp_to_type<TokenKind::DelimiterRParen>>,
ENABLED <nttp_to_type<LexState::DelimiterLCurly>, nttp_to_type<TokenKind::DelimiterLCurly>>,
ENABLED <nttp_to_type<LexState::DelimiterRCurly>, nttp_to_type<TokenKind::DelimiterRCurly>>,
ENABLED <nttp_to_type<LexState::DelimiterLSquare>, nttp_to_type<TokenKind::DelimiterLSquare>>,
ENABLED <nttp_to_type<LexState::DelimiterRSquare>, nttp_to_type<TokenKind::DelimiterRSquare>>,
ENABLED <nttp_to_type<LexState::DelimiterLAngle>, nttp_to_type<TokenKind::DelimiterLAngle>>,
ENABLED <nttp_to_type<LexState::DelimiterRAngle>, nttp_to_type<TokenKind::DelimiterRAngle>>,
ENABLED <nttp_to_type<LexState::Preprocessor>, nttp_to_type<TokenKind::Preprocessor>>,
ENABLED <nttp_to_type<LexState::Newline>, nttp_to_type<TokenKind::Newline>>,
ENABLED <nttp_to_type<LexState::Number>, nttp_to_type<TokenKind::Number>>,
CONDITIONAL <(0 == 0), nttp_to_type<LexState::Invalid>, nttp_to_type<TokenKind::Unknown>>
>
>;
using TokenKwrdCategorizer__final = TokenKeywordCategorizer<
TokenKeywordCategorizerConfiguration<
ENABLED <AccessKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordAccess>>,
ENABLED <AlignmentKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordAlignment>>,
ENABLED <ControlKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordControl>>,
ENABLED <ModifierKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordModifier>>,
DISABLED <QualifierKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordQualifier>>,
DISABLED <SpecifierKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordSpecifier>>,
DISABLED <TypeKeywordMatchingPolicy, nttp_to_type<TokenKind::KeywordType>>
>
>;
using LexicalAnalyzer =
Lexer<
LexingAutomaton__final,
LexStateToTokenMapper__final,
TokenKwrdCategorizer__final
>;
LexStateToTokenMapper__final mapper;
std::cout
<< "mapping result: "
<< static_cast<int>(mapper.find_target(LexState::Identifier))
<< "\n";
std::string content = generate_lexer_stress_test();
LexicalAnalyzer lexer;
auto start = std::chrono::steady_clock::now();
std::vector<Token> tokens =
lexer.tokenize<CharReader, PositionTracker>(content);
auto endA = std::chrono::steady_clock::now();
auto durationA =
std::chrono::duration_cast<std::chrono::nanoseconds>(endA - start);
std::cout
<< "[Tokenization] "
<< durationA.count()
<< " ns\n";
using test_configuration_static_dfa =
generate_expanded_dfa_config_t<
GenericConfigurationEntry<
true,
nttp_to_type<LexState::Start>,
charset<'a', 'b', 'c'>,
nttp_to_type<LexState::Newline>
>
>;
LexingAutomaton__final dfaf{};
std::cout
<< "[dfaf -> before] result: "
<< static_cast<int>(dfaf.get_current_state())
<< "\n";
dfaf.step('d');
std::cout
<< "[dfaf -> after] result: "
<< static_cast<int>(dfaf.get_current_state())
<< "\n";
std::cout
<< "Generation result: "
<< type_name<test_configuration_static_dfa>()
<< "\n";
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment