diff --git a/src/spider/compiler/assembler/AsmEBNF.cpp b/src/spider/compiler/assembler/AsmEBNF.cpp index e69de29..159c700 100644 --- a/src/spider/compiler/assembler/AsmEBNF.cpp +++ b/src/spider/compiler/assembler/AsmEBNF.cpp @@ -0,0 +1,58 @@ +#include "AsmEBNF.hpp" + +namespace spider::asm_ebnf { + + bool isUTF8Alpha(u32 ch) { + return false; + } + + bool isWhithespaceCharNotCrLf(u32 ch) { + return false; + } + + bool isUTF8CharNotCrLf(u32 ch) { + return ch != u32('\r') && ch != u32('\n'); + } + + bool isUTF8CharLitCont(u32 ch) { + return ch != u32('\''); + } + + bool isUTF8StringLitCont(u32 ch) { + return ch != u32('"'); + } + + LitToken numbers[] = { + "0", + "1","2","3", + "4","5","6", + "7","8","9", + }; + + LitToken hex_digits[][2] = { + {"A", "a"}, + {"B", "b"}, + {"C", "c"}, + {"D", "d"}, + {"E", "e"}, + {"F", "f"}, + }; + + LitToken new_line[] = { "\r\n", "\r", "\n" }; + + LitToken symbols[] = { + "\\", "\'", "\"", + "_" , ";" , "(" , + "#", + "$" , "." , "+" , "-", ",", ")", "@", ":", + }; + + LitToken lit_letter[] = { + "x", "c", "b" + }; + + LitToken type_letter[] = { + "B", "S", "I", "L", "F", "D" + }; + +} diff --git a/src/spider/compiler/assembler/AsmEBNF.hpp b/src/spider/compiler/assembler/AsmEBNF.hpp index eda512e..9c0523a 100644 --- a/src/spider/compiler/assembler/AsmEBNF.hpp +++ b/src/spider/compiler/assembler/AsmEBNF.hpp @@ -2,8 +2,8 @@ #include -namespace spider { +namespace spider::asm_ebnf { - + extern LitToken letter; } diff --git a/src/spider/compiler/text/Token.cpp b/src/spider/compiler/text/Token.cpp index 03bdf7a..8be0d20 100644 --- a/src/spider/compiler/text/Token.cpp +++ b/src/spider/compiler/text/Token.cpp @@ -30,6 +30,8 @@ namespace spider { if (!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!"); } + LitToken::LitToken(const char* lit) : LitToken(std::string_view(lit)) {} + LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {} TokenResult LitToken::test(TextReader& ctx) const { @@ -37,6 +39,13 @@ namespace spider { return { false, {} }; } + FnToken::FnToken(FnTokenFn chfn) : fn(chfn) {} + + TokenResult FnToken::test(TextReader& ctx) const { + std::u32string acc; + fn() + return { false, {} }; + } // ============================================================================ // SeqToken Implementation diff --git a/src/spider/compiler/text/Token.hpp b/src/spider/compiler/text/Token.hpp index db2ac16..d797f0a 100644 --- a/src/spider/compiler/text/Token.hpp +++ b/src/spider/compiler/text/Token.hpp @@ -91,7 +91,9 @@ namespace spider { * @brief Constructs a literal rule by transforming a standard UTF-8 string view. * @throws std::runtime_error If incoming character boundaries contain invalid UTF-8 formatting. */ - explicit LitToken(std::string_view lit); + LitToken(std::string_view lit); + + LitToken(const char* lit); /** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */ explicit LitToken(std::u32string lit); @@ -104,6 +106,26 @@ namespace spider { TokenResult test(TextReader& ctx) const override; }; + using FnTokenFn = std::function; + + /** + * @brief Function based token + */ + class FnToken : public Token { + private: + + FnTokenFn fn; + + public: + + FnToken(FnTokenFn chfn); + + public: + + TokenResult test(TextReader& ctx) const override; + + }; + /** * @brief Evaluates an unrolled sequence of ordered grammatical rules sequentially (AndToken). */