From 37798483554ff38ba55136f1cc804a80e853f629 Mon Sep 17 00:00:00 2001 From: Kittycannon Date: Sun, 2 Aug 2026 19:04:52 -0600 Subject: [PATCH] time to test --- samples/assembly.ebnf | 8 +- src/spider/compiler/assembler/AsmEBNF.cpp | 146 +++++++++++++---- src/spider/compiler/assembler/AsmEBNF.hpp | 2 + src/spider/compiler/text/Token.cpp | 191 ++++++++++++++-------- src/spider/compiler/text/Token.hpp | 115 +++++++------ 5 files changed, 310 insertions(+), 152 deletions(-) diff --git a/samples/assembly.ebnf b/samples/assembly.ebnf index 38d97a4..85c399b 100644 --- a/samples/assembly.ebnf +++ b/samples/assembly.ebnf @@ -31,10 +31,10 @@ exponent_marker = "e" | "E" ; exponent = exponent_marker , [ sign ] , digit , { digit } ; decimal_lit = [ sign ] , digit , { digit } , [ "B" | "S" | "I" | "L" ] ; -float_lit = [ sign ] , ( - ( digit , { digit } , "." , digit , { digit } , [ exponent ] ) | - ( "." , digit , { digit } , [ exponent ] ) | - ( digit , { digit } , exponent ) +float_lit = [ sign ] , ( + ( digit , { digit } , "." , digit , { digit } , [ exponent ] ) | + ( "." , digit , { digit } , [ exponent ] ) | + ( digit , { digit } , exponent ) ) , [ "F" | "D" ] ; hex_lit = [ sign ] , "0x" , hex_digit , { hex_digit } ; diff --git a/src/spider/compiler/assembler/AsmEBNF.cpp b/src/spider/compiler/assembler/AsmEBNF.cpp index 159c700..c49bec8 100644 --- a/src/spider/compiler/assembler/AsmEBNF.cpp +++ b/src/spider/compiler/assembler/AsmEBNF.cpp @@ -2,12 +2,18 @@ namespace spider::asm_ebnf { + // Token Factory + + static TokenFactory tf; + + // Char Functions + bool isUTF8Alpha(u32 ch) { - return false; + return (u32('a') <= ch && ch <= u32('z')) || (u32('A') <= ch && ch <= u32('Z')); } bool isWhithespaceCharNotCrLf(u32 ch) { - return false; + return ch == u32(' '); } bool isUTF8CharNotCrLf(u32 ch) { @@ -22,37 +28,117 @@ namespace spider::asm_ebnf { return ch != u32('"'); } - LitToken numbers[] = { - "0", - "1","2","3", - "4","5","6", - "7","8","9", - }; - - LitToken hex_digits[][2] = { - {"A", "a"}, - {"B", "b"}, - {"C", "c"}, - {"D", "d"}, - {"E", "e"}, - {"F", "f"}, - }; + // (* Characters & Basic Predicates *) + static const Token* letter = tf.fn(isUTF8Alpha); + static const Token* digit = tf.choice("0123456789"); + static const Token* alpha_num_char = tf.choice({ letter, digit }); - LitToken new_line[] = { "\r\n", "\r", "\n" }; + static const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef"); + static const Token* octal_digit = tf.choice("01234567"); + static const Token* binary_digit = tf.choice("01"); - LitToken symbols[] = { - "\\", "\'", "\"", - "_" , ";" , "(" , - "#", - "$" , "." , "+" , "-", ",", ")", "@", ":", - }; + static const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf); + static const Token* ws_optional = tf.rep(ws_char); + static const Token* whitespace = tf.seq({ ws_char, tf.rep(ws_char) }); + static const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }); + static const Token* utf8_char = tf.fn(isUTF8CharNotCrLf); - LitToken lit_letter[] = { - "x", "c", "b" - }; + static const Token* char_escape = tf.seq({ tf["\\"], utf8_char }); + static const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) }); + static const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] }); - LitToken type_letter[] = { - "B", "S", "I", "L", "F", "D" - }; + static const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) }); + static const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }); + + // (* Literals *) + static const Token* identifier = tf.seq({ + tf.choice({ letter, tf["_"] }), + tf.rep(tf.choice({ alpha_num_char, tf["_"] })) + }); + + static const Token* comment = tf.seq({ tf[";"], tf.rep(utf8_char) }); + + static const Token* sign = tf.choice("+-"); + static const Token* exponent_marker = tf.choice("eE"); + static const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) }); + + static const Token* decimal_lit = tf.seq({ + tf.opt(sign), + digit, + tf.rep(digit), + tf.opt(tf.choice("BSIL")) + }); + + static const Token* float_lit = tf.seq({ + tf.opt(sign), + tf.choice({ + tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }), + tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }), + tf.seq({ digit, tf.rep(digit), exponent }) + }), + tf.opt(tf.choice("FD")) + }); + + static const Token* hex_lit = tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }); + static const Token* octal_lit = tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }); + static const Token* binary_lit = tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }); + + static const Token* literal = tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }); + static const Token* literal_cast = tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }); + static const Token* literal_decl = tf.choice({ literal, literal_cast }); + + // (* Operands *) + static const Token* register_tok = tf.seq({ tf["R"], alpha_num_char }); + + static const Token* addrm_ind = tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }); + static const Token* addrm_ptr = tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }); + + static const Token* addrm_idx = tf.seq({ + tf["["], ws_optional, register_tok, ws_optional, + tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] + }); + + static const Token* addrm_sca = tf.seq({ + tf["["], ws_optional, register_tok, ws_optional, + tf["+"], ws_optional, register_tok, ws_optional, + tf["*"], ws_optional, literal_decl, ws_optional, tf["]"] + }); + + static const Token* addrm_dis = tf.seq({ + tf["["], ws_optional, register_tok, ws_optional, + tf["+"], ws_optional, register_tok, ws_optional, + tf["*"], ws_optional, literal_decl, ws_optional, + tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] + }); + + static const Token* addr_modes = tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }); + static const Token* operand = tf.choice({ register_tok, identifier, literal_decl, addr_modes }); + + // (* Generalized Instructions *) + + static const Token* opcode = tf.seq({ letter, tf.rep(alpha_num_char) }); + static const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }); + static const Token* instruction = tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) }); + + // (* Added Preprocessor, Annotation *) + + static const Token* annotation_named = tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }); + static const Token* annotation_arg = tf.choice({ annotation_named, literal_decl }); + static const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }); + static const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }); + static const Token* annotation = tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }); + + static const Token* preprocessor_val = tf.choice({ identifier, string_lit }); + static const Token* preprocessor = tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }); + + // (* Line Structure & Program *) + + static const Token* label = tf.seq({ identifier, tf[":"] }); + static const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }); + static const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }); + static const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction }); + static const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }); + static const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }); + static const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) }); } diff --git a/src/spider/compiler/assembler/AsmEBNF.hpp b/src/spider/compiler/assembler/AsmEBNF.hpp index 9c0523a..ebacb69 100644 --- a/src/spider/compiler/assembler/AsmEBNF.hpp +++ b/src/spider/compiler/assembler/AsmEBNF.hpp @@ -6,4 +6,6 @@ namespace spider::asm_ebnf { extern LitToken letter; + void createTokens(); + } diff --git a/src/spider/compiler/text/Token.cpp b/src/spider/compiler/text/Token.cpp index 8be0d20..7ba6509 100644 --- a/src/spider/compiler/text/Token.cpp +++ b/src/spider/compiler/text/Token.cpp @@ -1,25 +1,66 @@ #include "Token.hpp" +#include + namespace spider { // ============================================================================ - // Token Implementation + // Token Factory // ============================================================================ - SeqToken Token::operator&(const Token& tok) { - return SeqToken({ tok, *this }); + Token* TokenFactory::lit(std::string_view text) { + auto p = std::make_unique(text); + auto t = p.get(); + lit_cache.emplace(text, std::move(p)); + return t; } - OrToken Token::operator|(const Token& tok) { - return OrToken({ tok, *this }); + Token* TokenFactory::operator[](std::string_view text) { + return lit(text); } - OptToken Token::operator~() { - return OptToken(*this); + Token* TokenFactory::fn(FnTokenFn predicate) { + uptr p = std::make_unique(predicate); + auto t = p.get(); + arena.emplace_back(std::move(p)); + return t; } - RepToken Token::operator*() { - return RepToken(*this); + Token* TokenFactory::seq(const vector& tokens) { + uptr p = std::make_unique(tokens); + auto t = p.get(); + arena.emplace_back(std::move(p)); + return t; + } + + Token* TokenFactory::choice(std::string_view opts) { + vector toks; + for(char c : opts) { + std::string s = std::string(1, c); + toks.push_back(lit(s)); + } + return choice(toks); + } + + Token* TokenFactory::choice(const vector& tokens) { + uptr p = std::make_unique(tokens); + auto t = p.get(); + arena.emplace_back(std::move(p)); + return t; + } + + Token* TokenFactory::opt(const Token* target) { + uptr p = std::make_unique(target); + auto t = p.get(); + arena.emplace_back(std::move(p)); + return t; + } + + Token* TokenFactory::rep(const Token* target) { + uptr p = std::make_unique(target); + auto t = p.get(); + arena.emplace_back(std::move(p)); + return t; } // ============================================================================ @@ -30,57 +71,64 @@ namespace spider { if (!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!"); } - LitToken::LitToken(const char* lit) : LitToken(std::string_view(lit)) {} - LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {} TokenResult LitToken::test(TextReader& ctx) const { - if(ctx.eat(literal)) return { true, literal }; - return { false, {} }; + if (ctx.eat(literal)) return { .success = true, .match = literal }; + return { .success = false }; } FnToken::FnToken(FnTokenFn chfn) : fn(chfn) {} TokenResult FnToken::test(TextReader& ctx) const { - std::u32string acc; - fn() - return { false, {} }; + TokenResult r; + + for (;;) { + auto i = ctx.push(); + auto c = ctx.current(); + + if (!c || !fn(*c)) { + ctx.pop(i); + break; + } + + r.match += *c; + ctx.nextChar(); + } + + r.success = !r.match.empty(); + return r; } // ============================================================================ // SeqToken Implementation - // ============================================================================ + // ============================================================================30520370 - SeqToken::SeqToken(const ilist>& list) : tokens(list) {} + SeqToken::SeqToken(const vector& tokens) : tokens(tokens) {} TokenResult SeqToken::test(TextReader& ctx) const { // this is a common branch point - std::u32string acc; - auto tri = ctx.push(); + TokenResult r; + isize i = ctx.push(); // All matching steps within a sequence must pass consecutively. for (const auto& token_ref : tokens) { - TokenResult res = token_ref.get().test(ctx); + TokenResult res = token_ref->test(ctx); if (!res.success) { // Strict ACID Transaction: Roll back context pointer entirely // if any nested condition in the sequence fails. - ctx.pop(tri); - return { false, {} }; + ctx.pop(i); + return { .success = false }; } // Piecewise accumulation of individual matching sub-tokens - acc += res.match; + //r.match += res.match; + r.child.push_back(res); } - return { true, acc }; - } - - SeqToken SeqToken::operator&(const Token& tok) { - // Intrusive chaining optimization: Appends the next token directly into the existing - // registry vector instead of nesting structures, keeping the layout flattened. - tokens.push_back(std::cref(tok)); - return *this; + r.success = true; + return r; } @@ -88,42 +136,33 @@ namespace spider { // OrToken Implementation // ============================================================================ - OrToken::OrToken(const ilist>& list) : tokens(list) {} + OrToken::OrToken(const vector& tokens) : tokens(tokens) {} TokenResult OrToken::test(TextReader& ctx) const { - // this is a common branch point - auto tri = ctx.push(); - // All matching steps within a sequence must pass consecutively. + auto i = ctx.push(); + for (const auto& token_ref : tokens) { // Short-circuit branch: return immediately on first valid choice match - TokenResult res = token_ref.get().test(ctx); + TokenResult res = token_ref->test(ctx); if (res.success) return res; // Backtrack isolation: Reset the cursor position before testing the next alternative path - ctx.pop(tri); + ctx.pop(i); } - return { false, {} }; + return { .success = false }; } - OrToken OrToken::operator|(const Token& tok) { - // Intrusive grouping layout optimization: - // flattens alternative tokens at code evaluation time. - tokens.push_back(tok); - return *this; - } - - // ============================================================================ // OptToken Implementation // ============================================================================ - OptToken::OptToken(const Token& t) : target(t) {} + OptToken::OptToken(const Token* t) : target(t) {} TokenResult OptToken::test(TextReader& ctx) const { auto tri = ctx.push(); - TokenResult res = target.test(ctx); + TokenResult res = target->test(ctx); if (res.success) { return res; // Option matched exactly 1 instance successfully @@ -132,46 +171,58 @@ namespace spider { // Recovery path: If sub-rule fails, clean up the dirty state mutation // and successfully return an empty match payload (0 instances). ctx.pop(tri); - return { true, U"" }; + return { .success = false }; } - OptToken OptToken::operator~() { - // Redundant layer trap protection: returning self - // prevents wrapping an Optional in an Optional - return *this; - } - - // ============================================================================ // RepToken Implementation // ============================================================================ - RepToken::RepToken(const Token& t) : target(t) {} + RepToken::RepToken(const Token* t) : target(t) {} TokenResult RepToken::test(TextReader& ctx) const { - std::u32string acc; - auto tri = ctx.push(); + TokenResult r; - for(;;) { - TokenResult res = target.test(ctx); + for (;;) { + auto i = ctx.push(); + TokenResult res = target->test(ctx); // Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching // rule succeeded without advancing the buffer index (prevents dynamic parsing lockups). - if(!res.success || tri == ctx.push()) { - ctx.pop(tri); + if (!res.success || i == ctx.push()) { + ctx.pop(i); break; } - acc += res.match; + //r.match += res.match; + r.child.push_back(res); } // Repetition rules (* token) always evaluate to successful // completion state, even with 0 matches. - return { true, acc }; + r.success = true; + return r; } - RepToken RepToken::operator*() { - // Redundant layer trap protection: returning self prevents wrapping - // a Repetition rule inside a Repetition rule - return *this; + // Tagged Token + + TagToken::TagToken(const Token* t, std::string_view tag, bool doflatten) + : target(t), tag_name(tag), flatten(doflatten) { } + + TokenResult TagToken::test(TextReader& ctx) const { + auto r = target->test(ctx); + if (!r.success) return r; + + // Set tag of this result + r.tag = tag_name; + + if(flatten) { + r.match = U""; + for(auto e : r.child) { + r.match += e.match; + } + r.child.clear(); + } + + return r; } } diff --git a/src/spider/compiler/text/Token.hpp b/src/spider/compiler/text/Token.hpp index d797f0a..63da9f8 100644 --- a/src/spider/compiler/text/Token.hpp +++ b/src/spider/compiler/text/Token.hpp @@ -13,7 +13,9 @@ namespace spider { struct TokenResult { /** @brief Indicates if the token composition successfully matched the input boundary. */ - bool success; + bool success = false; + + optional tag; /** * @brief Holds the deep-copied UTF-32 matching substring upon victory. @@ -21,13 +23,14 @@ namespace spider { */ std::u32string match; + vector child; + }; - // Forward declarations required by the abstract interface for operator returns. - class SeqToken; - class OrToken; - class OptToken; - class RepToken; + class Token; + class TokenFactory; + + using FnTokenFn = std::function; /** * @brief Pure virtual base class defining the EBNF combinator node contract. @@ -50,31 +53,46 @@ namespace spider { */ virtual TokenResult test(TextReader& ctx) const = 0; + }; + + class TokenFactory { + private: + // Arena owning all created tokens + std::vector> arena; + + // Deduplication caches + std::unordered_map lit_cache; + public: - /** - * @brief Chains this token and another sequentially. - * @return A temporary structural bridge matching both tokens sequentially. - */ - virtual SeqToken operator&(const Token& tok); + TokenFactory() = default; - /** - * @brief Combines this token and another under alternation. - * @return A structural bridge matching either this token or the fallback selection. - */ - virtual OrToken operator|(const Token& tok); + // Prevent copying to maintain valid internal pointers + TokenFactory(const TokenFactory&) = delete; - /** - * @brief Wraps this node in an optional layout rule. - * @return A structure matching zero or one instances of this current node. - */ - virtual OptToken operator~(); + TokenFactory& operator=(const TokenFactory&) = delete; - /** - * @brief Wraps this node in a repetitive loop match framework. - * @return A structure matching zero or more occurrences of this current node. - */ - virtual RepToken operator*(); + public: + + // --- Primitive Constructors --- + + Token* lit(std::string_view text); + + Token* operator[](std::string_view text); + + Token* fn(FnTokenFn predicate); + + // --- Combinator Constructors --- + + Token* seq(const vector& tokens); + + Token* choice(std::string_view opts); + + Token* choice(const vector& tokens); + + Token* opt(const Token* target); + + Token* rep(const Token* target); }; @@ -93,8 +111,6 @@ namespace spider { */ LitToken(std::string_view lit); - LitToken(const char* lit); - /** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */ explicit LitToken(std::u32string lit); @@ -106,8 +122,6 @@ namespace spider { TokenResult test(TextReader& ctx) const override; }; - using FnTokenFn = std::function; - /** * @brief Function based token */ @@ -135,11 +149,10 @@ namespace spider { * @brief Internal contiguous layout registry storing lightweight, zero-overhead references. * @details Avoids heap allocation penalties by referencing static instances immutably. */ - vector> tokens; + vector tokens; public: - /** @brief Constructs a sequence container directly out of an inline brace-enclosed listing. */ - SeqToken(const ilist>& list); + SeqToken(const vector& tokens); public: /** @@ -149,8 +162,6 @@ namespace spider { */ TokenResult test(TextReader& ctx) const override; - /** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */ - SeqToken operator&(const Token& tok) override; }; /** @@ -159,11 +170,10 @@ namespace spider { class OrToken : public Token { private: /** @brief Ordered registry of possible alternate structural paths. */ - vector> tokens; + vector tokens; public: - /** @brief Constructs an alternation choice layout from brace-enclosed tokens. */ - OrToken(const ilist>& list); + OrToken(const vector& tokens); public: /** @@ -172,8 +182,6 @@ namespace spider { */ TokenResult test(TextReader& ctx) const override; - /** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */ - OrToken operator|(const Token& tok) override; }; /** @@ -182,11 +190,11 @@ namespace spider { class OptToken : public Token { private: /** @brief Read-only target node reference to test optional status against. */ - const Token& target; + const Token* target; public: /** @brief Binds the target node rule structural dependency layout wrapper. */ - explicit OptToken(const Token& t); + explicit OptToken(const Token* t); public: /** @@ -195,8 +203,6 @@ namespace spider { */ TokenResult test(TextReader& ctx) const override; - /** @brief Stub override providing standard compliance with the base Token interface signature. */ - OptToken operator~() override; }; /** @@ -205,11 +211,11 @@ namespace spider { class RepToken : public Token { private: /** @brief The base token node sequence layer evaluated in loops. */ - const Token& target; + const Token* target; public: /** @brief Binds the repeated structural blueprint node wrapper configuration. */ - explicit RepToken(const Token& t); + explicit RepToken(const Token* t); public: /** @@ -219,8 +225,21 @@ namespace spider { */ TokenResult test(TextReader& ctx) const override; - /** @brief Stub override providing standard compliance with the base Token interface signature. */ - RepToken operator*() override; + }; + + class TagToken : public Token { + private: + + const Token* target; + std::string tag_name; + bool flatten; + + public: + + TagToken(const Token* t, std::string_view tag, bool doflatten = false); + + TokenResult test(TextReader& ctx) const override; + }; }