diff --git a/src/spider/compiler/Compiler.cpp b/src/spider/compiler/Compiler.cpp index bd74202..dd2d113 100644 --- a/src/spider/compiler/Compiler.cpp +++ b/src/spider/compiler/Compiler.cpp @@ -7,74 +7,8 @@ namespace spider { -} - -// Test runner helper -void run_test(const std::string& name, const std::string& input) { - std::cout << "========================================\n"; - std::cout << " TEST: " << name << "\n"; - std::cout << "========================================\n"; - - spider::pos tracking_pos; - spider::utf8::hexdump(input.data(), input.size(), tracking_pos, std::cout); - std::cout << "\n"; -} - -void utf8sequences() { - // Permutation 1: Standard, valid mixed UTF-8 sequences (1, 2, 3, and 4 bytes) - // - 'A' -> 1 byte (U+0041) - // - 'ยข' (cents) -> 2 bytes (U+00A2) - // - 'โ‚ฌ' (euro) -> 3 bytes (U+20AC) - // - '๐ˆ' (gothic) -> 4 bytes (U+10348) - run_test("Valid Mixed Length Sequences", "A\xC2\xA2\xE2\x82\xAC\xF0\x90\x8D\x88"); - - // Permutation 2: Embedded Control Characters - // Should display mnemonics like (HT), (LF), (CR) without breaking formatting - run_test("ASCII Control Characters", "Text\tWith\r\nNewlines"); - - // Permutation 3: Invalid Lead Byte - // The byte 0xFF is structurally illegal under any UTF-8 definition. - // Expected behavior: Display single byte as INVALID LEAD, shift 1 byte over. - run_test("Invalid Lead Byte (0xFF)", "ABC\xFFXYZ"); - - // Permutation 4: Invalid Continuation Sequence - // A 3-byte header (\xE2) where the second byte (\x00) is a bad continuation. - // Expected behavior: Show the entire sequence up to 'm' bytes, flag as INVALID SEQUENCE. - run_test("Invalid Continuation Structure", std::string("Before \xE2\x00\xAC After", 16)); - - // Permutation 5: Truncated Sequence at End-of-Buffer - // A 4-byte emoji header (\xF0\x9F) but the string completely cuts off. - // Expected behavior: Display remaining space placeholders as '??' -> TRUNCATED SEQUENCE. - run_test("Truncated Sequence (Missing trailing bytes)", "Hello \xF0\x9F"); - - // Permutation 6: Overlong Encoding Security Vulnerability - // Attempting to write ASCII 'I' (normally 0x49) using 2 bytes: \xC1\x89 - // Expected behavior: Caught by constraints checks, flagged as INVALID SEQUENCE. - run_test("Security Hack: Overlong Encoding", "Safe\xC1\x89Hack"); - - // Permutation 7: Out-of-bounds / Restricted Ranges - // - \xED\xA0\x80 is a UTF-16 Surrogate (U+D800) - // - \xF4\x90\x80\x80 is outside valid Unicode space (> U+10FFFF) - // Expected behavior: Flagged securely as INVALID SEQUENCE. - run_test("Security Hack: Restricted Ranges (Surrogates & Out-of-bounds)", "Surrogate: \xED\xA0\x80 MaxBounds: \xF4\x90\x80\x80"); } int main() { - std::string test = "UTF-STR WITH EMOJIS ๐Ÿ˜€๐Ÿš€"; - // Extracted UTF-8 chars: - std::u32string out; - bool r = spider::utf8::toUTF32(test, out); - - std::cout << "INPUT: " << test << std::endl; - std::cout << "RESULT: " << int(r) << std::endl; - for(spider::u32 ch : out) { - std::cout << ch << " "; - } - std::cout << std::endl; - - std::cout << "Happy Day!" << std::endl; - spider::utf8::hexdump(test.data(), test.size(), spider::pos(), std::cout); - std::cout << std::endl; - utf8sequences(); return 0; } diff --git a/src/spider/compiler/assembler/AsmEBNF.cpp b/src/spider/compiler/assembler/AsmEBNF.cpp index c49bec8..c2e26a6 100644 --- a/src/spider/compiler/assembler/AsmEBNF.cpp +++ b/src/spider/compiler/assembler/AsmEBNF.cpp @@ -4,7 +4,7 @@ namespace spider::asm_ebnf { // Token Factory - static TokenFactory tf; + TokenFactory tf; // Char Functions @@ -29,47 +29,47 @@ namespace spider::asm_ebnf { } // (* Characters & Basic Predicates *) - static const Token* letter = tf.fn(isUTF8Alpha); - static const Token* digit = tf.choice("0123456789"); - static const Token* alpha_num_char = tf.choice({ letter, digit }); + const Token* letter = tf.fn(isUTF8Alpha); + const Token* digit = tf.choice("0123456789"); + const Token* alpha_num_char = tf.choice({ letter, digit }); - static const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef"); - static const Token* octal_digit = tf.choice("01234567"); - static const Token* binary_digit = tf.choice("01"); + const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef"); + const Token* octal_digit = tf.choice("01234567"); + const Token* binary_digit = tf.choice("01"); - static const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf); - static const Token* ws_optional = tf.rep(ws_char); - static const Token* whitespace = tf.seq({ ws_char, tf.rep(ws_char) }); - static const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }); - static const Token* utf8_char = tf.fn(isUTF8CharNotCrLf); + const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf); + const Token* ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true); + const Token* whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true); + const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }); + const Token* utf8_char = tf.fn(isUTF8CharNotCrLf); - static const Token* char_escape = tf.seq({ tf["\\"], utf8_char }); - static const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) }); - static const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] }); + const Token* char_escape = tf.seq({ tf["\\"], utf8_char }); + const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) }); + const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] }); - static const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) }); - static const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }); + const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) }); + const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }); // (* Literals *) - static const Token* identifier = tf.seq({ + const Token* identifier = tf.tag(tf.seq({ tf.choice({ letter, tf["_"] }), tf.rep(tf.choice({ alpha_num_char, tf["_"] })) - }); + }), "identifier", true); - static const Token* comment = tf.seq({ tf[";"], tf.rep(utf8_char) }); + const Token* comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true); - static const Token* sign = tf.choice("+-"); - static const Token* exponent_marker = tf.choice("eE"); - static const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) }); + const Token* sign = tf.choice("+-"); + const Token* exponent_marker = tf.choice("eE"); + const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) }); - static const Token* decimal_lit = tf.seq({ + const Token* decimal_lit = tf.tag(tf.seq({ tf.opt(sign), digit, tf.rep(digit), tf.opt(tf.choice("BSIL")) - }); + }), "decimal_lit", true); - static const Token* float_lit = tf.seq({ + const Token* float_lit = tf.tag(tf.seq({ tf.opt(sign), tf.choice({ tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }), @@ -77,68 +77,68 @@ namespace spider::asm_ebnf { tf.seq({ digit, tf.rep(digit), exponent }) }), tf.opt(tf.choice("FD")) - }); + }), "float_lit", true); - static const Token* hex_lit = tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }); - static const Token* octal_lit = tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }); - static const Token* binary_lit = tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }); + const Token* hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true); + const Token* octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true); + const Token* binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true); - static const Token* literal = tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }); - static const Token* literal_cast = tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }); - static const Token* literal_decl = tf.choice({ literal, literal_cast }); + const Token* literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal"); + const Token* literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast"); + const Token* literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl"); // (* Operands *) - static const Token* register_tok = tf.seq({ tf["R"], alpha_num_char }); + const Token* register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true); - static const Token* addrm_ind = tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }); - static const Token* addrm_ptr = tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }); + const Token* addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true); + const Token* addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true); - static const Token* addrm_idx = tf.seq({ + const Token* addrm_idx = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] - }); + }), "addrm_idx", true); - static const Token* addrm_sca = tf.seq({ + const Token* addrm_sca = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["+"], ws_optional, register_tok, ws_optional, tf["*"], ws_optional, literal_decl, ws_optional, tf["]"] - }); + }), "addrm_sca", true); - static const Token* addrm_dis = tf.seq({ + const Token* addrm_dis = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["+"], ws_optional, register_tok, ws_optional, tf["*"], ws_optional, literal_decl, ws_optional, tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] - }); + }), "addrm_dis", true); - static const Token* addr_modes = tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }); - static const Token* operand = tf.choice({ register_tok, identifier, literal_decl, addr_modes }); + const Token* addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm"); + const Token* operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand"); // (* Generalized Instructions *) - static const Token* opcode = tf.seq({ letter, tf.rep(alpha_num_char) }); - static const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }); - static const Token* instruction = tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) }); + const Token* opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true); + const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }); + const Token* instruction = tf.tag(tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction"); // (* Added Preprocessor, Annotation *) - static const Token* annotation_named = tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }); - static const Token* annotation_arg = tf.choice({ annotation_named, literal_decl }); - static const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }); - static const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }); - static const Token* annotation = tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }); + const Token* annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named"); + const Token* annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg"); + const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }); + const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }); + const Token* annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation"); - static const Token* preprocessor_val = tf.choice({ identifier, string_lit }); - static const Token* preprocessor = tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }); + const Token* preprocessor_val = tf.choice({ identifier, literal_decl }); + const Token* preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor"); // (* Line Structure & Program *) - static const Token* label = tf.seq({ identifier, tf[":"] }); - static const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }); - static const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }); - static const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction }); - static const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }); - static const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }); - static const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) }); + const Token* label = tf.tag(tf.seq({ identifier, tf[":"] }), "label"); + const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }); + const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }); + const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction }); + const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }); + const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }); + const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) }); } diff --git a/src/spider/compiler/assembler/AsmEBNF.hpp b/src/spider/compiler/assembler/AsmEBNF.hpp index ebacb69..26fafb4 100644 --- a/src/spider/compiler/assembler/AsmEBNF.hpp +++ b/src/spider/compiler/assembler/AsmEBNF.hpp @@ -4,8 +4,75 @@ namespace spider::asm_ebnf { - extern LitToken letter; + extern const Token* letter; + extern const Token* digit; + extern const Token* alpha_num_char; - void createTokens(); + extern const Token* hex_digit; + extern const Token* octal_digit; + extern const Token* binary_digit; + + extern const Token* ws_char; + extern const Token* ws_optional; + extern const Token* whitespace; + extern const Token* newline; + extern const Token* utf8_char; + + extern const Token* char_escape; + extern const Token* char_content; + extern const Token* char_lit; + + extern const Token* string_char; + extern const Token* string_lit; + + extern const Token* identifier; + extern const Token* comment; + + extern const Token* sign; + extern const Token* exponent_marker; + extern const Token* exponent; + + extern const Token* decimal_lit; + extern const Token* float_lit; + + extern const Token* hex_lit; + extern const Token* octal_lit; + extern const Token* binary_lit; + + extern const Token* literal; + extern const Token* literal_cast; + extern const Token* literal_decl; + + extern const Token* register_tok; + + extern const Token* addrm_ind; + extern const Token* addrm_ptr; + extern const Token* addrm_idx; + extern const Token* addrm_sca; + extern const Token* addrm_dis; + + extern const Token* addr_modes; + extern const Token* operand; + + extern const Token* opcode; + extern const Token* operand_list; + extern const Token* instruction; + + extern const Token* annotation_named; + extern const Token* annotation_arg; + extern const Token* annotation_args; + extern const Token* annotation_pars; + extern const Token* annotation; + + extern const Token* preprocessor_val; + extern const Token* preprocessor; + + extern const Token* label; + extern const Token* line_label; + extern const Token* line_annotation; + extern const Token* line_content; + extern const Token* line; + extern const Token* line_last; + extern const Token* program; } diff --git a/src/spider/compiler/text/Token.cpp b/src/spider/compiler/text/Token.cpp index 7ba6509..634bedd 100644 --- a/src/spider/compiler/text/Token.cpp +++ b/src/spider/compiler/text/Token.cpp @@ -63,6 +63,13 @@ namespace spider { return t; } + Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten = false) { + uptr p = std::make_unique(target, tagname, flatten); + auto t = p.get(); + arena.emplace_back(std::move(p)); + return t; + } + // ============================================================================ // LitToken Implementation // ============================================================================ diff --git a/src/spider/compiler/text/Token.hpp b/src/spider/compiler/text/Token.hpp index 63da9f8..b68e1df 100644 --- a/src/spider/compiler/text/Token.hpp +++ b/src/spider/compiler/text/Token.hpp @@ -94,6 +94,8 @@ namespace spider { Token* rep(const Token* target); + Token* tag(const Token* target, std::string_view tagname, bool flatten = false); + }; /**