#include #include #include #include #include #include #include using namespace spider; // ============================================================================ // Compiler Entry Points // ============================================================================ namespace spider { TokenFactory& assemblyGrammar() { static TokenFactory grammar; static bool loaded = false; if (!loaded) { asm_ebnf::initTokens(grammar); loaded = true; } return grammar; } bool compileProgram(const std::string& source, ParseTree& out) { return out.parse(asm_ebnf::program, source); } } // ============================================================================ // Test Bookkeeping // ============================================================================ // Suite wide bookkeeping, so a failing case can never hide behind a clean exit code. static isize total_tests = 0; static isize failed_tests = 0; static void report(bool ok, const std::string& name) { ++total_tests; if (ok) { std::cout << "[PASS] " << name << "\n"; } else { ++failed_tests; std::cerr << "[FAIL] " << name << "\n"; } } // Inline evaluator that executes the reader and prints standard output static bool run_test_case(const Token* token, const std::string& input, bool expectedSuccess, const std::u32string& expectedMatch = U"") { StringTextReader reader(input); TokenResult res = token->test(reader); bool status_ok = (res.success == expectedSuccess); bool match_ok = (!expectedSuccess) || expectedMatch.empty() || (res.flatMatch() == expectedMatch); ++total_tests; if (status_ok && match_ok) { std::cout << "[PASS] Input: \"" << input << "\" -> " << (res.success ? "SUCCESS" : "FAILURE") << (res.success ? (" (Matched: \"" + unicode::toUTF8(res.flatMatch()) + "\")") : "") << "\n"; return true; } std::cerr << "[FAIL] Input: \"" << input << "\"\n" << " Expected Success: " << (expectedSuccess ? "true" : "false") << ", Got: " << (res.success ? "true" : "false") << "\n"; if (expectedSuccess && !expectedMatch.empty()) { std::cerr << " Expected Match: \"" << unicode::toUTF8(expectedMatch) << "\", Got: \"" << unicode::toUTF8(res.flatMatch()) << "\"\n"; } ++failed_tests; return false; } // ============================================================================ // Core Combinator & Grammar Tests // ============================================================================ void test_primitives_and_literals(TokenFactory& tf) { std::cout << "\n--- Testing Primitives & Literals ---\n"; Token* hello = tf.lit("hello"); run_test_case(hello, "hello world", true, U"hello"); run_test_case(hello, "hell", false); run_test_case(asm_ebnf::letter, "a", true); run_test_case(asm_ebnf::letter, "f", true); run_test_case(asm_ebnf::letter, "A", true); run_test_case(asm_ebnf::letter, "Y", true); run_test_case(asm_ebnf::letter, "9", false); run_test_case(asm_ebnf::digit, "0", true); run_test_case(asm_ebnf::digit, "7", true); run_test_case(asm_ebnf::digit, "x", false); run_test_case(asm_ebnf::hex_digit, "F", true); run_test_case(asm_ebnf::hex_digit, "g", false); } void test_choice_and_seq(TokenFactory& tf) { std::cout << "\n--- Testing Choice & Sequence ---\n"; Token* seq_test = tf.seq({ tf["foo"], tf["bar"] }); run_test_case(seq_test, "foobar", true, U"foobar"); run_test_case(seq_test, "foobaz", false); Token* choice_test = tf.choice({ tf["apple"], tf["banana"] }); run_test_case(choice_test, "banana", true, U"banana"); run_test_case(choice_test, "cherry", false); } void test_opt_and_rep(TokenFactory& tf) { std::cout << "\n--- Testing Optional & Repeat ---\n"; Token* opt_test = tf.seq({ tf.opt(tf["+"]), asm_ebnf::digit }); run_test_case(opt_test, "+5", true, U"+5"); run_test_case(opt_test, "5", true, U"5"); Token* rep_digits = tf.rep(asm_ebnf::digit); run_test_case(rep_digits, "12345abc", true, U"12345"); run_test_case(rep_digits, "abc", true, U""); } void test_literals(TokenFactory& tf) { std::cout << "\n--- Testing Grammatical Literals ---\n"; run_test_case(asm_ebnf::identifier, "valid_var1", true, U"valid_var1"); run_test_case(asm_ebnf::identifier, "_private", true, U"_private"); run_test_case(asm_ebnf::identifier, "123invalid", false); run_test_case(asm_ebnf::decimal_lit, "1234", true); run_test_case(asm_ebnf::decimal_lit, "-567L", true); run_test_case(asm_ebnf::hex_lit, "0x1A2B", true); run_test_case(asm_ebnf::octal_lit, "0c755", true); run_test_case(asm_ebnf::binary_lit, "0b10101", true); run_test_case(asm_ebnf::float_lit, "3.14159F", true); run_test_case(asm_ebnf::float_lit, "1e-10D", true); run_test_case(asm_ebnf::char_lit, "'a'", true); run_test_case(asm_ebnf::char_lit, "'\\n'", true); run_test_case(asm_ebnf::string_lit, "\"Hello World\"", true); run_test_case(asm_ebnf::string_lit, "\"Escape \\\" Test\"", true); } void test_addressing_modes(TokenFactory& tf) { std::cout << "\n--- Testing Addressing Modes ---\n"; run_test_case(asm_ebnf::register_tok, "R0", true); run_test_case(asm_ebnf::register_tok, "R15", false); // R15 does not exist! run_test_case(asm_ebnf::addrm_ind, "[ 0x1000 ]", true); run_test_case(asm_ebnf::addrm_ptr, "[R1]", true); run_test_case(asm_ebnf::addrm_idx, "[R1 + 4]", true); run_test_case(asm_ebnf::addrm_sca, "[R1 + R2 * 4]", true); run_test_case(asm_ebnf::addrm_dis, "[ R1 + R2 * 4 + 16 ]", true); } void test_instructions_and_lines(TokenFactory& tf) { std::cout << "\n--- Testing Instructions & Lines ---\n"; run_test_case(asm_ebnf::instruction, "NOP", true); run_test_case(asm_ebnf::instruction, "MOV R1, [R2 + 4]", true); run_test_case(asm_ebnf::instruction, "ADD R0, 100", true); run_test_case(asm_ebnf::preprocessor, "#define MAX_BUF", true); run_test_case(asm_ebnf::annotation, "@inline", true); run_test_case(asm_ebnf::annotation, "@align(4)", true); run_test_case(asm_ebnf::annotation, "@section(name=\"text\", flags=1)", true); run_test_case(asm_ebnf::line, "main: MOV R0, R1 ; copy reg\n", true); run_test_case(asm_ebnf::line, " @deprecated NOP\n", true); run_test_case(asm_ebnf::line, "; only a comment line\n", true); } void test_full_program(TokenFactory& tf) { std::cout << "\n--- Testing Full Program Parser ---\n"; std::string asm_code = "#include stdio\n" "\n" "start:\n" " MOV R1, 0x20 ; Load constant\n" " @align(16) ADD R1, [R2 + R3 * 8 + 4]\n" " JMP start\n"; ParseTree tree; report(compileProgram(asm_code, tree), "full assembly program parses completely"); } void test_parse_tree(TokenFactory& tf) { std::cout << "\n--- Testing Parse Tree Inspection ---\n"; const std::string asm_code = "#include stdio\n" "\n" "start:\n" " MOV R1, 0x20 ; Load constant\n" " @align(16) ADD R1, [R2 + R3 * 8 + 4]\n" " JMP start\n"; ParseTree tree; report(compileProgram(asm_code, tree), "program becomes a tree of nodes"); ParseNode* root = tree.root(); report(root != nullptr, "tree exposes a root node"); if (root == nullptr) return; report(root->hasTag("program"), "root node is tagged as program"); report(root->textUtf8() == asm_code, "root text covers the whole source"); report(root->childCount() > 0, "root holds child nodes"); vector lines = root->findAll("line"); report(lines.size() == 6, "program holds one node per line"); if (lines.size() == 6) { report(lines.front()->parentNode() == root, "a line knows its parent"); report(lines.front()->depth() == 1, "lines sit one level under the root"); report(lines.front()->nextSibling() == lines[1], "sibling walk reaches the next line"); report(lines[1]->previousSibling() == lines.front(), "sibling walk goes back"); report(lines.back()->nextSibling() == nullptr, "the last line has no next sibling"); ParseNode* preprocessor = lines.front()->firstChild("preprocessor"); report(preprocessor != nullptr, "first line holds a preprocessor child"); report(preprocessor != nullptr && preprocessor->textUtf8() == "#include stdio", "preprocessor text is readable"); report(lines[2]->firstChild("line_label") != nullptr, "label line is recognized"); ParseNode* label = lines[2]->find("label"); report(label != nullptr && label->textUtf8() == "start:", "label text is readable"); ParseNode* mov = lines[3]->find("instruction"); report(mov != nullptr, "instruction is found inside its line"); if (mov != nullptr) { report(mov->firstChild("opcode") != nullptr, "instruction exposes its opcode child"); ParseNode* operands = mov->firstChild("operand_list"); report(operands != nullptr && operands->childCount() == 2, "operand list exposes both operands"); ParseNode* hexlit = mov->find("hex_lit"); report(hexlit != nullptr && hexlit->textUtf8() == "0x20", "hex literal text is readable"); } ParseNode* comment = lines[3]->firstChild("comment"); report(comment != nullptr && comment->textUtf8() == "; Load constant", "trailing comment is captured"); report(lines[4]->firstChild("line_annotation") != nullptr, "annotation line is recognized"); ParseNode* displacement = lines[4]->find("addrm_dis"); report(displacement != nullptr && displacement->textUtf8() == "[R2 + R3 * 8 + 4]", "displacement operand is readable"); report(tree.count("register") == 4, "every register of the program is reachable"); } ParseTree single; report(single.parse(asm_ebnf::instruction, "MOV R1, [R2 + 4]"), "a single instruction parses on its own"); ParseNode* instruction = single.root(); report(instruction != nullptr && instruction->hasTag("instruction"), "instruction root is tagged"); if (instruction != nullptr) { ParseNode* operands = instruction->firstChild("operand_list"); report(operands != nullptr && operands->childCount() == 2, "operand list holds two operands"); ParseNode* second = operands == nullptr ? nullptr : operands->childAt(1); report(second != nullptr && second->textUtf8() == "[R2 + 4]", "operand text is readable"); report(second != nullptr && second->ancestor("instruction") == instruction, "an operand can climb back to its instruction"); } std::cout << "\n--- Parsed Tree ---\n" << tree.toString() << "\n"; } // ============================================================================ // Main Execution // ============================================================================ int main() { TokenFactory tf; assemblyGrammar(); std::cout << "Running Token Framework Tests...\n"; test_primitives_and_literals(tf); test_choice_and_seq(tf); test_opt_and_rep(tf); test_literals(tf); test_addressing_modes(tf); test_instructions_and_lines(tf); test_full_program(tf); test_parse_tree(tf); std::cout << "\n========================================\n"; std::cout << "Test Results: " << (total_tests - failed_tests) << "/" << total_tests << " passed.\n"; std::cout << "========================================\n"; if (failed_tests > 0) { std::cerr << failed_tests << " test(s) FAILED.\n"; return 1; } std::cout << "All token framework tests passed successfully!\n"; return 0; }