297 lines
12 KiB
C++
297 lines
12 KiB
C++
#include <iostream>
|
|
|
|
#include <spider/compiler/Compiler.hpp>
|
|
#include <spider/compiler/common.hpp>
|
|
|
|
#include <spider/compiler/text/unicode.hpp>
|
|
#include <spider/compiler/text/TextReader.hpp>
|
|
|
|
#include <spider/compiler/assembler/AsmEBNF.hpp>
|
|
#include <spider/compiler/text/ParseTree.hpp>
|
|
|
|
using namespace spider;
|
|
|
|
// ============================================================================
|
|
// Compiler Entry Points
|
|
// ============================================================================
|
|
|
|
namespace spider {
|
|
|
|
TokenFactory& assemblyGrammar() {
|
|
static TokenFactory grammar;
|
|
static bool loaded = false;
|
|
if (!loaded) {
|
|
asm_ebnf::initTokens(grammar);
|
|
loaded = true;
|
|
}
|
|
return grammar;
|
|
}
|
|
|
|
bool compileProgram(const std::string& source, ParseTree& out) {
|
|
return out.parse(asm_ebnf::program, source);
|
|
}
|
|
|
|
}
|
|
|
|
// ============================================================================
|
|
// Test Bookkeeping
|
|
// ============================================================================
|
|
|
|
// Suite wide bookkeeping, so a failing case can never hide behind a clean exit code.
|
|
static isize total_tests = 0;
|
|
static isize failed_tests = 0;
|
|
|
|
static void report(bool ok, const std::string& name) {
|
|
++total_tests;
|
|
if (ok) {
|
|
std::cout << "[PASS] " << name << "\n";
|
|
} else {
|
|
++failed_tests;
|
|
std::cerr << "[FAIL] " << name << "\n";
|
|
}
|
|
}
|
|
|
|
// Inline evaluator that executes the reader and prints standard output
|
|
static bool run_test_case(const Token* token, const std::string& input, bool expectedSuccess, const std::u32string& expectedMatch = U"") {
|
|
StringTextReader reader(input);
|
|
TokenResult res = token->test(reader);
|
|
|
|
bool status_ok = (res.success == expectedSuccess);
|
|
bool match_ok = (!expectedSuccess) || expectedMatch.empty() || (res.flatMatch() == expectedMatch);
|
|
|
|
++total_tests;
|
|
if (status_ok && match_ok) {
|
|
std::cout << "[PASS] Input: \"" << input << "\" -> "
|
|
<< (res.success ? "SUCCESS" : "FAILURE")
|
|
<< (res.success ? (" (Matched: \"" + unicode::toUTF8(res.flatMatch()) + "\")") : "")
|
|
<< "\n";
|
|
return true;
|
|
}
|
|
|
|
std::cerr << "[FAIL] Input: \"" << input << "\"\n"
|
|
<< " Expected Success: " << (expectedSuccess ? "true" : "false") << ", Got: " << (res.success ? "true" : "false") << "\n";
|
|
if (expectedSuccess && !expectedMatch.empty()) {
|
|
std::cerr << " Expected Match: \"" << unicode::toUTF8(expectedMatch) << "\", Got: \"" << unicode::toUTF8(res.flatMatch()) << "\"\n";
|
|
}
|
|
++failed_tests;
|
|
return false;
|
|
}
|
|
|
|
// ============================================================================
|
|
// Core Combinator & Grammar Tests
|
|
// ============================================================================
|
|
|
|
void test_primitives_and_literals(TokenFactory& tf) {
|
|
std::cout << "\n--- Testing Primitives & Literals ---\n";
|
|
Token* hello = tf.lit("hello");
|
|
|
|
run_test_case(hello, "hello world", true, U"hello");
|
|
run_test_case(hello, "hell", false);
|
|
|
|
run_test_case(asm_ebnf::letter, "a", true);
|
|
run_test_case(asm_ebnf::letter, "f", true);
|
|
run_test_case(asm_ebnf::letter, "A", true);
|
|
run_test_case(asm_ebnf::letter, "Y", true);
|
|
run_test_case(asm_ebnf::letter, "9", false);
|
|
run_test_case(asm_ebnf::digit, "0", true);
|
|
run_test_case(asm_ebnf::digit, "7", true);
|
|
run_test_case(asm_ebnf::digit, "x", false);
|
|
run_test_case(asm_ebnf::hex_digit, "F", true);
|
|
run_test_case(asm_ebnf::hex_digit, "g", false);
|
|
}
|
|
|
|
void test_choice_and_seq(TokenFactory& tf) {
|
|
std::cout << "\n--- Testing Choice & Sequence ---\n";
|
|
Token* seq_test = tf.seq({ tf["foo"], tf["bar"] });
|
|
run_test_case(seq_test, "foobar", true, U"foobar");
|
|
run_test_case(seq_test, "foobaz", false);
|
|
|
|
Token* choice_test = tf.choice({ tf["apple"], tf["banana"] });
|
|
run_test_case(choice_test, "banana", true, U"banana");
|
|
run_test_case(choice_test, "cherry", false);
|
|
}
|
|
|
|
void test_opt_and_rep(TokenFactory& tf) {
|
|
std::cout << "\n--- Testing Optional & Repeat ---\n";
|
|
Token* opt_test = tf.seq({ tf.opt(tf["+"]), asm_ebnf::digit });
|
|
run_test_case(opt_test, "+5", true, U"+5");
|
|
run_test_case(opt_test, "5", true, U"5");
|
|
|
|
Token* rep_digits = tf.rep(asm_ebnf::digit);
|
|
run_test_case(rep_digits, "12345abc", true, U"12345");
|
|
run_test_case(rep_digits, "abc", true, U"");
|
|
}
|
|
|
|
void test_literals(TokenFactory& tf) {
|
|
std::cout << "\n--- Testing Grammatical Literals ---\n";
|
|
run_test_case(asm_ebnf::identifier, "valid_var1", true, U"valid_var1");
|
|
run_test_case(asm_ebnf::identifier, "_private", true, U"_private");
|
|
run_test_case(asm_ebnf::identifier, "123invalid", false);
|
|
|
|
run_test_case(asm_ebnf::decimal_lit, "1234", true);
|
|
run_test_case(asm_ebnf::decimal_lit, "-567L", true);
|
|
run_test_case(asm_ebnf::hex_lit, "0x1A2B", true);
|
|
run_test_case(asm_ebnf::octal_lit, "0c755", true);
|
|
run_test_case(asm_ebnf::binary_lit, "0b10101", true);
|
|
run_test_case(asm_ebnf::float_lit, "3.14159F", true);
|
|
run_test_case(asm_ebnf::float_lit, "1e-10D", true);
|
|
|
|
run_test_case(asm_ebnf::char_lit, "'a'", true);
|
|
run_test_case(asm_ebnf::char_lit, "'\\n'", true);
|
|
run_test_case(asm_ebnf::string_lit, "\"Hello World\"", true);
|
|
run_test_case(asm_ebnf::string_lit, "\"Escape \\\" Test\"", true);
|
|
}
|
|
|
|
void test_addressing_modes(TokenFactory& tf) {
|
|
std::cout << "\n--- Testing Addressing Modes ---\n";
|
|
run_test_case(asm_ebnf::register_tok, "R0", true);
|
|
run_test_case(asm_ebnf::register_tok, "R15", false); // R15 does not exist!
|
|
|
|
run_test_case(asm_ebnf::addrm_ind, "[ 0x1000 ]", true);
|
|
run_test_case(asm_ebnf::addrm_ptr, "[R1]", true);
|
|
run_test_case(asm_ebnf::addrm_idx, "[R1 + 4]", true);
|
|
run_test_case(asm_ebnf::addrm_sca, "[R1 + R2 * 4]", true);
|
|
run_test_case(asm_ebnf::addrm_dis, "[ R1 + R2 * 4 + 16 ]", true);
|
|
}
|
|
|
|
void test_instructions_and_lines(TokenFactory& tf) {
|
|
std::cout << "\n--- Testing Instructions & Lines ---\n";
|
|
run_test_case(asm_ebnf::instruction, "NOP", true);
|
|
run_test_case(asm_ebnf::instruction, "MOV R1, [R2 + 4]", true);
|
|
run_test_case(asm_ebnf::instruction, "ADD R0, 100", true);
|
|
|
|
run_test_case(asm_ebnf::preprocessor, "#define MAX_BUF", true);
|
|
|
|
run_test_case(asm_ebnf::annotation, "@inline", true);
|
|
run_test_case(asm_ebnf::annotation, "@align(4)", true);
|
|
run_test_case(asm_ebnf::annotation, "@section(name=\"text\", flags=1)", true);
|
|
|
|
run_test_case(asm_ebnf::line, "main: MOV R0, R1 ; copy reg\n", true);
|
|
run_test_case(asm_ebnf::line, " @deprecated NOP\n", true);
|
|
run_test_case(asm_ebnf::line, "; only a comment line\n", true);
|
|
}
|
|
|
|
void test_full_program(TokenFactory& tf) {
|
|
std::cout << "\n--- Testing Full Program Parser ---\n";
|
|
std::string asm_code =
|
|
"#include stdio\n"
|
|
"\n"
|
|
"start:\n"
|
|
" MOV R1, 0x20 ; Load constant\n"
|
|
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
|
|
" JMP start\n";
|
|
|
|
ParseTree tree;
|
|
report(compileProgram(asm_code, tree), "full assembly program parses completely");
|
|
}
|
|
|
|
void test_parse_tree(TokenFactory& tf) {
|
|
std::cout << "\n--- Testing Parse Tree Inspection ---\n";
|
|
const std::string asm_code =
|
|
"#include stdio\n"
|
|
"\n"
|
|
"start:\n"
|
|
" MOV R1, 0x20 ; Load constant\n"
|
|
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
|
|
" JMP start\n";
|
|
|
|
ParseTree tree;
|
|
report(compileProgram(asm_code, tree), "program becomes a tree of nodes");
|
|
|
|
ParseNode* root = tree.root();
|
|
report(root != nullptr, "tree exposes a root node");
|
|
if (root == nullptr) return;
|
|
|
|
report(root->hasTag("program"), "root node is tagged as program");
|
|
report(root->textUtf8() == asm_code, "root text covers the whole source");
|
|
report(root->childCount() > 0, "root holds child nodes");
|
|
|
|
vector<ParseNode*> lines = root->findAll("line");
|
|
report(lines.size() == 6, "program holds one node per line");
|
|
|
|
if (lines.size() == 6) {
|
|
report(lines.front()->parentNode() == root, "a line knows its parent");
|
|
report(lines.front()->depth() == 1, "lines sit one level under the root");
|
|
report(lines.front()->nextSibling() == lines[1], "sibling walk reaches the next line");
|
|
report(lines[1]->previousSibling() == lines.front(), "sibling walk goes back");
|
|
report(lines.back()->nextSibling() == nullptr, "the last line has no next sibling");
|
|
|
|
ParseNode* preprocessor = lines.front()->firstChild("preprocessor");
|
|
report(preprocessor != nullptr, "first line holds a preprocessor child");
|
|
report(preprocessor != nullptr && preprocessor->textUtf8() == "#include stdio", "preprocessor text is readable");
|
|
|
|
report(lines[2]->firstChild("line_label") != nullptr, "label line is recognized");
|
|
ParseNode* label = lines[2]->find("label");
|
|
report(label != nullptr && label->textUtf8() == "start:", "label text is readable");
|
|
|
|
ParseNode* mov = lines[3]->find("instruction");
|
|
report(mov != nullptr, "instruction is found inside its line");
|
|
if (mov != nullptr) {
|
|
report(mov->firstChild("opcode") != nullptr, "instruction exposes its opcode child");
|
|
|
|
ParseNode* operands = mov->firstChild("operand_list");
|
|
report(operands != nullptr && operands->childCount() == 2, "operand list exposes both operands");
|
|
|
|
ParseNode* hexlit = mov->find("hex_lit");
|
|
report(hexlit != nullptr && hexlit->textUtf8() == "0x20", "hex literal text is readable");
|
|
}
|
|
|
|
ParseNode* comment = lines[3]->firstChild("comment");
|
|
report(comment != nullptr && comment->textUtf8() == "; Load constant", "trailing comment is captured");
|
|
|
|
report(lines[4]->firstChild("line_annotation") != nullptr, "annotation line is recognized");
|
|
ParseNode* displacement = lines[4]->find("addrm_dis");
|
|
report(displacement != nullptr && displacement->textUtf8() == "[R2 + R3 * 8 + 4]", "displacement operand is readable");
|
|
report(tree.count("register") == 4, "every register of the program is reachable");
|
|
}
|
|
|
|
ParseTree single;
|
|
report(single.parse(asm_ebnf::instruction, "MOV R1, [R2 + 4]"), "a single instruction parses on its own");
|
|
|
|
ParseNode* instruction = single.root();
|
|
report(instruction != nullptr && instruction->hasTag("instruction"), "instruction root is tagged");
|
|
if (instruction != nullptr) {
|
|
ParseNode* operands = instruction->firstChild("operand_list");
|
|
report(operands != nullptr && operands->childCount() == 2, "operand list holds two operands");
|
|
|
|
ParseNode* second = operands == nullptr ? nullptr : operands->childAt(1);
|
|
report(second != nullptr && second->textUtf8() == "[R2 + 4]", "operand text is readable");
|
|
report(second != nullptr && second->ancestor("instruction") == instruction, "an operand can climb back to its instruction");
|
|
}
|
|
|
|
std::cout << "\n--- Parsed Tree ---\n" << tree.toString() << "\n";
|
|
}
|
|
|
|
// ============================================================================
|
|
// Main Execution
|
|
// ============================================================================
|
|
|
|
int main() {
|
|
TokenFactory tf;
|
|
assemblyGrammar();
|
|
|
|
std::cout << "Running Token Framework Tests...\n";
|
|
|
|
test_primitives_and_literals(tf);
|
|
test_choice_and_seq(tf);
|
|
test_opt_and_rep(tf);
|
|
test_literals(tf);
|
|
test_addressing_modes(tf);
|
|
test_instructions_and_lines(tf);
|
|
test_full_program(tf);
|
|
test_parse_tree(tf);
|
|
|
|
std::cout << "\n========================================\n";
|
|
std::cout << "Test Results: " << (total_tests - failed_tests) << "/" << total_tests << " passed.\n";
|
|
std::cout << "========================================\n";
|
|
|
|
if (failed_tests > 0) {
|
|
std::cerr << failed_tests << " test(s) FAILED.\n";
|
|
return 1;
|
|
}
|
|
|
|
std::cout << "All token framework tests passed successfully!\n";
|
|
return 0;
|
|
}
|
|
|