Files
spider-compiler/src/spider/compiler/Compiler.cpp
T
2026-09-27 01:05:03 -06:00

297 lines
12 KiB
C++

#include <iostream>
#include <spider/compiler/Compiler.hpp>
#include <spider/compiler/common.hpp>
#include <spider/compiler/text/unicode.hpp>
#include <spider/compiler/text/TextReader.hpp>
#include <spider/compiler/assembler/AsmEBNF.hpp>
#include <spider/compiler/text/ParseTree.hpp>
using namespace spider;
// ============================================================================
// Compiler Entry Points
// ============================================================================
namespace spider {
TokenFactory& assemblyGrammar() {
static TokenFactory grammar;
static bool loaded = false;
if (!loaded) {
asm_ebnf::initTokens(grammar);
loaded = true;
}
return grammar;
}
bool compileProgram(const std::string& source, ParseTree& out) {
return out.parse(asm_ebnf::program, source);
}
}
// ============================================================================
// Test Bookkeeping
// ============================================================================
// Suite wide bookkeeping, so a failing case can never hide behind a clean exit code.
static isize total_tests = 0;
static isize failed_tests = 0;
static void report(bool ok, const std::string& name) {
++total_tests;
if (ok) {
std::cout << "[PASS] " << name << "\n";
} else {
++failed_tests;
std::cerr << "[FAIL] " << name << "\n";
}
}
// Inline evaluator that executes the reader and prints standard output
static bool run_test_case(const Token* token, const std::string& input, bool expectedSuccess, const std::u32string& expectedMatch = U"") {
StringTextReader reader(input);
TokenResult res = token->test(reader);
bool status_ok = (res.success == expectedSuccess);
bool match_ok = (!expectedSuccess) || expectedMatch.empty() || (res.flatMatch() == expectedMatch);
++total_tests;
if (status_ok && match_ok) {
std::cout << "[PASS] Input: \"" << input << "\" -> "
<< (res.success ? "SUCCESS" : "FAILURE")
<< (res.success ? (" (Matched: \"" + unicode::toUTF8(res.flatMatch()) + "\")") : "")
<< "\n";
return true;
}
std::cerr << "[FAIL] Input: \"" << input << "\"\n"
<< " Expected Success: " << (expectedSuccess ? "true" : "false") << ", Got: " << (res.success ? "true" : "false") << "\n";
if (expectedSuccess && !expectedMatch.empty()) {
std::cerr << " Expected Match: \"" << unicode::toUTF8(expectedMatch) << "\", Got: \"" << unicode::toUTF8(res.flatMatch()) << "\"\n";
}
++failed_tests;
return false;
}
// ============================================================================
// Core Combinator & Grammar Tests
// ============================================================================
void test_primitives_and_literals(TokenFactory& tf) {
std::cout << "\n--- Testing Primitives & Literals ---\n";
Token* hello = tf.lit("hello");
run_test_case(hello, "hello world", true, U"hello");
run_test_case(hello, "hell", false);
run_test_case(asm_ebnf::letter, "a", true);
run_test_case(asm_ebnf::letter, "f", true);
run_test_case(asm_ebnf::letter, "A", true);
run_test_case(asm_ebnf::letter, "Y", true);
run_test_case(asm_ebnf::letter, "9", false);
run_test_case(asm_ebnf::digit, "0", true);
run_test_case(asm_ebnf::digit, "7", true);
run_test_case(asm_ebnf::digit, "x", false);
run_test_case(asm_ebnf::hex_digit, "F", true);
run_test_case(asm_ebnf::hex_digit, "g", false);
}
void test_choice_and_seq(TokenFactory& tf) {
std::cout << "\n--- Testing Choice & Sequence ---\n";
Token* seq_test = tf.seq({ tf["foo"], tf["bar"] });
run_test_case(seq_test, "foobar", true, U"foobar");
run_test_case(seq_test, "foobaz", false);
Token* choice_test = tf.choice({ tf["apple"], tf["banana"] });
run_test_case(choice_test, "banana", true, U"banana");
run_test_case(choice_test, "cherry", false);
}
void test_opt_and_rep(TokenFactory& tf) {
std::cout << "\n--- Testing Optional & Repeat ---\n";
Token* opt_test = tf.seq({ tf.opt(tf["+"]), asm_ebnf::digit });
run_test_case(opt_test, "+5", true, U"+5");
run_test_case(opt_test, "5", true, U"5");
Token* rep_digits = tf.rep(asm_ebnf::digit);
run_test_case(rep_digits, "12345abc", true, U"12345");
run_test_case(rep_digits, "abc", true, U"");
}
void test_literals(TokenFactory& tf) {
std::cout << "\n--- Testing Grammatical Literals ---\n";
run_test_case(asm_ebnf::identifier, "valid_var1", true, U"valid_var1");
run_test_case(asm_ebnf::identifier, "_private", true, U"_private");
run_test_case(asm_ebnf::identifier, "123invalid", false);
run_test_case(asm_ebnf::decimal_lit, "1234", true);
run_test_case(asm_ebnf::decimal_lit, "-567L", true);
run_test_case(asm_ebnf::hex_lit, "0x1A2B", true);
run_test_case(asm_ebnf::octal_lit, "0c755", true);
run_test_case(asm_ebnf::binary_lit, "0b10101", true);
run_test_case(asm_ebnf::float_lit, "3.14159F", true);
run_test_case(asm_ebnf::float_lit, "1e-10D", true);
run_test_case(asm_ebnf::char_lit, "'a'", true);
run_test_case(asm_ebnf::char_lit, "'\\n'", true);
run_test_case(asm_ebnf::string_lit, "\"Hello World\"", true);
run_test_case(asm_ebnf::string_lit, "\"Escape \\\" Test\"", true);
}
void test_addressing_modes(TokenFactory& tf) {
std::cout << "\n--- Testing Addressing Modes ---\n";
run_test_case(asm_ebnf::register_tok, "R0", true);
run_test_case(asm_ebnf::register_tok, "R15", false); // R15 does not exist!
run_test_case(asm_ebnf::addrm_ind, "[ 0x1000 ]", true);
run_test_case(asm_ebnf::addrm_ptr, "[R1]", true);
run_test_case(asm_ebnf::addrm_idx, "[R1 + 4]", true);
run_test_case(asm_ebnf::addrm_sca, "[R1 + R2 * 4]", true);
run_test_case(asm_ebnf::addrm_dis, "[ R1 + R2 * 4 + 16 ]", true);
}
void test_instructions_and_lines(TokenFactory& tf) {
std::cout << "\n--- Testing Instructions & Lines ---\n";
run_test_case(asm_ebnf::instruction, "NOP", true);
run_test_case(asm_ebnf::instruction, "MOV R1, [R2 + 4]", true);
run_test_case(asm_ebnf::instruction, "ADD R0, 100", true);
run_test_case(asm_ebnf::preprocessor, "#define MAX_BUF", true);
run_test_case(asm_ebnf::annotation, "@inline", true);
run_test_case(asm_ebnf::annotation, "@align(4)", true);
run_test_case(asm_ebnf::annotation, "@section(name=\"text\", flags=1)", true);
run_test_case(asm_ebnf::line, "main: MOV R0, R1 ; copy reg\n", true);
run_test_case(asm_ebnf::line, " @deprecated NOP\n", true);
run_test_case(asm_ebnf::line, "; only a comment line\n", true);
}
void test_full_program(TokenFactory& tf) {
std::cout << "\n--- Testing Full Program Parser ---\n";
std::string asm_code =
"#include stdio\n"
"\n"
"start:\n"
" MOV R1, 0x20 ; Load constant\n"
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
" JMP start\n";
ParseTree tree;
report(compileProgram(asm_code, tree), "full assembly program parses completely");
}
void test_parse_tree(TokenFactory& tf) {
std::cout << "\n--- Testing Parse Tree Inspection ---\n";
const std::string asm_code =
"#include stdio\n"
"\n"
"start:\n"
" MOV R1, 0x20 ; Load constant\n"
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
" JMP start\n";
ParseTree tree;
report(compileProgram(asm_code, tree), "program becomes a tree of nodes");
ParseNode* root = tree.root();
report(root != nullptr, "tree exposes a root node");
if (root == nullptr) return;
report(root->hasTag("program"), "root node is tagged as program");
report(root->textUtf8() == asm_code, "root text covers the whole source");
report(root->childCount() > 0, "root holds child nodes");
vector<ParseNode*> lines = root->findAll("line");
report(lines.size() == 6, "program holds one node per line");
if (lines.size() == 6) {
report(lines.front()->parentNode() == root, "a line knows its parent");
report(lines.front()->depth() == 1, "lines sit one level under the root");
report(lines.front()->nextSibling() == lines[1], "sibling walk reaches the next line");
report(lines[1]->previousSibling() == lines.front(), "sibling walk goes back");
report(lines.back()->nextSibling() == nullptr, "the last line has no next sibling");
ParseNode* preprocessor = lines.front()->firstChild("preprocessor");
report(preprocessor != nullptr, "first line holds a preprocessor child");
report(preprocessor != nullptr && preprocessor->textUtf8() == "#include stdio", "preprocessor text is readable");
report(lines[2]->firstChild("line_label") != nullptr, "label line is recognized");
ParseNode* label = lines[2]->find("label");
report(label != nullptr && label->textUtf8() == "start:", "label text is readable");
ParseNode* mov = lines[3]->find("instruction");
report(mov != nullptr, "instruction is found inside its line");
if (mov != nullptr) {
report(mov->firstChild("opcode") != nullptr, "instruction exposes its opcode child");
ParseNode* operands = mov->firstChild("operand_list");
report(operands != nullptr && operands->childCount() == 2, "operand list exposes both operands");
ParseNode* hexlit = mov->find("hex_lit");
report(hexlit != nullptr && hexlit->textUtf8() == "0x20", "hex literal text is readable");
}
ParseNode* comment = lines[3]->firstChild("comment");
report(comment != nullptr && comment->textUtf8() == "; Load constant", "trailing comment is captured");
report(lines[4]->firstChild("line_annotation") != nullptr, "annotation line is recognized");
ParseNode* displacement = lines[4]->find("addrm_dis");
report(displacement != nullptr && displacement->textUtf8() == "[R2 + R3 * 8 + 4]", "displacement operand is readable");
report(tree.count("register") == 4, "every register of the program is reachable");
}
ParseTree single;
report(single.parse(asm_ebnf::instruction, "MOV R1, [R2 + 4]"), "a single instruction parses on its own");
ParseNode* instruction = single.root();
report(instruction != nullptr && instruction->hasTag("instruction"), "instruction root is tagged");
if (instruction != nullptr) {
ParseNode* operands = instruction->firstChild("operand_list");
report(operands != nullptr && operands->childCount() == 2, "operand list holds two operands");
ParseNode* second = operands == nullptr ? nullptr : operands->childAt(1);
report(second != nullptr && second->textUtf8() == "[R2 + 4]", "operand text is readable");
report(second != nullptr && second->ancestor("instruction") == instruction, "an operand can climb back to its instruction");
}
std::cout << "\n--- Parsed Tree ---\n" << tree.toString() << "\n";
}
// ============================================================================
// Main Execution
// ============================================================================
int main() {
TokenFactory tf;
assemblyGrammar();
std::cout << "Running Token Framework Tests...\n";
test_primitives_and_literals(tf);
test_choice_and_seq(tf);
test_opt_and_rep(tf);
test_literals(tf);
test_addressing_modes(tf);
test_instructions_and_lines(tf);
test_full_program(tf);
test_parse_tree(tf);
std::cout << "\n========================================\n";
std::cout << "Test Results: " << (total_tests - failed_tests) << "/" << total_tests << " passed.\n";
std::cout << "========================================\n";
if (failed_tests > 0) {
std::cerr << failed_tests << " test(s) FAILED.\n";
return 1;
}
std::cout << "All token framework tests passed successfully!\n";
return 0;
}