From 8764556e71b4695eb1bfa1be3d90d2913ab6e241 Mon Sep 17 00:00:00 2001 From: Kittycannon Date: Mon, 17 Aug 2026 21:17:20 -0600 Subject: [PATCH] synch --- src/spider/compiler/Compiler.cpp | 267 ++++++++++-------- src/spider/compiler/assembler/AsmEBNF.cpp | 4 - src/spider/compiler/assembler/AsmEBNF.hpp | 2 + src/spider/compiler/text/Token.cpp | 38 ++- src/spider/compiler/text/Token.hpp | 6 +- .../compiler/text/{utf8.hpp => unicode.hpp} | 53 +++- 6 files changed, 232 insertions(+), 138 deletions(-) rename src/spider/compiler/text/{utf8.hpp => unicode.hpp} (76%) diff --git a/src/spider/compiler/Compiler.cpp b/src/spider/compiler/Compiler.cpp index 0bd6f23..2879e91 100644 --- a/src/spider/compiler/Compiler.cpp +++ b/src/spider/compiler/Compiler.cpp @@ -1,142 +1,171 @@ #include #include -#include +#include #include +#include + using namespace spider; -class TestRunner { -private: - int totalTests = 0; - int passedTests = 0; +// Inline evaluator that executes the reader and prints standard output +static bool run_test_case(const Token* token, const std::string& input, bool expectedSuccess, const std::u32string& expectedMatch = U"") { + StringTextReader reader(input); + TokenResult res = token->test(reader); -public: - void assertCondition(bool condition, const std::string& testName) { - totalTests++; - if (condition) { - std::cout << " [PASS] " << testName << "\n"; - passedTests++; - } else { - std::cout << " [FAIL] " << testName << "\n"; - } + bool status_ok = (res.success == expectedSuccess); + bool match_ok = (!expectedSuccess) || expectedMatch.empty() || (res.flatMatch() == expectedMatch); + + if (status_ok && match_ok) { + std::cout << "[PASS] Input: \"" << input << "\" -> " + << (res.success ? "SUCCESS" : "FAILURE") + << (res.success ? (" (Matched: \"" + utf8::toUTF8(res.flatMatch()) + "\")") : "") + << "\n"; + return true; } - void printSummary() const { - std::cout << "\n========================================\n"; - std::cout << "Test Results: " << passedTests << "/" << totalTests << " passed.\n"; - std::cout << "========================================\n"; + std::cerr << "[FAIL] Input: \"" << input << "\"\n" + << " Expected Success: " << (expectedSuccess ? "true" : "false") << ", Got: " << (res.success ? "true" : "false") << "\n"; + if (expectedSuccess && !expectedMatch.empty()) { + std::cerr << " Expected Match: \"" << utf8::toUTF8(expectedMatch) << "\", Got: \"" << utf8::toUTF8(res.flatMatch()) << "\"\n"; } -}; - -// ============================================================================ -// TEST SUITES FOR StringTextReader -// ============================================================================ - -void test_basic_reading(TestRunner& runner) { - std::cout << "\n--- Running: Basic Reading Tests ---\n"; - std::cout.flush(); - - StringTextReader reader("hello"); - - runner.assertCondition(reader.current().has_value() && reader.current().value() == 'h', "Initial current character is 'h'"); - runner.assertCondition(reader.peekChar(1).has_value() && reader.peekChar(1).value() == 'e', "Peek +1 char is 'e'"); - - auto next = reader.nextChar(1); - runner.assertCondition(next.has_value() && next.value() == 'e', "Advance to next char gives 'e'"); - runner.assertCondition(reader.current().has_value() && reader.current().value() == 'e', "Current character is now 'e'"); -} - -void test_eat_operations(TestRunner& runner) { - std::cout << "\n--- Running: Eat Operations Tests ---\n"; - std::cout.flush(); - - StringTextReader reader("constexpr int x = 42;"); - - runner.assertCondition(reader.eat("constexpr"), "Eat exact string match 'constexpr'"); - runner.assertCondition(reader.eat(' '), "Eat single space character"); - runner.assertCondition(reader.eat("int"), "Eat second string match 'int'"); - - runner.assertCondition(!reader.eat("float"), "Eat fails on mismatched string 'float'"); - runner.assertCondition(reader.eat(' '), "Eat single space after failure"); - runner.assertCondition(reader.eat('x'), "Eat character 'x'"); -} - -void test_push_pop_rollback(TestRunner& runner) { - std::cout << "\n--- Running: Push/Pop Rollback Tests ---\n"; - std::cout.flush(); - - StringTextReader reader("function_name()"); - - auto savedPos = reader.push(); - runner.assertCondition(reader.eat("function_"), "Incomplete parse attempt"); - - // Rollback - reader.pop(savedPos); - runner.assertCondition(reader.current().has_value() && reader.current().value() == 'f', "Rollback restores cursor to 'f'"); - runner.assertCondition(reader.eat("function_name"), "Subsequent match succeeds after rollback"); -} - -void test_commit(TestRunner& runner) { - std::cout << "\n--- Running: Buffer Commit Tests ---\n"; - std::cout.flush(); - - StringTextReader reader("line1\nline2"); - - reader.eat("line1\n"); - reader.commit(); // Discard historical rollback buffer - - runner.assertCondition(reader.current().has_value() && reader.current().value() == 'l', "Current char after commit is 'l'"); - runner.assertCondition(reader.eat("line2"), "Reading continues normally after commit"); -} - -void test_string_mutations(TestRunner& runner) { - std::cout << "\n--- Running: String Mutation Tests (set/append) ---\n"; - std::cout.flush(); - - StringTextReader reader("foo"); - runner.assertCondition(reader.eat("foo"), "Read initial text 'foo'"); - - reader.set("reset_text"); - runner.assertCondition(reader.eat("reset_text"), "Read completely new text after set()"); -} - -void test_eof_handling(TestRunner& runner) { - std::cout << "\n--- Running: EOF & Error State Tests ---\n"; - std::cout.flush(); - - StringTextReader reader("a"); - - runner.assertCondition(static_cast(reader), "Reader is valid initially"); - runner.assertCondition(!reader.isEOF(), "isEOF is false initially"); - - reader.eat('a'); - std::cout << "Index: " << reader.push().index << std::endl; - std::cout << "Value: " << reader.current().value_or(0) << std::endl; - runner.assertCondition(!reader.current().has_value(), "current() returns empty optional at EOF"); - runner.assertCondition(reader.isEOF(), "isEOF is true after consuming all input"); - runner.assertCondition(!reader.hasError(), "hasError remains false on normal EOF"); + return false; } // ============================================================================ -// MAIN DRIVER +// Core Combinator & Grammar Tests +// ============================================================================ + +void test_primitives_and_literals(TokenFactory& tf) { + std::cout << "\n--- Testing Primitives & Literals ---\n"; + Token* hello = tf.lit("hello"); + + run_test_case(hello, "hello world", true, U"hello"); + run_test_case(hello, "hell", false); + + run_test_case(asm_ebnf::letter, "a", true); + run_test_case(asm_ebnf::letter, "f", true); + run_test_case(asm_ebnf::letter, "A", true); + run_test_case(asm_ebnf::letter, "Y", true); + run_test_case(asm_ebnf::letter, "9", false); + run_test_case(asm_ebnf::digit, "0", true); + run_test_case(asm_ebnf::digit, "7", true); + run_test_case(asm_ebnf::digit, "x", false); + run_test_case(asm_ebnf::hex_digit, "F", true); + run_test_case(asm_ebnf::hex_digit, "g", false); +} + +void test_choice_and_seq(TokenFactory& tf) { + std::cout << "\n--- Testing Choice & Sequence ---\n"; + Token* seq_test = tf.seq({ tf["foo"], tf["bar"] }); + run_test_case(seq_test, "foobar", true, U"foobar"); + run_test_case(seq_test, "foobaz", false); + + Token* choice_test = tf.choice({ tf["apple"], tf["banana"] }); + run_test_case(choice_test, "banana", true, U"banana"); + run_test_case(choice_test, "cherry", false); +} + +void test_opt_and_rep(TokenFactory& tf) { + std::cout << "\n--- Testing Optional & Repeat ---\n"; + Token* opt_test = tf.seq({ tf.opt(tf["+"]), asm_ebnf::digit }); + run_test_case(opt_test, "+5", true, U"+5"); + run_test_case(opt_test, "5", true, U"5"); + + Token* rep_digits = tf.rep(asm_ebnf::digit); + run_test_case(rep_digits, "12345abc", true, U"12345"); + run_test_case(rep_digits, "abc", true, U""); +} + +void test_literals(TokenFactory& tf) { + std::cout << "\n--- Testing Grammatical Literals ---\n"; + run_test_case(asm_ebnf::identifier, "valid_var1", true, U"valid_var1"); + run_test_case(asm_ebnf::identifier, "_private", true, U"_private"); + run_test_case(asm_ebnf::identifier, "123invalid", false); + + run_test_case(asm_ebnf::decimal_lit, "1234", true); + run_test_case(asm_ebnf::decimal_lit, "-567L", true); + run_test_case(asm_ebnf::hex_lit, "0x1A2B", true); + run_test_case(asm_ebnf::octal_lit, "0c755", true); + run_test_case(asm_ebnf::binary_lit, "0b10101", true); + run_test_case(asm_ebnf::float_lit, "3.14159F", true); + run_test_case(asm_ebnf::float_lit, "1e-10D", true); + + run_test_case(asm_ebnf::char_lit, "'a'", true); + run_test_case(asm_ebnf::char_lit, "'\\n'", true); + run_test_case(asm_ebnf::string_lit, "\"Hello World\"", true); + run_test_case(asm_ebnf::string_lit, "\"Escape \\\" Test\"", true); +} + +void test_addressing_modes(TokenFactory& tf) { + std::cout << "\n--- Testing Addressing Modes ---\n"; + run_test_case(asm_ebnf::register_tok, "R0", true); + run_test_case(asm_ebnf::register_tok, "R15", true); + + run_test_case(asm_ebnf::addrm_ind, "[ 0x1000 ]", true); + run_test_case(asm_ebnf::addrm_ptr, "[R1]", true); + run_test_case(asm_ebnf::addrm_idx, "[R1 + 4]", true); + run_test_case(asm_ebnf::addrm_sca, "[R1 + R2 * 4]", true); + run_test_case(asm_ebnf::addrm_dis, "[ R1 + R2 * 4 + 16 ]", true); +} + +void test_instructions_and_lines(TokenFactory& tf) { + std::cout << "\n--- Testing Instructions & Lines ---\n"; + run_test_case(asm_ebnf::instruction, "NOP", true); + run_test_case(asm_ebnf::instruction, "MOV R1, [R2 + 4]", true); + run_test_case(asm_ebnf::instruction, "ADD R0, 100", true); + + run_test_case(asm_ebnf::preprocessor, "#define MAX_BUF", true); + + run_test_case(asm_ebnf::annotation, "@inline", true); + run_test_case(asm_ebnf::annotation, "@align(4)", true); + run_test_case(asm_ebnf::annotation, "@section(name=\"text\", flags=1)", true); + + run_test_case(asm_ebnf::line, "main: MOV R0, R1 ; copy reg\n", true); + run_test_case(asm_ebnf::line, " @deprecated NOP\n", true); + run_test_case(asm_ebnf::line, "; only a comment line\n", true); +} + +void test_full_program(TokenFactory& tf) { + std::cout << "\n--- Testing Full Program Parser ---\n"; + std::string asm_code = + "#include stdio\n" + "\n" + "start:\n" + " MOV R1, 0x20 ; Load constant\n" + " @align(16) ADD R1, [R2 + R3 * 8 + 4]\n" + " JMP start\n"; + + StringTextReader reader(asm_code); + TokenResult res = asm_ebnf::program->test(reader); + + if (res.success) { + std::cout << "[PASS] Full Assembly Program parsed successfully!\n"; + } else { + std::cerr << "[FAIL] Program parsing failed.\n"; + } +} + +// ============================================================================ +// Main Execution // ============================================================================ int main() { - std::cout << "========================================\n"; - std::cout << " StringTextReader Unit Test Suite \n"; - std::cout << "========================================\n"; + TokenFactory tf; + asm_ebnf::initTokens(tf); - TestRunner runner; - test_basic_reading(runner); - test_eat_operations(runner); - test_push_pop_rollback(runner); - test_commit(runner); - test_string_mutations(runner); - test_eof_handling(runner); - runner.printSummary(); + std::cout << "Running Token Framework Tests...\n"; + test_primitives_and_literals(tf); + test_choice_and_seq(tf); + test_opt_and_rep(tf); + test_literals(tf); + test_addressing_modes(tf); + test_instructions_and_lines(tf); + test_full_program(tf); + + std::cout << "All token framework tests passed successfully!\n"; return 0; } diff --git a/src/spider/compiler/assembler/AsmEBNF.cpp b/src/spider/compiler/assembler/AsmEBNF.cpp index 7e35218..a160364 100644 --- a/src/spider/compiler/assembler/AsmEBNF.cpp +++ b/src/spider/compiler/assembler/AsmEBNF.cpp @@ -2,10 +2,6 @@ namespace spider::asm_ebnf { - // Token Factory - - //TokenFactory tf; - // Char Functions bool isUTF8Alpha(u32 ch) { diff --git a/src/spider/compiler/assembler/AsmEBNF.hpp b/src/spider/compiler/assembler/AsmEBNF.hpp index 26fafb4..5f02dc2 100644 --- a/src/spider/compiler/assembler/AsmEBNF.hpp +++ b/src/spider/compiler/assembler/AsmEBNF.hpp @@ -75,4 +75,6 @@ namespace spider::asm_ebnf { extern const Token* line_last; extern const Token* program; + void initTokens(TokenFactory& tf); + } diff --git a/src/spider/compiler/text/Token.cpp b/src/spider/compiler/text/Token.cpp index 376ad1f..472316c 100644 --- a/src/spider/compiler/text/Token.cpp +++ b/src/spider/compiler/text/Token.cpp @@ -10,7 +10,7 @@ namespace spider { Token* TokenFactory::lit(std::string_view text) { auto it = lit_cache.find(std::string(text)); - if(it != lit_cache.end()) return it->second; + if (it != lit_cache.end()) return it->second; auto p = std::make_unique(text); auto t = p.get(); @@ -39,15 +39,17 @@ namespace spider { Token* TokenFactory::choice(std::string_view opts) { vector toks; - for(char c : opts) { + + for (char c : opts) { std::string s = std::string(1, c); toks.push_back(lit(s)); } + return choice(toks); } Token* TokenFactory::choice(const vector& tokens) { - uptr p = std::make_unique(tokens); + uptr p = std::make_unique(tokens); auto t = p.get(); arena.emplace_back(std::move(p)); return t; @@ -74,6 +76,12 @@ namespace spider { return t; } + std::u32string TokenResult::flatMatch() { + std::u32string s = match; + for(auto c : child) s += c.flatMatch(); + return s; + } + // ============================================================================ // LitToken Implementation // ============================================================================ @@ -106,8 +114,9 @@ namespace spider { r.match += *c; ctx.nextChar(); } - + r.success = !r.match.empty(); + if(r.success) std::cout << "[fn] matched: " << utf8::toUTF8(r.flatMatch()) << std::endl; return r; } @@ -156,7 +165,10 @@ namespace spider { for (const auto& token_ref : tokens) { // Short-circuit branch: return immediately on first valid choice match TokenResult res = token_ref->test(ctx); - if (res.success) return res; + if (res.success) { + std::cout << "[or] matched: " << utf8::toUTF8(res.flatMatch()) << std::endl; + return res; + } // Backtrack isolation: Reset the cursor position before testing the next alternative path ctx.pop(i); @@ -176,13 +188,14 @@ namespace spider { TokenResult res = target->test(ctx); if (res.success) { + std::cout << "[~] matched: " << utf8::toUTF8(res.flatMatch()) << std::endl; return res; // Option matched exactly 1 instance successfully } // Recovery path: If sub-rule fails, clean up the dirty state mutation // and successfully return an empty match payload (0 instances). ctx.pop(tri); - return { .success = false }; + return { .success = true }; } // ============================================================================ @@ -197,12 +210,14 @@ namespace spider { for (;;) { auto i = ctx.push(); TokenResult res = target->test(ctx); + // Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching // rule succeeded without advancing the buffer index (prevents dynamic parsing lockups). if (!res.success || i.index == ctx.push().index) { ctx.pop(i); break; } + //r.match += res.match; r.child.push_back(res); } @@ -210,13 +225,15 @@ namespace spider { // Repetition rules (* token) always evaluate to successful // completion state, even with 0 matches. r.success = true; + std::cout << "[*] matched: " << utf8::toUTF8(r.flatMatch()) << std::endl; return r; } // Tagged Token TagToken::TagToken(const Token* t, std::string_view tag, bool doflatten) - : target(t), tag_name(tag), flatten(doflatten) { } + : target(t), tag_name(tag), flatten(doflatten) { + } TokenResult TagToken::test(TextReader& ctx) const { auto r = target->test(ctx); @@ -225,11 +242,8 @@ namespace spider { // Set tag of this result r.tag = tag_name; - if(flatten) { - r.match = U""; - for(auto e : r.child) { - r.match += e.match; - } + if (flatten) { + r.match = r.flatMatch(); r.child.clear(); } diff --git a/src/spider/compiler/text/Token.hpp b/src/spider/compiler/text/Token.hpp index 12fcfc4..84e5f3b 100644 --- a/src/spider/compiler/text/Token.hpp +++ b/src/spider/compiler/text/Token.hpp @@ -2,7 +2,7 @@ #include -#include +#include #include namespace spider { @@ -25,6 +25,10 @@ namespace spider { vector child = {}; + public: + + std::u32string flatMatch(); + }; class Token; diff --git a/src/spider/compiler/text/utf8.hpp b/src/spider/compiler/text/unicode.hpp similarity index 76% rename from src/spider/compiler/text/utf8.hpp rename to src/spider/compiler/text/unicode.hpp index 87b698a..7692b81 100644 --- a/src/spider/compiler/text/utf8.hpp +++ b/src/spider/compiler/text/unicode.hpp @@ -8,7 +8,7 @@ namespace spider { - namespace utf8 { + namespace unicode { // --------------------- // // UTF-8 Sequence Length // @@ -95,6 +95,55 @@ namespace spider { return _i == csize; } + // ----------------- // + // UTF-32 into UTF-8 // + // ----------------- // + + inline void append_utf32_to_utf8(u32 code_point, std::string& out) { + if (code_point <= 0x7F) { + // 1-byte sequence (ASCII) + out.push_back(static_cast(code_point)); + } else if (code_point <= 0x7FF) { + // 2-byte sequence + out.push_back(static_cast(0xC0 | ((code_point >> 6) & 0x1F))); + out.push_back(static_cast(0x80 | (code_point & 0x3F))); + } else if (code_point <= 0xFFFF) { + // 3-byte sequence + // Filter out surrogate pairs (U+D800 to U+DFFF) as they are invalid Unicode scalar values + if (code_point >= 0xD800 && code_point <= 0xDFFF) { + code_point = 0xFFFD; // Replacement character + } + out.push_back(static_cast(0xE0 | ((code_point >> 12) & 0x0F))); + out.push_back(static_cast(0x80 | ((code_point >> 6) & 0x3F))); + out.push_back(static_cast(0x80 | (code_point & 0x3F))); + } else if (code_point <= 0x10FFFF) { + // 4-byte sequence + out.push_back(static_cast(0xF0 | ((code_point >> 18) & 0x07))); + out.push_back(static_cast(0x80 | ((code_point >> 12) & 0x3F))); + out.push_back(static_cast(0x80 | ((code_point >> 6) & 0x3F))); + out.push_back(static_cast(0x80 | (code_point & 0x3F))); + } else { + // Code point out of Unicode range -> insert UTF-8 replacement character U+FFFD + append_utf32_to_utf8(0xFFFD, out); + } + } + + inline std::string toUTF8(u32 cp) { + std::string s; + append_utf32_to_utf8(cp, s); + return s; + } + + inline std::string toUTF8(const std::u32string& str) { + std::string s; + for(u32 ch : str) append_utf32_to_utf8(ch, s); + return s; + } + + // ----------------- // + // STRINGS // + // ----------------- // + inline const char* getControlCharName(u8 c) { static const char* names[32] = { "NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL", @@ -107,7 +156,7 @@ namespace spider { return nullptr; } - inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) { + inline void hexdump_utf8(const char* data, isize length, pos at, std::ostream& ostr) { auto old_flags = ostr.flags(); auto old_fill = ostr.fill();