This commit is contained in:
2026-08-17 21:17:20 -06:00
parent 897e155b0e
commit 8764556e71
6 changed files with 232 additions and 138 deletions
+149 -120
View File
@@ -1,142 +1,171 @@
#include <iostream> #include <iostream>
#include <spider/compiler/common.hpp> #include <spider/compiler/common.hpp>
#include <spider/compiler/text/utf8.hpp>
#include <spider/compiler/text/unicode.hpp>
#include <spider/compiler/text/TextReader.hpp> #include <spider/compiler/text/TextReader.hpp>
#include <spider/compiler/assembler/AsmEBNF.hpp>
using namespace spider; using namespace spider;
class TestRunner { // Inline evaluator that executes the reader and prints standard output
private: static bool run_test_case(const Token* token, const std::string& input, bool expectedSuccess, const std::u32string& expectedMatch = U"") {
int totalTests = 0; StringTextReader reader(input);
int passedTests = 0; TokenResult res = token->test(reader);
public: bool status_ok = (res.success == expectedSuccess);
void assertCondition(bool condition, const std::string& testName) { bool match_ok = (!expectedSuccess) || expectedMatch.empty() || (res.flatMatch() == expectedMatch);
totalTests++;
if (condition) { if (status_ok && match_ok) {
std::cout << " [PASS] " << testName << "\n"; std::cout << "[PASS] Input: \"" << input << "\" -> "
passedTests++; << (res.success ? "SUCCESS" : "FAILURE")
<< (res.success ? (" (Matched: \"" + utf8::toUTF8(res.flatMatch()) + "\")") : "")
<< "\n";
return true;
}
std::cerr << "[FAIL] Input: \"" << input << "\"\n"
<< " Expected Success: " << (expectedSuccess ? "true" : "false") << ", Got: " << (res.success ? "true" : "false") << "\n";
if (expectedSuccess && !expectedMatch.empty()) {
std::cerr << " Expected Match: \"" << utf8::toUTF8(expectedMatch) << "\", Got: \"" << utf8::toUTF8(res.flatMatch()) << "\"\n";
}
return false;
}
// ============================================================================
// Core Combinator & Grammar Tests
// ============================================================================
void test_primitives_and_literals(TokenFactory& tf) {
std::cout << "\n--- Testing Primitives & Literals ---\n";
Token* hello = tf.lit("hello");
run_test_case(hello, "hello world", true, U"hello");
run_test_case(hello, "hell", false);
run_test_case(asm_ebnf::letter, "a", true);
run_test_case(asm_ebnf::letter, "f", true);
run_test_case(asm_ebnf::letter, "A", true);
run_test_case(asm_ebnf::letter, "Y", true);
run_test_case(asm_ebnf::letter, "9", false);
run_test_case(asm_ebnf::digit, "0", true);
run_test_case(asm_ebnf::digit, "7", true);
run_test_case(asm_ebnf::digit, "x", false);
run_test_case(asm_ebnf::hex_digit, "F", true);
run_test_case(asm_ebnf::hex_digit, "g", false);
}
void test_choice_and_seq(TokenFactory& tf) {
std::cout << "\n--- Testing Choice & Sequence ---\n";
Token* seq_test = tf.seq({ tf["foo"], tf["bar"] });
run_test_case(seq_test, "foobar", true, U"foobar");
run_test_case(seq_test, "foobaz", false);
Token* choice_test = tf.choice({ tf["apple"], tf["banana"] });
run_test_case(choice_test, "banana", true, U"banana");
run_test_case(choice_test, "cherry", false);
}
void test_opt_and_rep(TokenFactory& tf) {
std::cout << "\n--- Testing Optional & Repeat ---\n";
Token* opt_test = tf.seq({ tf.opt(tf["+"]), asm_ebnf::digit });
run_test_case(opt_test, "+5", true, U"+5");
run_test_case(opt_test, "5", true, U"5");
Token* rep_digits = tf.rep(asm_ebnf::digit);
run_test_case(rep_digits, "12345abc", true, U"12345");
run_test_case(rep_digits, "abc", true, U"");
}
void test_literals(TokenFactory& tf) {
std::cout << "\n--- Testing Grammatical Literals ---\n";
run_test_case(asm_ebnf::identifier, "valid_var1", true, U"valid_var1");
run_test_case(asm_ebnf::identifier, "_private", true, U"_private");
run_test_case(asm_ebnf::identifier, "123invalid", false);
run_test_case(asm_ebnf::decimal_lit, "1234", true);
run_test_case(asm_ebnf::decimal_lit, "-567L", true);
run_test_case(asm_ebnf::hex_lit, "0x1A2B", true);
run_test_case(asm_ebnf::octal_lit, "0c755", true);
run_test_case(asm_ebnf::binary_lit, "0b10101", true);
run_test_case(asm_ebnf::float_lit, "3.14159F", true);
run_test_case(asm_ebnf::float_lit, "1e-10D", true);
run_test_case(asm_ebnf::char_lit, "'a'", true);
run_test_case(asm_ebnf::char_lit, "'\\n'", true);
run_test_case(asm_ebnf::string_lit, "\"Hello World\"", true);
run_test_case(asm_ebnf::string_lit, "\"Escape \\\" Test\"", true);
}
void test_addressing_modes(TokenFactory& tf) {
std::cout << "\n--- Testing Addressing Modes ---\n";
run_test_case(asm_ebnf::register_tok, "R0", true);
run_test_case(asm_ebnf::register_tok, "R15", true);
run_test_case(asm_ebnf::addrm_ind, "[ 0x1000 ]", true);
run_test_case(asm_ebnf::addrm_ptr, "[R1]", true);
run_test_case(asm_ebnf::addrm_idx, "[R1 + 4]", true);
run_test_case(asm_ebnf::addrm_sca, "[R1 + R2 * 4]", true);
run_test_case(asm_ebnf::addrm_dis, "[ R1 + R2 * 4 + 16 ]", true);
}
void test_instructions_and_lines(TokenFactory& tf) {
std::cout << "\n--- Testing Instructions & Lines ---\n";
run_test_case(asm_ebnf::instruction, "NOP", true);
run_test_case(asm_ebnf::instruction, "MOV R1, [R2 + 4]", true);
run_test_case(asm_ebnf::instruction, "ADD R0, 100", true);
run_test_case(asm_ebnf::preprocessor, "#define MAX_BUF", true);
run_test_case(asm_ebnf::annotation, "@inline", true);
run_test_case(asm_ebnf::annotation, "@align(4)", true);
run_test_case(asm_ebnf::annotation, "@section(name=\"text\", flags=1)", true);
run_test_case(asm_ebnf::line, "main: MOV R0, R1 ; copy reg\n", true);
run_test_case(asm_ebnf::line, " @deprecated NOP\n", true);
run_test_case(asm_ebnf::line, "; only a comment line\n", true);
}
void test_full_program(TokenFactory& tf) {
std::cout << "\n--- Testing Full Program Parser ---\n";
std::string asm_code =
"#include stdio\n"
"\n"
"start:\n"
" MOV R1, 0x20 ; Load constant\n"
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
" JMP start\n";
StringTextReader reader(asm_code);
TokenResult res = asm_ebnf::program->test(reader);
if (res.success) {
std::cout << "[PASS] Full Assembly Program parsed successfully!\n";
} else { } else {
std::cout << " [FAIL] " << testName << "\n"; std::cerr << "[FAIL] Program parsing failed.\n";
} }
}
void printSummary() const {
std::cout << "\n========================================\n";
std::cout << "Test Results: " << passedTests << "/" << totalTests << " passed.\n";
std::cout << "========================================\n";
}
};
// ============================================================================
// TEST SUITES FOR StringTextReader
// ============================================================================
void test_basic_reading(TestRunner& runner) {
std::cout << "\n--- Running: Basic Reading Tests ---\n";
std::cout.flush();
StringTextReader reader("hello");
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'h', "Initial current character is 'h'");
runner.assertCondition(reader.peekChar(1).has_value() && reader.peekChar(1).value() == 'e', "Peek +1 char is 'e'");
auto next = reader.nextChar(1);
runner.assertCondition(next.has_value() && next.value() == 'e', "Advance to next char gives 'e'");
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'e', "Current character is now 'e'");
}
void test_eat_operations(TestRunner& runner) {
std::cout << "\n--- Running: Eat Operations Tests ---\n";
std::cout.flush();
StringTextReader reader("constexpr int x = 42;");
runner.assertCondition(reader.eat("constexpr"), "Eat exact string match 'constexpr'");
runner.assertCondition(reader.eat(' '), "Eat single space character");
runner.assertCondition(reader.eat("int"), "Eat second string match 'int'");
runner.assertCondition(!reader.eat("float"), "Eat fails on mismatched string 'float'");
runner.assertCondition(reader.eat(' '), "Eat single space after failure");
runner.assertCondition(reader.eat('x'), "Eat character 'x'");
}
void test_push_pop_rollback(TestRunner& runner) {
std::cout << "\n--- Running: Push/Pop Rollback Tests ---\n";
std::cout.flush();
StringTextReader reader("function_name()");
auto savedPos = reader.push();
runner.assertCondition(reader.eat("function_"), "Incomplete parse attempt");
// Rollback
reader.pop(savedPos);
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'f', "Rollback restores cursor to 'f'");
runner.assertCondition(reader.eat("function_name"), "Subsequent match succeeds after rollback");
}
void test_commit(TestRunner& runner) {
std::cout << "\n--- Running: Buffer Commit Tests ---\n";
std::cout.flush();
StringTextReader reader("line1\nline2");
reader.eat("line1\n");
reader.commit(); // Discard historical rollback buffer
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'l', "Current char after commit is 'l'");
runner.assertCondition(reader.eat("line2"), "Reading continues normally after commit");
}
void test_string_mutations(TestRunner& runner) {
std::cout << "\n--- Running: String Mutation Tests (set/append) ---\n";
std::cout.flush();
StringTextReader reader("foo");
runner.assertCondition(reader.eat("foo"), "Read initial text 'foo'");
reader.set("reset_text");
runner.assertCondition(reader.eat("reset_text"), "Read completely new text after set()");
}
void test_eof_handling(TestRunner& runner) {
std::cout << "\n--- Running: EOF & Error State Tests ---\n";
std::cout.flush();
StringTextReader reader("a");
runner.assertCondition(static_cast<bool>(reader), "Reader is valid initially");
runner.assertCondition(!reader.isEOF(), "isEOF is false initially");
reader.eat('a');
std::cout << "Index: " << reader.push().index << std::endl;
std::cout << "Value: " << reader.current().value_or(0) << std::endl;
runner.assertCondition(!reader.current().has_value(), "current() returns empty optional at EOF");
runner.assertCondition(reader.isEOF(), "isEOF is true after consuming all input");
runner.assertCondition(!reader.hasError(), "hasError remains false on normal EOF");
} }
// ============================================================================ // ============================================================================
// MAIN DRIVER // Main Execution
// ============================================================================ // ============================================================================
int main() { int main() {
std::cout << "========================================\n"; TokenFactory tf;
std::cout << " StringTextReader Unit Test Suite \n"; asm_ebnf::initTokens(tf);
std::cout << "========================================\n";
TestRunner runner; std::cout << "Running Token Framework Tests...\n";
test_basic_reading(runner);
test_eat_operations(runner);
test_push_pop_rollback(runner);
test_commit(runner);
test_string_mutations(runner);
test_eof_handling(runner);
runner.printSummary();
test_primitives_and_literals(tf);
test_choice_and_seq(tf);
test_opt_and_rep(tf);
test_literals(tf);
test_addressing_modes(tf);
test_instructions_and_lines(tf);
test_full_program(tf);
std::cout << "All token framework tests passed successfully!\n";
return 0; return 0;
} }
@@ -2,10 +2,6 @@
namespace spider::asm_ebnf { namespace spider::asm_ebnf {
// Token Factory
//TokenFactory tf;
// Char Functions // Char Functions
bool isUTF8Alpha(u32 ch) { bool isUTF8Alpha(u32 ch) {
@@ -75,4 +75,6 @@ namespace spider::asm_ebnf {
extern const Token* line_last; extern const Token* line_last;
extern const Token* program; extern const Token* program;
void initTokens(TokenFactory& tf);
} }
+25 -11
View File
@@ -10,7 +10,7 @@ namespace spider {
Token* TokenFactory::lit(std::string_view text) { Token* TokenFactory::lit(std::string_view text) {
auto it = lit_cache.find(std::string(text)); auto it = lit_cache.find(std::string(text));
if(it != lit_cache.end()) return it->second; if (it != lit_cache.end()) return it->second;
auto p = std::make_unique<LitToken>(text); auto p = std::make_unique<LitToken>(text);
auto t = p.get(); auto t = p.get();
@@ -39,15 +39,17 @@ namespace spider {
Token* TokenFactory::choice(std::string_view opts) { Token* TokenFactory::choice(std::string_view opts) {
vector<const Token*> toks; vector<const Token*> toks;
for(char c : opts) {
for (char c : opts) {
std::string s = std::string(1, c); std::string s = std::string(1, c);
toks.push_back(lit(s)); toks.push_back(lit(s));
} }
return choice(toks); return choice(toks);
} }
Token* TokenFactory::choice(const vector<const Token*>& tokens) { Token* TokenFactory::choice(const vector<const Token*>& tokens) {
uptr<Token> p = std::make_unique<SeqToken>(tokens); uptr<Token> p = std::make_unique<OrToken>(tokens);
auto t = p.get(); auto t = p.get();
arena.emplace_back(std::move(p)); arena.emplace_back(std::move(p));
return t; return t;
@@ -74,6 +76,12 @@ namespace spider {
return t; return t;
} }
std::u32string TokenResult::flatMatch() {
std::u32string s = match;
for(auto c : child) s += c.flatMatch();
return s;
}
// ============================================================================ // ============================================================================
// LitToken Implementation // LitToken Implementation
// ============================================================================ // ============================================================================
@@ -108,6 +116,7 @@ namespace spider {
} }
r.success = !r.match.empty(); r.success = !r.match.empty();
if(r.success) std::cout << "[fn] matched: " << utf8::toUTF8(r.flatMatch()) << std::endl;
return r; return r;
} }
@@ -156,7 +165,10 @@ namespace spider {
for (const auto& token_ref : tokens) { for (const auto& token_ref : tokens) {
// Short-circuit branch: return immediately on first valid choice match // Short-circuit branch: return immediately on first valid choice match
TokenResult res = token_ref->test(ctx); TokenResult res = token_ref->test(ctx);
if (res.success) return res; if (res.success) {
std::cout << "[or] matched: " << utf8::toUTF8(res.flatMatch()) << std::endl;
return res;
}
// Backtrack isolation: Reset the cursor position before testing the next alternative path // Backtrack isolation: Reset the cursor position before testing the next alternative path
ctx.pop(i); ctx.pop(i);
@@ -176,13 +188,14 @@ namespace spider {
TokenResult res = target->test(ctx); TokenResult res = target->test(ctx);
if (res.success) { if (res.success) {
std::cout << "[~] matched: " << utf8::toUTF8(res.flatMatch()) << std::endl;
return res; // Option matched exactly 1 instance successfully return res; // Option matched exactly 1 instance successfully
} }
// Recovery path: If sub-rule fails, clean up the dirty state mutation // Recovery path: If sub-rule fails, clean up the dirty state mutation
// and successfully return an empty match payload (0 instances). // and successfully return an empty match payload (0 instances).
ctx.pop(tri); ctx.pop(tri);
return { .success = false }; return { .success = true };
} }
// ============================================================================ // ============================================================================
@@ -197,12 +210,14 @@ namespace spider {
for (;;) { for (;;) {
auto i = ctx.push(); auto i = ctx.push();
TokenResult res = target->test(ctx); TokenResult res = target->test(ctx);
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching // Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups). // rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
if (!res.success || i.index == ctx.push().index) { if (!res.success || i.index == ctx.push().index) {
ctx.pop(i); ctx.pop(i);
break; break;
} }
//r.match += res.match; //r.match += res.match;
r.child.push_back(res); r.child.push_back(res);
} }
@@ -210,13 +225,15 @@ namespace spider {
// Repetition rules (* token) always evaluate to successful // Repetition rules (* token) always evaluate to successful
// completion state, even with 0 matches. // completion state, even with 0 matches.
r.success = true; r.success = true;
std::cout << "[*] matched: " << utf8::toUTF8(r.flatMatch()) << std::endl;
return r; return r;
} }
// Tagged Token // Tagged Token
TagToken::TagToken(const Token* t, std::string_view tag, bool doflatten) TagToken::TagToken(const Token* t, std::string_view tag, bool doflatten)
: target(t), tag_name(tag), flatten(doflatten) { } : target(t), tag_name(tag), flatten(doflatten) {
}
TokenResult TagToken::test(TextReader& ctx) const { TokenResult TagToken::test(TextReader& ctx) const {
auto r = target->test(ctx); auto r = target->test(ctx);
@@ -225,11 +242,8 @@ namespace spider {
// Set tag of this result // Set tag of this result
r.tag = tag_name; r.tag = tag_name;
if(flatten) { if (flatten) {
r.match = U""; r.match = r.flatMatch();
for(auto e : r.child) {
r.match += e.match;
}
r.child.clear(); r.child.clear();
} }
+5 -1
View File
@@ -2,7 +2,7 @@
#include <spider/compiler/common.hpp> #include <spider/compiler/common.hpp>
#include <spider/compiler/text/utf8.hpp> #include <spider/compiler/text/unicode.hpp>
#include <spider/compiler/text/TextReader.hpp> #include <spider/compiler/text/TextReader.hpp>
namespace spider { namespace spider {
@@ -25,6 +25,10 @@ namespace spider {
vector<TokenResult> child = {}; vector<TokenResult> child = {};
public:
std::u32string flatMatch();
}; };
class Token; class Token;
@@ -8,7 +8,7 @@
namespace spider { namespace spider {
namespace utf8 { namespace unicode {
// --------------------- // // --------------------- //
// UTF-8 Sequence Length // // UTF-8 Sequence Length //
@@ -95,6 +95,55 @@ namespace spider {
return _i == csize; return _i == csize;
} }
// ----------------- //
// UTF-32 into UTF-8 //
// ----------------- //
inline void append_utf32_to_utf8(u32 code_point, std::string& out) {
if (code_point <= 0x7F) {
// 1-byte sequence (ASCII)
out.push_back(static_cast<char>(code_point));
} else if (code_point <= 0x7FF) {
// 2-byte sequence
out.push_back(static_cast<char>(0xC0 | ((code_point >> 6) & 0x1F)));
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
} else if (code_point <= 0xFFFF) {
// 3-byte sequence
// Filter out surrogate pairs (U+D800 to U+DFFF) as they are invalid Unicode scalar values
if (code_point >= 0xD800 && code_point <= 0xDFFF) {
code_point = 0xFFFD; // Replacement character
}
out.push_back(static_cast<char>(0xE0 | ((code_point >> 12) & 0x0F)));
out.push_back(static_cast<char>(0x80 | ((code_point >> 6) & 0x3F)));
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
} else if (code_point <= 0x10FFFF) {
// 4-byte sequence
out.push_back(static_cast<char>(0xF0 | ((code_point >> 18) & 0x07)));
out.push_back(static_cast<char>(0x80 | ((code_point >> 12) & 0x3F)));
out.push_back(static_cast<char>(0x80 | ((code_point >> 6) & 0x3F)));
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
} else {
// Code point out of Unicode range -> insert UTF-8 replacement character U+FFFD
append_utf32_to_utf8(0xFFFD, out);
}
}
inline std::string toUTF8(u32 cp) {
std::string s;
append_utf32_to_utf8(cp, s);
return s;
}
inline std::string toUTF8(const std::u32string& str) {
std::string s;
for(u32 ch : str) append_utf32_to_utf8(ch, s);
return s;
}
// ----------------- //
// STRINGS //
// ----------------- //
inline const char* getControlCharName(u8 c) { inline const char* getControlCharName(u8 c) {
static const char* names[32] = { static const char* names[32] = {
"NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL", "NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL",
@@ -107,7 +156,7 @@ namespace spider {
return nullptr; return nullptr;
} }
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) { inline void hexdump_utf8(const char* data, isize length, pos at, std::ostream& ostr) {
auto old_flags = ostr.flags(); auto old_flags = ostr.flags();
auto old_fill = ostr.fill(); auto old_fill = ostr.fill();