From 897e155b0e08958f7bdce7ff90159031d17d64ac Mon Sep 17 00:00:00 2001 From: Kittycannon Date: Tue, 4 Aug 2026 14:22:57 -0600 Subject: [PATCH] text reader passes tests --- src/spider/compiler/Compiler.cpp | 130 ++++++++++- src/spider/compiler/assembler/AsmEBNF.cpp | 251 ++++++++++++++-------- src/spider/compiler/text/TextReader.cpp | 110 +++++----- src/spider/compiler/text/TextReader.hpp | 51 +++-- src/spider/compiler/text/Token.cpp | 16 +- src/spider/compiler/text/Token.hpp | 10 +- 6 files changed, 406 insertions(+), 162 deletions(-) diff --git a/src/spider/compiler/Compiler.cpp b/src/spider/compiler/Compiler.cpp index dd2d113..0bd6f23 100644 --- a/src/spider/compiler/Compiler.cpp +++ b/src/spider/compiler/Compiler.cpp @@ -3,12 +3,140 @@ #include #include -namespace spider { +#include +using namespace spider; +class TestRunner { +private: + int totalTests = 0; + int passedTests = 0; +public: + void assertCondition(bool condition, const std::string& testName) { + totalTests++; + if (condition) { + std::cout << " [PASS] " << testName << "\n"; + passedTests++; + } else { + std::cout << " [FAIL] " << testName << "\n"; + } + } + + void printSummary() const { + std::cout << "\n========================================\n"; + std::cout << "Test Results: " << passedTests << "/" << totalTests << " passed.\n"; + std::cout << "========================================\n"; + } +}; + +// ============================================================================ +// TEST SUITES FOR StringTextReader +// ============================================================================ + +void test_basic_reading(TestRunner& runner) { + std::cout << "\n--- Running: Basic Reading Tests ---\n"; + std::cout.flush(); + + StringTextReader reader("hello"); + + runner.assertCondition(reader.current().has_value() && reader.current().value() == 'h', "Initial current character is 'h'"); + runner.assertCondition(reader.peekChar(1).has_value() && reader.peekChar(1).value() == 'e', "Peek +1 char is 'e'"); + + auto next = reader.nextChar(1); + runner.assertCondition(next.has_value() && next.value() == 'e', "Advance to next char gives 'e'"); + runner.assertCondition(reader.current().has_value() && reader.current().value() == 'e', "Current character is now 'e'"); } +void test_eat_operations(TestRunner& runner) { + std::cout << "\n--- Running: Eat Operations Tests ---\n"; + std::cout.flush(); + + StringTextReader reader("constexpr int x = 42;"); + + runner.assertCondition(reader.eat("constexpr"), "Eat exact string match 'constexpr'"); + runner.assertCondition(reader.eat(' '), "Eat single space character"); + runner.assertCondition(reader.eat("int"), "Eat second string match 'int'"); + + runner.assertCondition(!reader.eat("float"), "Eat fails on mismatched string 'float'"); + runner.assertCondition(reader.eat(' '), "Eat single space after failure"); + runner.assertCondition(reader.eat('x'), "Eat character 'x'"); +} + +void test_push_pop_rollback(TestRunner& runner) { + std::cout << "\n--- Running: Push/Pop Rollback Tests ---\n"; + std::cout.flush(); + + StringTextReader reader("function_name()"); + + auto savedPos = reader.push(); + runner.assertCondition(reader.eat("function_"), "Incomplete parse attempt"); + + // Rollback + reader.pop(savedPos); + runner.assertCondition(reader.current().has_value() && reader.current().value() == 'f', "Rollback restores cursor to 'f'"); + runner.assertCondition(reader.eat("function_name"), "Subsequent match succeeds after rollback"); +} + +void test_commit(TestRunner& runner) { + std::cout << "\n--- Running: Buffer Commit Tests ---\n"; + std::cout.flush(); + + StringTextReader reader("line1\nline2"); + + reader.eat("line1\n"); + reader.commit(); // Discard historical rollback buffer + + runner.assertCondition(reader.current().has_value() && reader.current().value() == 'l', "Current char after commit is 'l'"); + runner.assertCondition(reader.eat("line2"), "Reading continues normally after commit"); +} + +void test_string_mutations(TestRunner& runner) { + std::cout << "\n--- Running: String Mutation Tests (set/append) ---\n"; + std::cout.flush(); + + StringTextReader reader("foo"); + runner.assertCondition(reader.eat("foo"), "Read initial text 'foo'"); + + reader.set("reset_text"); + runner.assertCondition(reader.eat("reset_text"), "Read completely new text after set()"); +} + +void test_eof_handling(TestRunner& runner) { + std::cout << "\n--- Running: EOF & Error State Tests ---\n"; + std::cout.flush(); + + StringTextReader reader("a"); + + runner.assertCondition(static_cast(reader), "Reader is valid initially"); + runner.assertCondition(!reader.isEOF(), "isEOF is false initially"); + + reader.eat('a'); + std::cout << "Index: " << reader.push().index << std::endl; + std::cout << "Value: " << reader.current().value_or(0) << std::endl; + runner.assertCondition(!reader.current().has_value(), "current() returns empty optional at EOF"); + runner.assertCondition(reader.isEOF(), "isEOF is true after consuming all input"); + runner.assertCondition(!reader.hasError(), "hasError remains false on normal EOF"); +} + +// ============================================================================ +// MAIN DRIVER +// ============================================================================ + int main() { + std::cout << "========================================\n"; + std::cout << " StringTextReader Unit Test Suite \n"; + std::cout << "========================================\n"; + + TestRunner runner; + test_basic_reading(runner); + test_eat_operations(runner); + test_push_pop_rollback(runner); + test_commit(runner); + test_string_mutations(runner); + test_eof_handling(runner); + runner.printSummary(); + return 0; } + diff --git a/src/spider/compiler/assembler/AsmEBNF.cpp b/src/spider/compiler/assembler/AsmEBNF.cpp index c2e26a6..7e35218 100644 --- a/src/spider/compiler/assembler/AsmEBNF.cpp +++ b/src/spider/compiler/assembler/AsmEBNF.cpp @@ -4,7 +4,7 @@ namespace spider::asm_ebnf { // Token Factory - TokenFactory tf; + //TokenFactory tf; // Char Functions @@ -28,117 +28,190 @@ namespace spider::asm_ebnf { return ch != u32('"'); } - // (* Characters & Basic Predicates *) - const Token* letter = tf.fn(isUTF8Alpha); - const Token* digit = tf.choice("0123456789"); - const Token* alpha_num_char = tf.choice({ letter, digit }); + const Token* letter; + const Token* digit; + const Token* alpha_num_char; - const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef"); - const Token* octal_digit = tf.choice("01234567"); - const Token* binary_digit = tf.choice("01"); + const Token* hex_digit; + const Token* octal_digit; + const Token* binary_digit; - const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf); - const Token* ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true); - const Token* whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true); - const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }); - const Token* utf8_char = tf.fn(isUTF8CharNotCrLf); + const Token* ws_char; + const Token* ws_optional; + const Token* whitespace; + const Token* newline; + const Token* utf8_char; - const Token* char_escape = tf.seq({ tf["\\"], utf8_char }); - const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) }); - const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] }); + const Token* char_escape; + const Token* char_content; + const Token* char_lit; - const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) }); - const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }); + const Token* string_char; + const Token* string_lit; - // (* Literals *) - const Token* identifier = tf.tag(tf.seq({ - tf.choice({ letter, tf["_"] }), - tf.rep(tf.choice({ alpha_num_char, tf["_"] })) - }), "identifier", true); + const Token* identifier; + const Token* comment; - const Token* comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true); + const Token* sign; + const Token* exponent_marker; + const Token* exponent; - const Token* sign = tf.choice("+-"); - const Token* exponent_marker = tf.choice("eE"); - const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) }); + const Token* decimal_lit; + const Token* float_lit; - const Token* decimal_lit = tf.tag(tf.seq({ - tf.opt(sign), - digit, - tf.rep(digit), - tf.opt(tf.choice("BSIL")) - }), "decimal_lit", true); + const Token* hex_lit; + const Token* octal_lit; + const Token* binary_lit; - const Token* float_lit = tf.tag(tf.seq({ - tf.opt(sign), - tf.choice({ - tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }), - tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }), - tf.seq({ digit, tf.rep(digit), exponent }) - }), - tf.opt(tf.choice("FD")) - }), "float_lit", true); + const Token* literal; + const Token* literal_cast; + const Token* literal_decl; - const Token* hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true); - const Token* octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true); - const Token* binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true); + const Token* register_tok; - const Token* literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal"); - const Token* literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast"); - const Token* literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl"); + const Token* addrm_ind; + const Token* addrm_ptr; + const Token* addrm_idx; + const Token* addrm_sca; + const Token* addrm_dis; - // (* Operands *) - const Token* register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true); + const Token* addr_modes; + const Token* operand; - const Token* addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true); - const Token* addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true); + const Token* opcode; + const Token* operand_list; + const Token* instruction; - const Token* addrm_idx = tf.tag(tf.seq({ - tf["["], ws_optional, register_tok, ws_optional, - tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] - }), "addrm_idx", true); + const Token* annotation_named; + const Token* annotation_arg; + const Token* annotation_args; + const Token* annotation_pars; + const Token* annotation; - const Token* addrm_sca = tf.tag(tf.seq({ - tf["["], ws_optional, register_tok, ws_optional, - tf["+"], ws_optional, register_tok, ws_optional, - tf["*"], ws_optional, literal_decl, ws_optional, tf["]"] - }), "addrm_sca", true); + const Token* preprocessor_val; + const Token* preprocessor; - const Token* addrm_dis = tf.tag(tf.seq({ - tf["["], ws_optional, register_tok, ws_optional, - tf["+"], ws_optional, register_tok, ws_optional, - tf["*"], ws_optional, literal_decl, ws_optional, - tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] - }), "addrm_dis", true); + const Token* label; + const Token* line_label; + const Token* line_annotation; + const Token* line_content; + const Token* line; + const Token* line_last; + const Token* program; - const Token* addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm"); - const Token* operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand"); + void initTokens(TokenFactory& tf) { + // (* Characters & Basic Predicates *) + letter = tf.fn(isUTF8Alpha); + digit = tf.choice("0123456789"); + alpha_num_char = tf.choice({ letter, digit }); - // (* Generalized Instructions *) + hex_digit = tf.choice("0123456789ABCDEFabcdef"); + octal_digit = tf.choice("01234567"); + binary_digit = tf.choice("01"); - const Token* opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true); - const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }); - const Token* instruction = tf.tag(tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction"); + ws_char = tf.fn(isWhithespaceCharNotCrLf); + ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true); + whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true); + newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }); + utf8_char = tf.fn(isUTF8CharNotCrLf); - // (* Added Preprocessor, Annotation *) + char_escape = tf.seq({ tf["\\"], utf8_char }); + char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) }); + char_lit = tf.seq({ tf["'"], char_content, tf["'"] }); - const Token* annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named"); - const Token* annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg"); - const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }); - const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }); - const Token* annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation"); + string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) }); + string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }); - const Token* preprocessor_val = tf.choice({ identifier, literal_decl }); - const Token* preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor"); + // (* Literals *) + identifier = tf.tag(tf.seq({ + tf.choice({ letter, tf["_"] }), + tf.rep(tf.choice({ alpha_num_char, tf["_"] })) + }), "identifier", true); - // (* Line Structure & Program *) + comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true); - const Token* label = tf.tag(tf.seq({ identifier, tf[":"] }), "label"); - const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }); - const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }); - const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction }); - const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }); - const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }); - const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) }); + sign = tf.choice("+-"); + exponent_marker = tf.choice("eE"); + exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) }); + + decimal_lit = tf.tag(tf.seq({ + tf.opt(sign), + digit, + tf.rep(digit), + tf.opt(tf.choice("BSIL")) + }), "decimal_lit", true); + + float_lit = tf.tag(tf.seq({ + tf.opt(sign), + tf.choice({ + tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }), + tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }), + tf.seq({ digit, tf.rep(digit), exponent }) + }), + tf.opt(tf.choice("FD")) + }), "float_lit", true); + + hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true); + octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true); + binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true); + + literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal"); + literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast"); + literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl"); + + // (* Operands *) + register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true); + + addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true); + addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true); + + addrm_idx = tf.tag(tf.seq({ + tf["["], ws_optional, register_tok, ws_optional, + tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] + }), "addrm_idx", true); + + addrm_sca = tf.tag(tf.seq({ + tf["["], ws_optional, register_tok, ws_optional, + tf["+"], ws_optional, register_tok, ws_optional, + tf["*"], ws_optional, literal_decl, ws_optional, tf["]"] + }), "addrm_sca", true); + + addrm_dis = tf.tag(tf.seq({ + tf["["], ws_optional, register_tok, ws_optional, + tf["+"], ws_optional, register_tok, ws_optional, + tf["*"], ws_optional, literal_decl, ws_optional, + tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] + }), "addrm_dis", true); + + addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm"); + operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand"); + + // (* Generalized Instructions *) + + opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true); + operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }); + instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction"); + + // (* Added Preprocessor, Annotation *) + + annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named"); + annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg"); + annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }); + annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }); + annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation"); + + preprocessor_val = tf.choice({ identifier, literal_decl }); + preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor"); + + // (* Line Structure & Program *) + + label = tf.tag(tf.seq({ identifier, tf[":"] }), "label"); + line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }); + line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }); + line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction }); + line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }); + line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }); + program = tf.seq({ tf.rep(line), tf.opt(line_last) }); + } } diff --git a/src/spider/compiler/text/TextReader.cpp b/src/spider/compiler/text/TextReader.cpp index 75accde..6f157ca 100644 --- a/src/spider/compiler/text/TextReader.cpp +++ b/src/spider/compiler/text/TextReader.cpp @@ -8,11 +8,7 @@ namespace spider { // Text Reader // - TextReader::TextReader() : err(false), eof(false), bufferIndex(0) { - // Prime the buffer with the first character - // so current() is immediately valid - fillBufferTo(0); - } + TextReader::TextReader() : err(false), eof(false), bufferIndex(0) {} TextReader::~TextReader() {} @@ -37,14 +33,25 @@ namespace spider { } bool TextReader::eat(const std::u32string& str) { + // case 0: no str + if(str.empty()) return true; + + // prepare n chars + isize index_space = str.size() - 1; + fillBufferTo(index_space); + + // fast reject + if(!hasBufferTo(index_space)) return false; + // compare now - isize index; - for(index = 0; index < str.size(); index++) { - if(str[index] != peekChar(index)) return false; + for(isize i = 0; i <= index_space; i++) { + if(str[i] != buffer[bufferIndex + i]) { + return false; + } } // success! - nextChar(index); + consumeChars(str.size()); return true; } @@ -68,37 +75,30 @@ namespace spider { return char(ch); } - /** - * Returns the current character. - */ optional TextReader::current() { + fillBufferTo(0); if (bufferIndex < buffer.size()) { return buffer[bufferIndex]; } return {}; } - /** - * Reads the next character and advances the position tracker. - */ optional TextReader::nextChar(isize n) { // Ensure the character we are moving TO exists - if (fillBufferTo(n)) { - // advance n characters - while(n--) { - advance(buffer[bufferIndex]); - bufferIndex++; - } - return current(); - } - return {}; + fillBufferTo(n); // index = n will be accessible + // from [0, n] inclusive, equal to (n + 1) chars + + // remember partial success + consumeChars(n); // n chars will be removed + + // return current char + // current char, index = 0 + return current(); } - /** - * Keeps the next n-th character (n = 0 is current). - */ optional TextReader::peekChar(isize n) { - if (fillBufferTo(n)) return buffer[bufferIndex + n]; + fillBufferTo(n); + if (hasBufferTo(n)) return buffer[bufferIndex + n]; return {}; } @@ -110,12 +110,16 @@ namespace spider { } } - isize TextReader::push() { - return bufferIndex; + TextReader::State TextReader::push() { + return { .err = err, .eof = eof, .at = at, .errmsg = errmsg, .index = bufferIndex }; } - void TextReader::pop(isize index) { - bufferIndex = std::min(index, bufferIndex); + void TextReader::pop(TextReader::State s) { + err = s.err; + eof = s.eof; + at = s.at; + errmsg = s.errmsg; + bufferIndex = s.index; } /** @@ -163,19 +167,27 @@ namespace spider { return true; } - /** - * Fills the buffer sequentially until it contains at least up - * to (bufferIndex + targetOffset). - */ bool TextReader::fillBufferTo(isize targetOffset) { - isize targetSize = bufferIndex + targetOffset + 1; - while (buffer.size() < targetSize) { + isize targetSize = bufferIndex + targetOffset; + while (targetSize >= buffer.size()) { if(readChar()) continue; return false; } return true; } + bool TextReader::hasBufferTo(isize index) { + return bufferIndex + index < buffer.size(); + } + + void TextReader::consumeChars(isize n) { + // advance up to specified char. + for(isize i = 0; i < n && hasBufferTo(i); i++) { + advance(buffer[bufferIndex]); + } + bufferIndex += n; + } + pos TextReader::getPosition() const { return at; } @@ -212,24 +224,24 @@ namespace spider { // String Reader // StringTextReader::StringTextReader(std::string initialText) - : buffer(std::move(initialText)), - stringStream(std::make_unique(buffer)) { - } + : txt_buffer(std::move(initialText)), + stringStream(std::make_unique(txt_buffer)) { } std::istream& StringTextReader::getStream() { return *stringStream; } void StringTextReader::set(const std::string& newText) { - buffer = newText; - stringStream = std::make_unique(buffer); - } + txt_buffer = newText; + stringStream = std::make_unique(txt_buffer); - void StringTextReader::append(const std::string& extraText) { - std::streampos pos = stringStream->tellg(); - buffer += extraText; - stringStream = std::make_unique(buffer); - stringStream->seekg(pos); + buffer.clear(); + txt_buffer.clear(); + + bufferIndex = 0; + err = false; + eof = false; + at = pos(); } } diff --git a/src/spider/compiler/text/TextReader.hpp b/src/spider/compiler/text/TextReader.hpp index 4bf1bb0..0fb96fd 100644 --- a/src/spider/compiler/text/TextReader.hpp +++ b/src/spider/compiler/text/TextReader.hpp @@ -34,11 +34,6 @@ namespace spider { std::string errmsg; - struct stored_char { - u8 byte_count; - u32 value; - }; - /** * Buffer of extracted characters. */ @@ -52,6 +47,16 @@ namespace spider { */ isize bufferIndex; + public: + + struct State { + bool err; + bool eof; + pos at; + std::string errmsg; + isize index; + }; + public: TextReader(); @@ -94,13 +99,16 @@ namespace spider { optional current(); /** - * Reads the next n-th character. + * Skips n number of characters and returns + * the current one in that position. * n = 0 is a noop, since it's the current one. + * + * Will advance until the EOF is reached. */ optional nextChar(isize n = 1); /** - * Keeps the next n-th character + * Returns the n-th character following the current one. * n = 0 is the current one. */ optional peekChar(isize n = 1); @@ -121,14 +129,14 @@ namespace spider { * Inside a parser, this allows to roll * back the index to a specific position. */ - isize push(); + State push(); /** * Sets the current buffer index. * Inside a parser, rolls back to * a previous position. */ - void pop(isize index); + void pop(State s); /** * Returns true if the end of the stream has been reached. @@ -160,8 +168,29 @@ namespace spider { virtual std::istream& getStream() = 0; + /** + * Fills the buffer sequentially until it the passed + * index can be safely accessed, relative to the current + * buffer position. + * + * Returns false if that index could not be reached. + * Partial success is possible, check buffer.size()! + */ bool fillBufferTo(isize index); + /** + * Verifies that the index can be safely accessed, + * relative to the current buffer position. + */ + bool hasBufferTo(isize index); + + /** + * Triggers the buffer to consume this number + * of characters from the buffer. This is a reverseable + * operation. + */ + void consumeChars(isize n); + }; /** @@ -188,7 +217,7 @@ namespace spider { class StringTextReader : public TextReader { private: - std::string buffer; + std::string txt_buffer; std::unique_ptr stringStream; public: @@ -199,8 +228,6 @@ namespace spider { void set(const std::string& newText); - void append(const std::string& extraText); - protected: std::istream& getStream() override; diff --git a/src/spider/compiler/text/Token.cpp b/src/spider/compiler/text/Token.cpp index 634bedd..376ad1f 100644 --- a/src/spider/compiler/text/Token.cpp +++ b/src/spider/compiler/text/Token.cpp @@ -9,9 +9,13 @@ namespace spider { // ============================================================================ Token* TokenFactory::lit(std::string_view text) { + auto it = lit_cache.find(std::string(text)); + if(it != lit_cache.end()) return it->second; + auto p = std::make_unique(text); auto t = p.get(); - lit_cache.emplace(text, std::move(p)); + arena.emplace_back(std::move(p)); + lit_cache.emplace(text, t); return t; } @@ -63,7 +67,7 @@ namespace spider { return t; } - Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten = false) { + Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten) { uptr p = std::make_unique(target, tagname, flatten); auto t = p.get(); arena.emplace_back(std::move(p)); @@ -111,12 +115,12 @@ namespace spider { // SeqToken Implementation // ============================================================================30520370 - SeqToken::SeqToken(const vector& tokens) : tokens(tokens) {} + SeqToken::SeqToken(const vector& _tokens) : tokens(_tokens) {} TokenResult SeqToken::test(TextReader& ctx) const { // this is a common branch point TokenResult r; - isize i = ctx.push(); + auto i = ctx.push(); // All matching steps within a sequence must pass consecutively. for (const auto& token_ref : tokens) { @@ -143,7 +147,7 @@ namespace spider { // OrToken Implementation // ============================================================================ - OrToken::OrToken(const vector& tokens) : tokens(tokens) {} + OrToken::OrToken(const vector& _tokens) : tokens(_tokens) {} TokenResult OrToken::test(TextReader& ctx) const { // All matching steps within a sequence must pass consecutively. @@ -195,7 +199,7 @@ namespace spider { TokenResult res = target->test(ctx); // Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching // rule succeeded without advancing the buffer index (prevents dynamic parsing lockups). - if (!res.success || i == ctx.push()) { + if (!res.success || i.index == ctx.push().index) { ctx.pop(i); break; } diff --git a/src/spider/compiler/text/Token.hpp b/src/spider/compiler/text/Token.hpp index b68e1df..12fcfc4 100644 --- a/src/spider/compiler/text/Token.hpp +++ b/src/spider/compiler/text/Token.hpp @@ -15,15 +15,15 @@ namespace spider { /** @brief Indicates if the token composition successfully matched the input boundary. */ bool success = false; - optional tag; + optional tag = {}; /** * @brief Holds the deep-copied UTF-32 matching substring upon victory. * @note Returns empty when success is false. */ - std::u32string match; + std::u32string match = U""; - vector child; + vector child = {}; }; @@ -61,7 +61,7 @@ namespace spider { std::vector> arena; // Deduplication caches - std::unordered_map lit_cache; + std::unordered_map lit_cache; public: @@ -238,7 +238,7 @@ namespace spider { public: - TagToken(const Token* t, std::string_view tag, bool doflatten = false); + TagToken(const Token* t, std::string_view tag, bool doflatten); TokenResult test(TextReader& ctx) const override;