text reader passes tests

This commit is contained in:
2026-08-04 14:22:57 -06:00
parent fa511520c6
commit 897e155b0e
6 changed files with 406 additions and 162 deletions
+129 -1
View File
@@ -3,12 +3,140 @@
#include <spider/compiler/common.hpp> #include <spider/compiler/common.hpp>
#include <spider/compiler/text/utf8.hpp> #include <spider/compiler/text/utf8.hpp>
namespace spider { #include <spider/compiler/text/TextReader.hpp>
using namespace spider;
class TestRunner {
private:
int totalTests = 0;
int passedTests = 0;
public:
void assertCondition(bool condition, const std::string& testName) {
totalTests++;
if (condition) {
std::cout << " [PASS] " << testName << "\n";
passedTests++;
} else {
std::cout << " [FAIL] " << testName << "\n";
}
}
void printSummary() const {
std::cout << "\n========================================\n";
std::cout << "Test Results: " << passedTests << "/" << totalTests << " passed.\n";
std::cout << "========================================\n";
}
};
// ============================================================================
// TEST SUITES FOR StringTextReader
// ============================================================================
void test_basic_reading(TestRunner& runner) {
std::cout << "\n--- Running: Basic Reading Tests ---\n";
std::cout.flush();
StringTextReader reader("hello");
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'h', "Initial current character is 'h'");
runner.assertCondition(reader.peekChar(1).has_value() && reader.peekChar(1).value() == 'e', "Peek +1 char is 'e'");
auto next = reader.nextChar(1);
runner.assertCondition(next.has_value() && next.value() == 'e', "Advance to next char gives 'e'");
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'e', "Current character is now 'e'");
} }
void test_eat_operations(TestRunner& runner) {
std::cout << "\n--- Running: Eat Operations Tests ---\n";
std::cout.flush();
StringTextReader reader("constexpr int x = 42;");
runner.assertCondition(reader.eat("constexpr"), "Eat exact string match 'constexpr'");
runner.assertCondition(reader.eat(' '), "Eat single space character");
runner.assertCondition(reader.eat("int"), "Eat second string match 'int'");
runner.assertCondition(!reader.eat("float"), "Eat fails on mismatched string 'float'");
runner.assertCondition(reader.eat(' '), "Eat single space after failure");
runner.assertCondition(reader.eat('x'), "Eat character 'x'");
}
void test_push_pop_rollback(TestRunner& runner) {
std::cout << "\n--- Running: Push/Pop Rollback Tests ---\n";
std::cout.flush();
StringTextReader reader("function_name()");
auto savedPos = reader.push();
runner.assertCondition(reader.eat("function_"), "Incomplete parse attempt");
// Rollback
reader.pop(savedPos);
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'f', "Rollback restores cursor to 'f'");
runner.assertCondition(reader.eat("function_name"), "Subsequent match succeeds after rollback");
}
void test_commit(TestRunner& runner) {
std::cout << "\n--- Running: Buffer Commit Tests ---\n";
std::cout.flush();
StringTextReader reader("line1\nline2");
reader.eat("line1\n");
reader.commit(); // Discard historical rollback buffer
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'l', "Current char after commit is 'l'");
runner.assertCondition(reader.eat("line2"), "Reading continues normally after commit");
}
void test_string_mutations(TestRunner& runner) {
std::cout << "\n--- Running: String Mutation Tests (set/append) ---\n";
std::cout.flush();
StringTextReader reader("foo");
runner.assertCondition(reader.eat("foo"), "Read initial text 'foo'");
reader.set("reset_text");
runner.assertCondition(reader.eat("reset_text"), "Read completely new text after set()");
}
void test_eof_handling(TestRunner& runner) {
std::cout << "\n--- Running: EOF & Error State Tests ---\n";
std::cout.flush();
StringTextReader reader("a");
runner.assertCondition(static_cast<bool>(reader), "Reader is valid initially");
runner.assertCondition(!reader.isEOF(), "isEOF is false initially");
reader.eat('a');
std::cout << "Index: " << reader.push().index << std::endl;
std::cout << "Value: " << reader.current().value_or(0) << std::endl;
runner.assertCondition(!reader.current().has_value(), "current() returns empty optional at EOF");
runner.assertCondition(reader.isEOF(), "isEOF is true after consuming all input");
runner.assertCondition(!reader.hasError(), "hasError remains false on normal EOF");
}
// ============================================================================
// MAIN DRIVER
// ============================================================================
int main() { int main() {
std::cout << "========================================\n";
std::cout << " StringTextReader Unit Test Suite \n";
std::cout << "========================================\n";
TestRunner runner;
test_basic_reading(runner);
test_eat_operations(runner);
test_push_pop_rollback(runner);
test_commit(runner);
test_string_mutations(runner);
test_eof_handling(runner);
runner.printSummary();
return 0; return 0;
} }
+162 -89
View File
@@ -4,7 +4,7 @@ namespace spider::asm_ebnf {
// Token Factory // Token Factory
TokenFactory tf; //TokenFactory tf;
// Char Functions // Char Functions
@@ -28,117 +28,190 @@ namespace spider::asm_ebnf {
return ch != u32('"'); return ch != u32('"');
} }
// (* Characters & Basic Predicates *) const Token* letter;
const Token* letter = tf.fn(isUTF8Alpha); const Token* digit;
const Token* digit = tf.choice("0123456789"); const Token* alpha_num_char;
const Token* alpha_num_char = tf.choice({ letter, digit });
const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef"); const Token* hex_digit;
const Token* octal_digit = tf.choice("01234567"); const Token* octal_digit;
const Token* binary_digit = tf.choice("01"); const Token* binary_digit;
const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf); const Token* ws_char;
const Token* ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true); const Token* ws_optional;
const Token* whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true); const Token* whitespace;
const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }); const Token* newline;
const Token* utf8_char = tf.fn(isUTF8CharNotCrLf); const Token* utf8_char;
const Token* char_escape = tf.seq({ tf["\\"], utf8_char }); const Token* char_escape;
const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) }); const Token* char_content;
const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] }); const Token* char_lit;
const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) }); const Token* string_char;
const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }); const Token* string_lit;
// (* Literals *) const Token* identifier;
const Token* identifier = tf.tag(tf.seq({ const Token* comment;
tf.choice({ letter, tf["_"] }),
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
}), "identifier", true);
const Token* comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true); const Token* sign;
const Token* exponent_marker;
const Token* exponent;
const Token* sign = tf.choice("+-"); const Token* decimal_lit;
const Token* exponent_marker = tf.choice("eE"); const Token* float_lit;
const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
const Token* decimal_lit = tf.tag(tf.seq({ const Token* hex_lit;
tf.opt(sign), const Token* octal_lit;
digit, const Token* binary_lit;
tf.rep(digit),
tf.opt(tf.choice("BSIL"))
}), "decimal_lit", true);
const Token* float_lit = tf.tag(tf.seq({ const Token* literal;
tf.opt(sign), const Token* literal_cast;
tf.choice({ const Token* literal_decl;
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ digit, tf.rep(digit), exponent })
}),
tf.opt(tf.choice("FD"))
}), "float_lit", true);
const Token* hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true); const Token* register_tok;
const Token* octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true);
const Token* binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true);
const Token* literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal"); const Token* addrm_ind;
const Token* literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast"); const Token* addrm_ptr;
const Token* literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl"); const Token* addrm_idx;
const Token* addrm_sca;
const Token* addrm_dis;
// (* Operands *) const Token* addr_modes;
const Token* register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true); const Token* operand;
const Token* addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true); const Token* opcode;
const Token* addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true); const Token* operand_list;
const Token* instruction;
const Token* addrm_idx = tf.tag(tf.seq({ const Token* annotation_named;
tf["["], ws_optional, register_tok, ws_optional, const Token* annotation_arg;
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] const Token* annotation_args;
}), "addrm_idx", true); const Token* annotation_pars;
const Token* annotation;
const Token* addrm_sca = tf.tag(tf.seq({ const Token* preprocessor_val;
tf["["], ws_optional, register_tok, ws_optional, const Token* preprocessor;
tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_sca", true);
const Token* addrm_dis = tf.tag(tf.seq({ const Token* label;
tf["["], ws_optional, register_tok, ws_optional, const Token* line_label;
tf["+"], ws_optional, register_tok, ws_optional, const Token* line_annotation;
tf["*"], ws_optional, literal_decl, ws_optional, const Token* line_content;
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] const Token* line;
}), "addrm_dis", true); const Token* line_last;
const Token* program;
const Token* addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm"); void initTokens(TokenFactory& tf) {
const Token* operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand"); // (* Characters & Basic Predicates *)
letter = tf.fn(isUTF8Alpha);
digit = tf.choice("0123456789");
alpha_num_char = tf.choice({ letter, digit });
// (* Generalized Instructions *) hex_digit = tf.choice("0123456789ABCDEFabcdef");
octal_digit = tf.choice("01234567");
binary_digit = tf.choice("01");
const Token* opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true); ws_char = tf.fn(isWhithespaceCharNotCrLf);
const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }); ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true);
const Token* instruction = tf.tag(tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction"); whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true);
newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
utf8_char = tf.fn(isUTF8CharNotCrLf);
// (* Added Preprocessor, Annotation *) char_escape = tf.seq({ tf["\\"], utf8_char });
char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
const Token* annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named"); string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
const Token* annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg"); string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
const Token* annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation");
const Token* preprocessor_val = tf.choice({ identifier, literal_decl }); // (* Literals *)
const Token* preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor"); identifier = tf.tag(tf.seq({
tf.choice({ letter, tf["_"] }),
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
}), "identifier", true);
// (* Line Structure & Program *) comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true);
const Token* label = tf.tag(tf.seq({ identifier, tf[":"] }), "label"); sign = tf.choice("+-");
const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }); exponent_marker = tf.choice("eE");
const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }); exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }); decimal_lit = tf.tag(tf.seq({
const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }); tf.opt(sign),
const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) }); digit,
tf.rep(digit),
tf.opt(tf.choice("BSIL"))
}), "decimal_lit", true);
float_lit = tf.tag(tf.seq({
tf.opt(sign),
tf.choice({
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ digit, tf.rep(digit), exponent })
}),
tf.opt(tf.choice("FD"))
}), "float_lit", true);
hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true);
octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true);
binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true);
literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal");
literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast");
literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl");
// (* Operands *)
register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true);
addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true);
addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true);
addrm_idx = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_idx", true);
addrm_sca = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_sca", true);
addrm_dis = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_dis", true);
addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm");
operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand");
// (* Generalized Instructions *)
opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction");
// (* Added Preprocessor, Annotation *)
annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named");
annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg");
annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation");
preprocessor_val = tf.choice({ identifier, literal_decl });
preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor");
// (* Line Structure & Program *)
label = tf.tag(tf.seq({ identifier, tf[":"] }), "label");
line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
program = tf.seq({ tf.rep(line), tf.opt(line_last) });
}
} }
+61 -49
View File
@@ -8,11 +8,7 @@ namespace spider {
// Text Reader // // Text Reader //
TextReader::TextReader() : err(false), eof(false), bufferIndex(0) { TextReader::TextReader() : err(false), eof(false), bufferIndex(0) {}
// Prime the buffer with the first character
// so current() is immediately valid
fillBufferTo(0);
}
TextReader::~TextReader() {} TextReader::~TextReader() {}
@@ -37,14 +33,25 @@ namespace spider {
} }
bool TextReader::eat(const std::u32string& str) { bool TextReader::eat(const std::u32string& str) {
// case 0: no str
if(str.empty()) return true;
// prepare n chars
isize index_space = str.size() - 1;
fillBufferTo(index_space);
// fast reject
if(!hasBufferTo(index_space)) return false;
// compare now // compare now
isize index; for(isize i = 0; i <= index_space; i++) {
for(index = 0; index < str.size(); index++) { if(str[i] != buffer[bufferIndex + i]) {
if(str[index] != peekChar(index)) return false; return false;
}
} }
// success! // success!
nextChar(index); consumeChars(str.size());
return true; return true;
} }
@@ -68,37 +75,30 @@ namespace spider {
return char(ch); return char(ch);
} }
/**
* Returns the current character.
*/
optional<u32> TextReader::current() { optional<u32> TextReader::current() {
fillBufferTo(0);
if (bufferIndex < buffer.size()) { if (bufferIndex < buffer.size()) {
return buffer[bufferIndex]; return buffer[bufferIndex];
} }
return {}; return {};
} }
/**
* Reads the next character and advances the position tracker.
*/
optional<u32> TextReader::nextChar(isize n) { optional<u32> TextReader::nextChar(isize n) {
// Ensure the character we are moving TO exists // Ensure the character we are moving TO exists
if (fillBufferTo(n)) { fillBufferTo(n); // index = n will be accessible
// advance n characters // from [0, n] inclusive, equal to (n + 1) chars
while(n--) {
advance(buffer[bufferIndex]); // remember partial success
bufferIndex++; consumeChars(n); // n chars will be removed
}
return current(); // return current char
} // current char, index = 0
return {}; return current();
} }
/**
* Keeps the next n-th character (n = 0 is current).
*/
optional<u32> TextReader::peekChar(isize n) { optional<u32> TextReader::peekChar(isize n) {
if (fillBufferTo(n)) return buffer[bufferIndex + n]; fillBufferTo(n);
if (hasBufferTo(n)) return buffer[bufferIndex + n];
return {}; return {};
} }
@@ -110,12 +110,16 @@ namespace spider {
} }
} }
isize TextReader::push() { TextReader::State TextReader::push() {
return bufferIndex; return { .err = err, .eof = eof, .at = at, .errmsg = errmsg, .index = bufferIndex };
} }
void TextReader::pop(isize index) { void TextReader::pop(TextReader::State s) {
bufferIndex = std::min(index, bufferIndex); err = s.err;
eof = s.eof;
at = s.at;
errmsg = s.errmsg;
bufferIndex = s.index;
} }
/** /**
@@ -163,19 +167,27 @@ namespace spider {
return true; return true;
} }
/**
* Fills the buffer sequentially until it contains at least up
* to (bufferIndex + targetOffset).
*/
bool TextReader::fillBufferTo(isize targetOffset) { bool TextReader::fillBufferTo(isize targetOffset) {
isize targetSize = bufferIndex + targetOffset + 1; isize targetSize = bufferIndex + targetOffset;
while (buffer.size() < targetSize) { while (targetSize >= buffer.size()) {
if(readChar()) continue; if(readChar()) continue;
return false; return false;
} }
return true; return true;
} }
bool TextReader::hasBufferTo(isize index) {
return bufferIndex + index < buffer.size();
}
void TextReader::consumeChars(isize n) {
// advance up to specified char.
for(isize i = 0; i < n && hasBufferTo(i); i++) {
advance(buffer[bufferIndex]);
}
bufferIndex += n;
}
pos TextReader::getPosition() const { pos TextReader::getPosition() const {
return at; return at;
} }
@@ -212,24 +224,24 @@ namespace spider {
// String Reader // // String Reader //
StringTextReader::StringTextReader(std::string initialText) StringTextReader::StringTextReader(std::string initialText)
: buffer(std::move(initialText)), : txt_buffer(std::move(initialText)),
stringStream(std::make_unique<std::istringstream>(buffer)) { stringStream(std::make_unique<std::istringstream>(txt_buffer)) { }
}
std::istream& StringTextReader::getStream() { std::istream& StringTextReader::getStream() {
return *stringStream; return *stringStream;
} }
void StringTextReader::set(const std::string& newText) { void StringTextReader::set(const std::string& newText) {
buffer = newText; txt_buffer = newText;
stringStream = std::make_unique<std::istringstream>(buffer); stringStream = std::make_unique<std::istringstream>(txt_buffer);
}
void StringTextReader::append(const std::string& extraText) { buffer.clear();
std::streampos pos = stringStream->tellg(); txt_buffer.clear();
buffer += extraText;
stringStream = std::make_unique<std::istringstream>(buffer); bufferIndex = 0;
stringStream->seekg(pos); err = false;
eof = false;
at = pos();
} }
} }
+39 -12
View File
@@ -34,11 +34,6 @@ namespace spider {
std::string errmsg; std::string errmsg;
struct stored_char {
u8 byte_count;
u32 value;
};
/** /**
* Buffer of extracted characters. * Buffer of extracted characters.
*/ */
@@ -52,6 +47,16 @@ namespace spider {
*/ */
isize bufferIndex; isize bufferIndex;
public:
struct State {
bool err;
bool eof;
pos at;
std::string errmsg;
isize index;
};
public: public:
TextReader(); TextReader();
@@ -94,13 +99,16 @@ namespace spider {
optional<u32> current(); optional<u32> current();
/** /**
* Reads the next n-th character. * Skips n number of characters and returns
* the current one in that position.
* n = 0 is a noop, since it's the current one. * n = 0 is a noop, since it's the current one.
*
* Will advance until the EOF is reached.
*/ */
optional<u32> nextChar(isize n = 1); optional<u32> nextChar(isize n = 1);
/** /**
* Keeps the next n-th character * Returns the n-th character following the current one.
* n = 0 is the current one. * n = 0 is the current one.
*/ */
optional<u32> peekChar(isize n = 1); optional<u32> peekChar(isize n = 1);
@@ -121,14 +129,14 @@ namespace spider {
* Inside a parser, this allows to roll * Inside a parser, this allows to roll
* back the index to a specific position. * back the index to a specific position.
*/ */
isize push(); State push();
/** /**
* Sets the current buffer index. * Sets the current buffer index.
* Inside a parser, rolls back to * Inside a parser, rolls back to
* a previous position. * a previous position.
*/ */
void pop(isize index); void pop(State s);
/** /**
* Returns true if the end of the stream has been reached. * Returns true if the end of the stream has been reached.
@@ -160,8 +168,29 @@ namespace spider {
virtual std::istream& getStream() = 0; virtual std::istream& getStream() = 0;
/**
* Fills the buffer sequentially until it the passed
* index can be safely accessed, relative to the current
* buffer position.
*
* Returns false if that index could not be reached.
* Partial success is possible, check buffer.size()!
*/
bool fillBufferTo(isize index); bool fillBufferTo(isize index);
/**
* Verifies that the index can be safely accessed,
* relative to the current buffer position.
*/
bool hasBufferTo(isize index);
/**
* Triggers the buffer to consume this number
* of characters from the buffer. This is a reverseable
* operation.
*/
void consumeChars(isize n);
}; };
/** /**
@@ -188,7 +217,7 @@ namespace spider {
class StringTextReader : public TextReader { class StringTextReader : public TextReader {
private: private:
std::string buffer; std::string txt_buffer;
std::unique_ptr<std::istringstream> stringStream; std::unique_ptr<std::istringstream> stringStream;
public: public:
@@ -199,8 +228,6 @@ namespace spider {
void set(const std::string& newText); void set(const std::string& newText);
void append(const std::string& extraText);
protected: protected:
std::istream& getStream() override; std::istream& getStream() override;
+10 -6
View File
@@ -9,9 +9,13 @@ namespace spider {
// ============================================================================ // ============================================================================
Token* TokenFactory::lit(std::string_view text) { Token* TokenFactory::lit(std::string_view text) {
auto it = lit_cache.find(std::string(text));
if(it != lit_cache.end()) return it->second;
auto p = std::make_unique<LitToken>(text); auto p = std::make_unique<LitToken>(text);
auto t = p.get(); auto t = p.get();
lit_cache.emplace(text, std::move(p)); arena.emplace_back(std::move(p));
lit_cache.emplace(text, t);
return t; return t;
} }
@@ -63,7 +67,7 @@ namespace spider {
return t; return t;
} }
Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten = false) { Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten) {
uptr<Token> p = std::make_unique<TagToken>(target, tagname, flatten); uptr<Token> p = std::make_unique<TagToken>(target, tagname, flatten);
auto t = p.get(); auto t = p.get();
arena.emplace_back(std::move(p)); arena.emplace_back(std::move(p));
@@ -111,12 +115,12 @@ namespace spider {
// SeqToken Implementation // SeqToken Implementation
// ============================================================================30520370 // ============================================================================30520370
SeqToken::SeqToken(const vector<const Token*>& tokens) : tokens(tokens) {} SeqToken::SeqToken(const vector<const Token*>& _tokens) : tokens(_tokens) {}
TokenResult SeqToken::test(TextReader& ctx) const { TokenResult SeqToken::test(TextReader& ctx) const {
// this is a common branch point // this is a common branch point
TokenResult r; TokenResult r;
isize i = ctx.push(); auto i = ctx.push();
// All matching steps within a sequence must pass consecutively. // All matching steps within a sequence must pass consecutively.
for (const auto& token_ref : tokens) { for (const auto& token_ref : tokens) {
@@ -143,7 +147,7 @@ namespace spider {
// OrToken Implementation // OrToken Implementation
// ============================================================================ // ============================================================================
OrToken::OrToken(const vector<const Token*>& tokens) : tokens(tokens) {} OrToken::OrToken(const vector<const Token*>& _tokens) : tokens(_tokens) {}
TokenResult OrToken::test(TextReader& ctx) const { TokenResult OrToken::test(TextReader& ctx) const {
// All matching steps within a sequence must pass consecutively. // All matching steps within a sequence must pass consecutively.
@@ -195,7 +199,7 @@ namespace spider {
TokenResult res = target->test(ctx); TokenResult res = target->test(ctx);
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching // Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups). // rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
if (!res.success || i == ctx.push()) { if (!res.success || i.index == ctx.push().index) {
ctx.pop(i); ctx.pop(i);
break; break;
} }
+5 -5
View File
@@ -15,15 +15,15 @@ namespace spider {
/** @brief Indicates if the token composition successfully matched the input boundary. */ /** @brief Indicates if the token composition successfully matched the input boundary. */
bool success = false; bool success = false;
optional<std::string_view> tag; optional<std::string_view> tag = {};
/** /**
* @brief Holds the deep-copied UTF-32 matching substring upon victory. * @brief Holds the deep-copied UTF-32 matching substring upon victory.
* @note Returns empty when success is false. * @note Returns empty when success is false.
*/ */
std::u32string match; std::u32string match = U"";
vector<TokenResult> child; vector<TokenResult> child = {};
}; };
@@ -61,7 +61,7 @@ namespace spider {
std::vector<std::unique_ptr<Token>> arena; std::vector<std::unique_ptr<Token>> arena;
// Deduplication caches // Deduplication caches
std::unordered_map<std::string, const Token*> lit_cache; std::unordered_map<std::string, Token*> lit_cache;
public: public:
@@ -238,7 +238,7 @@ namespace spider {
public: public:
TagToken(const Token* t, std::string_view tag, bool doflatten = false); TagToken(const Token* t, std::string_view tag, bool doflatten);
TokenResult test(TextReader& ctx) const override; TokenResult test(TextReader& ctx) const override;