text reader passes tests
This commit is contained in:
@@ -3,12 +3,140 @@
|
||||
#include <spider/compiler/common.hpp>
|
||||
#include <spider/compiler/text/utf8.hpp>
|
||||
|
||||
namespace spider {
|
||||
#include <spider/compiler/text/TextReader.hpp>
|
||||
|
||||
using namespace spider;
|
||||
|
||||
class TestRunner {
|
||||
private:
|
||||
int totalTests = 0;
|
||||
int passedTests = 0;
|
||||
|
||||
public:
|
||||
void assertCondition(bool condition, const std::string& testName) {
|
||||
totalTests++;
|
||||
if (condition) {
|
||||
std::cout << " [PASS] " << testName << "\n";
|
||||
passedTests++;
|
||||
} else {
|
||||
std::cout << " [FAIL] " << testName << "\n";
|
||||
}
|
||||
}
|
||||
|
||||
void printSummary() const {
|
||||
std::cout << "\n========================================\n";
|
||||
std::cout << "Test Results: " << passedTests << "/" << totalTests << " passed.\n";
|
||||
std::cout << "========================================\n";
|
||||
}
|
||||
};
|
||||
|
||||
// ============================================================================
|
||||
// TEST SUITES FOR StringTextReader
|
||||
// ============================================================================
|
||||
|
||||
void test_basic_reading(TestRunner& runner) {
|
||||
std::cout << "\n--- Running: Basic Reading Tests ---\n";
|
||||
std::cout.flush();
|
||||
|
||||
StringTextReader reader("hello");
|
||||
|
||||
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'h', "Initial current character is 'h'");
|
||||
runner.assertCondition(reader.peekChar(1).has_value() && reader.peekChar(1).value() == 'e', "Peek +1 char is 'e'");
|
||||
|
||||
auto next = reader.nextChar(1);
|
||||
runner.assertCondition(next.has_value() && next.value() == 'e', "Advance to next char gives 'e'");
|
||||
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'e', "Current character is now 'e'");
|
||||
}
|
||||
|
||||
void test_eat_operations(TestRunner& runner) {
|
||||
std::cout << "\n--- Running: Eat Operations Tests ---\n";
|
||||
std::cout.flush();
|
||||
|
||||
StringTextReader reader("constexpr int x = 42;");
|
||||
|
||||
runner.assertCondition(reader.eat("constexpr"), "Eat exact string match 'constexpr'");
|
||||
runner.assertCondition(reader.eat(' '), "Eat single space character");
|
||||
runner.assertCondition(reader.eat("int"), "Eat second string match 'int'");
|
||||
|
||||
runner.assertCondition(!reader.eat("float"), "Eat fails on mismatched string 'float'");
|
||||
runner.assertCondition(reader.eat(' '), "Eat single space after failure");
|
||||
runner.assertCondition(reader.eat('x'), "Eat character 'x'");
|
||||
}
|
||||
|
||||
void test_push_pop_rollback(TestRunner& runner) {
|
||||
std::cout << "\n--- Running: Push/Pop Rollback Tests ---\n";
|
||||
std::cout.flush();
|
||||
|
||||
StringTextReader reader("function_name()");
|
||||
|
||||
auto savedPos = reader.push();
|
||||
runner.assertCondition(reader.eat("function_"), "Incomplete parse attempt");
|
||||
|
||||
// Rollback
|
||||
reader.pop(savedPos);
|
||||
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'f', "Rollback restores cursor to 'f'");
|
||||
runner.assertCondition(reader.eat("function_name"), "Subsequent match succeeds after rollback");
|
||||
}
|
||||
|
||||
void test_commit(TestRunner& runner) {
|
||||
std::cout << "\n--- Running: Buffer Commit Tests ---\n";
|
||||
std::cout.flush();
|
||||
|
||||
StringTextReader reader("line1\nline2");
|
||||
|
||||
reader.eat("line1\n");
|
||||
reader.commit(); // Discard historical rollback buffer
|
||||
|
||||
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'l', "Current char after commit is 'l'");
|
||||
runner.assertCondition(reader.eat("line2"), "Reading continues normally after commit");
|
||||
}
|
||||
|
||||
void test_string_mutations(TestRunner& runner) {
|
||||
std::cout << "\n--- Running: String Mutation Tests (set/append) ---\n";
|
||||
std::cout.flush();
|
||||
|
||||
StringTextReader reader("foo");
|
||||
runner.assertCondition(reader.eat("foo"), "Read initial text 'foo'");
|
||||
|
||||
reader.set("reset_text");
|
||||
runner.assertCondition(reader.eat("reset_text"), "Read completely new text after set()");
|
||||
}
|
||||
|
||||
void test_eof_handling(TestRunner& runner) {
|
||||
std::cout << "\n--- Running: EOF & Error State Tests ---\n";
|
||||
std::cout.flush();
|
||||
|
||||
StringTextReader reader("a");
|
||||
|
||||
runner.assertCondition(static_cast<bool>(reader), "Reader is valid initially");
|
||||
runner.assertCondition(!reader.isEOF(), "isEOF is false initially");
|
||||
|
||||
reader.eat('a');
|
||||
std::cout << "Index: " << reader.push().index << std::endl;
|
||||
std::cout << "Value: " << reader.current().value_or(0) << std::endl;
|
||||
runner.assertCondition(!reader.current().has_value(), "current() returns empty optional at EOF");
|
||||
runner.assertCondition(reader.isEOF(), "isEOF is true after consuming all input");
|
||||
runner.assertCondition(!reader.hasError(), "hasError remains false on normal EOF");
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// MAIN DRIVER
|
||||
// ============================================================================
|
||||
|
||||
int main() {
|
||||
std::cout << "========================================\n";
|
||||
std::cout << " StringTextReader Unit Test Suite \n";
|
||||
std::cout << "========================================\n";
|
||||
|
||||
TestRunner runner;
|
||||
test_basic_reading(runner);
|
||||
test_eat_operations(runner);
|
||||
test_push_pop_rollback(runner);
|
||||
test_commit(runner);
|
||||
test_string_mutations(runner);
|
||||
test_eof_handling(runner);
|
||||
runner.printSummary();
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
@@ -4,7 +4,7 @@ namespace spider::asm_ebnf {
|
||||
|
||||
// Token Factory
|
||||
|
||||
TokenFactory tf;
|
||||
//TokenFactory tf;
|
||||
|
||||
// Char Functions
|
||||
|
||||
@@ -28,117 +28,190 @@ namespace spider::asm_ebnf {
|
||||
return ch != u32('"');
|
||||
}
|
||||
|
||||
// (* Characters & Basic Predicates *)
|
||||
const Token* letter = tf.fn(isUTF8Alpha);
|
||||
const Token* digit = tf.choice("0123456789");
|
||||
const Token* alpha_num_char = tf.choice({ letter, digit });
|
||||
const Token* letter;
|
||||
const Token* digit;
|
||||
const Token* alpha_num_char;
|
||||
|
||||
const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef");
|
||||
const Token* octal_digit = tf.choice("01234567");
|
||||
const Token* binary_digit = tf.choice("01");
|
||||
const Token* hex_digit;
|
||||
const Token* octal_digit;
|
||||
const Token* binary_digit;
|
||||
|
||||
const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf);
|
||||
const Token* ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true);
|
||||
const Token* whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true);
|
||||
const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
|
||||
const Token* utf8_char = tf.fn(isUTF8CharNotCrLf);
|
||||
const Token* ws_char;
|
||||
const Token* ws_optional;
|
||||
const Token* whitespace;
|
||||
const Token* newline;
|
||||
const Token* utf8_char;
|
||||
|
||||
const Token* char_escape = tf.seq({ tf["\\"], utf8_char });
|
||||
const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
|
||||
const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
|
||||
const Token* char_escape;
|
||||
const Token* char_content;
|
||||
const Token* char_lit;
|
||||
|
||||
const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
|
||||
const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
|
||||
const Token* string_char;
|
||||
const Token* string_lit;
|
||||
|
||||
// (* Literals *)
|
||||
const Token* identifier = tf.tag(tf.seq({
|
||||
tf.choice({ letter, tf["_"] }),
|
||||
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
|
||||
}), "identifier", true);
|
||||
const Token* identifier;
|
||||
const Token* comment;
|
||||
|
||||
const Token* comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true);
|
||||
const Token* sign;
|
||||
const Token* exponent_marker;
|
||||
const Token* exponent;
|
||||
|
||||
const Token* sign = tf.choice("+-");
|
||||
const Token* exponent_marker = tf.choice("eE");
|
||||
const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
|
||||
const Token* decimal_lit;
|
||||
const Token* float_lit;
|
||||
|
||||
const Token* decimal_lit = tf.tag(tf.seq({
|
||||
tf.opt(sign),
|
||||
digit,
|
||||
tf.rep(digit),
|
||||
tf.opt(tf.choice("BSIL"))
|
||||
}), "decimal_lit", true);
|
||||
const Token* hex_lit;
|
||||
const Token* octal_lit;
|
||||
const Token* binary_lit;
|
||||
|
||||
const Token* float_lit = tf.tag(tf.seq({
|
||||
tf.opt(sign),
|
||||
tf.choice({
|
||||
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
||||
tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
||||
tf.seq({ digit, tf.rep(digit), exponent })
|
||||
}),
|
||||
tf.opt(tf.choice("FD"))
|
||||
}), "float_lit", true);
|
||||
const Token* literal;
|
||||
const Token* literal_cast;
|
||||
const Token* literal_decl;
|
||||
|
||||
const Token* hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true);
|
||||
const Token* octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true);
|
||||
const Token* binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true);
|
||||
const Token* register_tok;
|
||||
|
||||
const Token* literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal");
|
||||
const Token* literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast");
|
||||
const Token* literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl");
|
||||
const Token* addrm_ind;
|
||||
const Token* addrm_ptr;
|
||||
const Token* addrm_idx;
|
||||
const Token* addrm_sca;
|
||||
const Token* addrm_dis;
|
||||
|
||||
// (* Operands *)
|
||||
const Token* register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true);
|
||||
const Token* addr_modes;
|
||||
const Token* operand;
|
||||
|
||||
const Token* addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true);
|
||||
const Token* addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true);
|
||||
const Token* opcode;
|
||||
const Token* operand_list;
|
||||
const Token* instruction;
|
||||
|
||||
const Token* addrm_idx = tf.tag(tf.seq({
|
||||
tf["["], ws_optional, register_tok, ws_optional,
|
||||
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||
}), "addrm_idx", true);
|
||||
const Token* annotation_named;
|
||||
const Token* annotation_arg;
|
||||
const Token* annotation_args;
|
||||
const Token* annotation_pars;
|
||||
const Token* annotation;
|
||||
|
||||
const Token* addrm_sca = tf.tag(tf.seq({
|
||||
tf["["], ws_optional, register_tok, ws_optional,
|
||||
tf["+"], ws_optional, register_tok, ws_optional,
|
||||
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||
}), "addrm_sca", true);
|
||||
const Token* preprocessor_val;
|
||||
const Token* preprocessor;
|
||||
|
||||
const Token* addrm_dis = tf.tag(tf.seq({
|
||||
tf["["], ws_optional, register_tok, ws_optional,
|
||||
tf["+"], ws_optional, register_tok, ws_optional,
|
||||
tf["*"], ws_optional, literal_decl, ws_optional,
|
||||
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||
}), "addrm_dis", true);
|
||||
const Token* label;
|
||||
const Token* line_label;
|
||||
const Token* line_annotation;
|
||||
const Token* line_content;
|
||||
const Token* line;
|
||||
const Token* line_last;
|
||||
const Token* program;
|
||||
|
||||
const Token* addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm");
|
||||
const Token* operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand");
|
||||
void initTokens(TokenFactory& tf) {
|
||||
// (* Characters & Basic Predicates *)
|
||||
letter = tf.fn(isUTF8Alpha);
|
||||
digit = tf.choice("0123456789");
|
||||
alpha_num_char = tf.choice({ letter, digit });
|
||||
|
||||
// (* Generalized Instructions *)
|
||||
hex_digit = tf.choice("0123456789ABCDEFabcdef");
|
||||
octal_digit = tf.choice("01234567");
|
||||
binary_digit = tf.choice("01");
|
||||
|
||||
const Token* opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
|
||||
const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
|
||||
const Token* instruction = tf.tag(tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction");
|
||||
ws_char = tf.fn(isWhithespaceCharNotCrLf);
|
||||
ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true);
|
||||
whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true);
|
||||
newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
|
||||
utf8_char = tf.fn(isUTF8CharNotCrLf);
|
||||
|
||||
// (* Added Preprocessor, Annotation *)
|
||||
char_escape = tf.seq({ tf["\\"], utf8_char });
|
||||
char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
|
||||
char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
|
||||
|
||||
const Token* annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named");
|
||||
const Token* annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg");
|
||||
const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
|
||||
const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
|
||||
const Token* annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation");
|
||||
string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
|
||||
string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
|
||||
|
||||
const Token* preprocessor_val = tf.choice({ identifier, literal_decl });
|
||||
const Token* preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor");
|
||||
// (* Literals *)
|
||||
identifier = tf.tag(tf.seq({
|
||||
tf.choice({ letter, tf["_"] }),
|
||||
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
|
||||
}), "identifier", true);
|
||||
|
||||
// (* Line Structure & Program *)
|
||||
comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true);
|
||||
|
||||
const Token* label = tf.tag(tf.seq({ identifier, tf[":"] }), "label");
|
||||
const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||
const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||
const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
|
||||
const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
|
||||
const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
|
||||
const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) });
|
||||
sign = tf.choice("+-");
|
||||
exponent_marker = tf.choice("eE");
|
||||
exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
|
||||
|
||||
decimal_lit = tf.tag(tf.seq({
|
||||
tf.opt(sign),
|
||||
digit,
|
||||
tf.rep(digit),
|
||||
tf.opt(tf.choice("BSIL"))
|
||||
}), "decimal_lit", true);
|
||||
|
||||
float_lit = tf.tag(tf.seq({
|
||||
tf.opt(sign),
|
||||
tf.choice({
|
||||
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
||||
tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
||||
tf.seq({ digit, tf.rep(digit), exponent })
|
||||
}),
|
||||
tf.opt(tf.choice("FD"))
|
||||
}), "float_lit", true);
|
||||
|
||||
hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true);
|
||||
octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true);
|
||||
binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true);
|
||||
|
||||
literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal");
|
||||
literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast");
|
||||
literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl");
|
||||
|
||||
// (* Operands *)
|
||||
register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true);
|
||||
|
||||
addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true);
|
||||
addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true);
|
||||
|
||||
addrm_idx = tf.tag(tf.seq({
|
||||
tf["["], ws_optional, register_tok, ws_optional,
|
||||
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||
}), "addrm_idx", true);
|
||||
|
||||
addrm_sca = tf.tag(tf.seq({
|
||||
tf["["], ws_optional, register_tok, ws_optional,
|
||||
tf["+"], ws_optional, register_tok, ws_optional,
|
||||
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||
}), "addrm_sca", true);
|
||||
|
||||
addrm_dis = tf.tag(tf.seq({
|
||||
tf["["], ws_optional, register_tok, ws_optional,
|
||||
tf["+"], ws_optional, register_tok, ws_optional,
|
||||
tf["*"], ws_optional, literal_decl, ws_optional,
|
||||
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||
}), "addrm_dis", true);
|
||||
|
||||
addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm");
|
||||
operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand");
|
||||
|
||||
// (* Generalized Instructions *)
|
||||
|
||||
opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
|
||||
operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
|
||||
instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction");
|
||||
|
||||
// (* Added Preprocessor, Annotation *)
|
||||
|
||||
annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named");
|
||||
annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg");
|
||||
annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
|
||||
annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
|
||||
annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation");
|
||||
|
||||
preprocessor_val = tf.choice({ identifier, literal_decl });
|
||||
preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor");
|
||||
|
||||
// (* Line Structure & Program *)
|
||||
|
||||
label = tf.tag(tf.seq({ identifier, tf[":"] }), "label");
|
||||
line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||
line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||
line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
|
||||
line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
|
||||
line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
|
||||
program = tf.seq({ tf.rep(line), tf.opt(line_last) });
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -8,11 +8,7 @@ namespace spider {
|
||||
|
||||
// Text Reader //
|
||||
|
||||
TextReader::TextReader() : err(false), eof(false), bufferIndex(0) {
|
||||
// Prime the buffer with the first character
|
||||
// so current() is immediately valid
|
||||
fillBufferTo(0);
|
||||
}
|
||||
TextReader::TextReader() : err(false), eof(false), bufferIndex(0) {}
|
||||
|
||||
TextReader::~TextReader() {}
|
||||
|
||||
@@ -37,14 +33,25 @@ namespace spider {
|
||||
}
|
||||
|
||||
bool TextReader::eat(const std::u32string& str) {
|
||||
// case 0: no str
|
||||
if(str.empty()) return true;
|
||||
|
||||
// prepare n chars
|
||||
isize index_space = str.size() - 1;
|
||||
fillBufferTo(index_space);
|
||||
|
||||
// fast reject
|
||||
if(!hasBufferTo(index_space)) return false;
|
||||
|
||||
// compare now
|
||||
isize index;
|
||||
for(index = 0; index < str.size(); index++) {
|
||||
if(str[index] != peekChar(index)) return false;
|
||||
for(isize i = 0; i <= index_space; i++) {
|
||||
if(str[i] != buffer[bufferIndex + i]) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// success!
|
||||
nextChar(index);
|
||||
consumeChars(str.size());
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -68,37 +75,30 @@ namespace spider {
|
||||
return char(ch);
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns the current character.
|
||||
*/
|
||||
optional<u32> TextReader::current() {
|
||||
fillBufferTo(0);
|
||||
if (bufferIndex < buffer.size()) {
|
||||
return buffer[bufferIndex];
|
||||
}
|
||||
return {};
|
||||
}
|
||||
|
||||
/**
|
||||
* Reads the next character and advances the position tracker.
|
||||
*/
|
||||
optional<u32> TextReader::nextChar(isize n) {
|
||||
// Ensure the character we are moving TO exists
|
||||
if (fillBufferTo(n)) {
|
||||
// advance n characters
|
||||
while(n--) {
|
||||
advance(buffer[bufferIndex]);
|
||||
bufferIndex++;
|
||||
}
|
||||
return current();
|
||||
}
|
||||
return {};
|
||||
fillBufferTo(n); // index = n will be accessible
|
||||
// from [0, n] inclusive, equal to (n + 1) chars
|
||||
|
||||
// remember partial success
|
||||
consumeChars(n); // n chars will be removed
|
||||
|
||||
// return current char
|
||||
// current char, index = 0
|
||||
return current();
|
||||
}
|
||||
|
||||
/**
|
||||
* Keeps the next n-th character (n = 0 is current).
|
||||
*/
|
||||
optional<u32> TextReader::peekChar(isize n) {
|
||||
if (fillBufferTo(n)) return buffer[bufferIndex + n];
|
||||
fillBufferTo(n);
|
||||
if (hasBufferTo(n)) return buffer[bufferIndex + n];
|
||||
return {};
|
||||
}
|
||||
|
||||
@@ -110,12 +110,16 @@ namespace spider {
|
||||
}
|
||||
}
|
||||
|
||||
isize TextReader::push() {
|
||||
return bufferIndex;
|
||||
TextReader::State TextReader::push() {
|
||||
return { .err = err, .eof = eof, .at = at, .errmsg = errmsg, .index = bufferIndex };
|
||||
}
|
||||
|
||||
void TextReader::pop(isize index) {
|
||||
bufferIndex = std::min(index, bufferIndex);
|
||||
void TextReader::pop(TextReader::State s) {
|
||||
err = s.err;
|
||||
eof = s.eof;
|
||||
at = s.at;
|
||||
errmsg = s.errmsg;
|
||||
bufferIndex = s.index;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -163,19 +167,27 @@ namespace spider {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fills the buffer sequentially until it contains at least up
|
||||
* to (bufferIndex + targetOffset).
|
||||
*/
|
||||
bool TextReader::fillBufferTo(isize targetOffset) {
|
||||
isize targetSize = bufferIndex + targetOffset + 1;
|
||||
while (buffer.size() < targetSize) {
|
||||
isize targetSize = bufferIndex + targetOffset;
|
||||
while (targetSize >= buffer.size()) {
|
||||
if(readChar()) continue;
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool TextReader::hasBufferTo(isize index) {
|
||||
return bufferIndex + index < buffer.size();
|
||||
}
|
||||
|
||||
void TextReader::consumeChars(isize n) {
|
||||
// advance up to specified char.
|
||||
for(isize i = 0; i < n && hasBufferTo(i); i++) {
|
||||
advance(buffer[bufferIndex]);
|
||||
}
|
||||
bufferIndex += n;
|
||||
}
|
||||
|
||||
pos TextReader::getPosition() const {
|
||||
return at;
|
||||
}
|
||||
@@ -212,24 +224,24 @@ namespace spider {
|
||||
// String Reader //
|
||||
|
||||
StringTextReader::StringTextReader(std::string initialText)
|
||||
: buffer(std::move(initialText)),
|
||||
stringStream(std::make_unique<std::istringstream>(buffer)) {
|
||||
}
|
||||
: txt_buffer(std::move(initialText)),
|
||||
stringStream(std::make_unique<std::istringstream>(txt_buffer)) { }
|
||||
|
||||
std::istream& StringTextReader::getStream() {
|
||||
return *stringStream;
|
||||
}
|
||||
|
||||
void StringTextReader::set(const std::string& newText) {
|
||||
buffer = newText;
|
||||
stringStream = std::make_unique<std::istringstream>(buffer);
|
||||
}
|
||||
txt_buffer = newText;
|
||||
stringStream = std::make_unique<std::istringstream>(txt_buffer);
|
||||
|
||||
void StringTextReader::append(const std::string& extraText) {
|
||||
std::streampos pos = stringStream->tellg();
|
||||
buffer += extraText;
|
||||
stringStream = std::make_unique<std::istringstream>(buffer);
|
||||
stringStream->seekg(pos);
|
||||
buffer.clear();
|
||||
txt_buffer.clear();
|
||||
|
||||
bufferIndex = 0;
|
||||
err = false;
|
||||
eof = false;
|
||||
at = pos();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -34,11 +34,6 @@ namespace spider {
|
||||
|
||||
std::string errmsg;
|
||||
|
||||
struct stored_char {
|
||||
u8 byte_count;
|
||||
u32 value;
|
||||
};
|
||||
|
||||
/**
|
||||
* Buffer of extracted characters.
|
||||
*/
|
||||
@@ -52,6 +47,16 @@ namespace spider {
|
||||
*/
|
||||
isize bufferIndex;
|
||||
|
||||
public:
|
||||
|
||||
struct State {
|
||||
bool err;
|
||||
bool eof;
|
||||
pos at;
|
||||
std::string errmsg;
|
||||
isize index;
|
||||
};
|
||||
|
||||
public:
|
||||
|
||||
TextReader();
|
||||
@@ -94,13 +99,16 @@ namespace spider {
|
||||
optional<u32> current();
|
||||
|
||||
/**
|
||||
* Reads the next n-th character.
|
||||
* Skips n number of characters and returns
|
||||
* the current one in that position.
|
||||
* n = 0 is a noop, since it's the current one.
|
||||
*
|
||||
* Will advance until the EOF is reached.
|
||||
*/
|
||||
optional<u32> nextChar(isize n = 1);
|
||||
|
||||
/**
|
||||
* Keeps the next n-th character
|
||||
* Returns the n-th character following the current one.
|
||||
* n = 0 is the current one.
|
||||
*/
|
||||
optional<u32> peekChar(isize n = 1);
|
||||
@@ -121,14 +129,14 @@ namespace spider {
|
||||
* Inside a parser, this allows to roll
|
||||
* back the index to a specific position.
|
||||
*/
|
||||
isize push();
|
||||
State push();
|
||||
|
||||
/**
|
||||
* Sets the current buffer index.
|
||||
* Inside a parser, rolls back to
|
||||
* a previous position.
|
||||
*/
|
||||
void pop(isize index);
|
||||
void pop(State s);
|
||||
|
||||
/**
|
||||
* Returns true if the end of the stream has been reached.
|
||||
@@ -160,8 +168,29 @@ namespace spider {
|
||||
|
||||
virtual std::istream& getStream() = 0;
|
||||
|
||||
/**
|
||||
* Fills the buffer sequentially until it the passed
|
||||
* index can be safely accessed, relative to the current
|
||||
* buffer position.
|
||||
*
|
||||
* Returns false if that index could not be reached.
|
||||
* Partial success is possible, check buffer.size()!
|
||||
*/
|
||||
bool fillBufferTo(isize index);
|
||||
|
||||
/**
|
||||
* Verifies that the index can be safely accessed,
|
||||
* relative to the current buffer position.
|
||||
*/
|
||||
bool hasBufferTo(isize index);
|
||||
|
||||
/**
|
||||
* Triggers the buffer to consume this number
|
||||
* of characters from the buffer. This is a reverseable
|
||||
* operation.
|
||||
*/
|
||||
void consumeChars(isize n);
|
||||
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -188,7 +217,7 @@ namespace spider {
|
||||
class StringTextReader : public TextReader {
|
||||
private:
|
||||
|
||||
std::string buffer;
|
||||
std::string txt_buffer;
|
||||
std::unique_ptr<std::istringstream> stringStream;
|
||||
|
||||
public:
|
||||
@@ -199,8 +228,6 @@ namespace spider {
|
||||
|
||||
void set(const std::string& newText);
|
||||
|
||||
void append(const std::string& extraText);
|
||||
|
||||
protected:
|
||||
|
||||
std::istream& getStream() override;
|
||||
|
||||
@@ -9,9 +9,13 @@ namespace spider {
|
||||
// ============================================================================
|
||||
|
||||
Token* TokenFactory::lit(std::string_view text) {
|
||||
auto it = lit_cache.find(std::string(text));
|
||||
if(it != lit_cache.end()) return it->second;
|
||||
|
||||
auto p = std::make_unique<LitToken>(text);
|
||||
auto t = p.get();
|
||||
lit_cache.emplace(text, std::move(p));
|
||||
arena.emplace_back(std::move(p));
|
||||
lit_cache.emplace(text, t);
|
||||
return t;
|
||||
}
|
||||
|
||||
@@ -63,7 +67,7 @@ namespace spider {
|
||||
return t;
|
||||
}
|
||||
|
||||
Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten = false) {
|
||||
Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten) {
|
||||
uptr<Token> p = std::make_unique<TagToken>(target, tagname, flatten);
|
||||
auto t = p.get();
|
||||
arena.emplace_back(std::move(p));
|
||||
@@ -111,12 +115,12 @@ namespace spider {
|
||||
// SeqToken Implementation
|
||||
// ============================================================================30520370
|
||||
|
||||
SeqToken::SeqToken(const vector<const Token*>& tokens) : tokens(tokens) {}
|
||||
SeqToken::SeqToken(const vector<const Token*>& _tokens) : tokens(_tokens) {}
|
||||
|
||||
TokenResult SeqToken::test(TextReader& ctx) const {
|
||||
// this is a common branch point
|
||||
TokenResult r;
|
||||
isize i = ctx.push();
|
||||
auto i = ctx.push();
|
||||
|
||||
// All matching steps within a sequence must pass consecutively.
|
||||
for (const auto& token_ref : tokens) {
|
||||
@@ -143,7 +147,7 @@ namespace spider {
|
||||
// OrToken Implementation
|
||||
// ============================================================================
|
||||
|
||||
OrToken::OrToken(const vector<const Token*>& tokens) : tokens(tokens) {}
|
||||
OrToken::OrToken(const vector<const Token*>& _tokens) : tokens(_tokens) {}
|
||||
|
||||
TokenResult OrToken::test(TextReader& ctx) const {
|
||||
// All matching steps within a sequence must pass consecutively.
|
||||
@@ -195,7 +199,7 @@ namespace spider {
|
||||
TokenResult res = target->test(ctx);
|
||||
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
||||
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
||||
if (!res.success || i == ctx.push()) {
|
||||
if (!res.success || i.index == ctx.push().index) {
|
||||
ctx.pop(i);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -15,15 +15,15 @@ namespace spider {
|
||||
/** @brief Indicates if the token composition successfully matched the input boundary. */
|
||||
bool success = false;
|
||||
|
||||
optional<std::string_view> tag;
|
||||
optional<std::string_view> tag = {};
|
||||
|
||||
/**
|
||||
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
|
||||
* @note Returns empty when success is false.
|
||||
*/
|
||||
std::u32string match;
|
||||
std::u32string match = U"";
|
||||
|
||||
vector<TokenResult> child;
|
||||
vector<TokenResult> child = {};
|
||||
|
||||
};
|
||||
|
||||
@@ -61,7 +61,7 @@ namespace spider {
|
||||
std::vector<std::unique_ptr<Token>> arena;
|
||||
|
||||
// Deduplication caches
|
||||
std::unordered_map<std::string, const Token*> lit_cache;
|
||||
std::unordered_map<std::string, Token*> lit_cache;
|
||||
|
||||
public:
|
||||
|
||||
@@ -238,7 +238,7 @@ namespace spider {
|
||||
|
||||
public:
|
||||
|
||||
TagToken(const Token* t, std::string_view tag, bool doflatten = false);
|
||||
TagToken(const Token* t, std::string_view tag, bool doflatten);
|
||||
|
||||
TokenResult test(TextReader& ctx) const override;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user