text reader passes tests
This commit is contained in:
@@ -3,12 +3,140 @@
|
|||||||
#include <spider/compiler/common.hpp>
|
#include <spider/compiler/common.hpp>
|
||||||
#include <spider/compiler/text/utf8.hpp>
|
#include <spider/compiler/text/utf8.hpp>
|
||||||
|
|
||||||
namespace spider {
|
#include <spider/compiler/text/TextReader.hpp>
|
||||||
|
|
||||||
|
using namespace spider;
|
||||||
|
|
||||||
|
class TestRunner {
|
||||||
|
private:
|
||||||
|
int totalTests = 0;
|
||||||
|
int passedTests = 0;
|
||||||
|
|
||||||
|
public:
|
||||||
|
void assertCondition(bool condition, const std::string& testName) {
|
||||||
|
totalTests++;
|
||||||
|
if (condition) {
|
||||||
|
std::cout << " [PASS] " << testName << "\n";
|
||||||
|
passedTests++;
|
||||||
|
} else {
|
||||||
|
std::cout << " [FAIL] " << testName << "\n";
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void printSummary() const {
|
||||||
|
std::cout << "\n========================================\n";
|
||||||
|
std::cout << "Test Results: " << passedTests << "/" << totalTests << " passed.\n";
|
||||||
|
std::cout << "========================================\n";
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// TEST SUITES FOR StringTextReader
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
void test_basic_reading(TestRunner& runner) {
|
||||||
|
std::cout << "\n--- Running: Basic Reading Tests ---\n";
|
||||||
|
std::cout.flush();
|
||||||
|
|
||||||
|
StringTextReader reader("hello");
|
||||||
|
|
||||||
|
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'h', "Initial current character is 'h'");
|
||||||
|
runner.assertCondition(reader.peekChar(1).has_value() && reader.peekChar(1).value() == 'e', "Peek +1 char is 'e'");
|
||||||
|
|
||||||
|
auto next = reader.nextChar(1);
|
||||||
|
runner.assertCondition(next.has_value() && next.value() == 'e', "Advance to next char gives 'e'");
|
||||||
|
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'e', "Current character is now 'e'");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void test_eat_operations(TestRunner& runner) {
|
||||||
|
std::cout << "\n--- Running: Eat Operations Tests ---\n";
|
||||||
|
std::cout.flush();
|
||||||
|
|
||||||
|
StringTextReader reader("constexpr int x = 42;");
|
||||||
|
|
||||||
|
runner.assertCondition(reader.eat("constexpr"), "Eat exact string match 'constexpr'");
|
||||||
|
runner.assertCondition(reader.eat(' '), "Eat single space character");
|
||||||
|
runner.assertCondition(reader.eat("int"), "Eat second string match 'int'");
|
||||||
|
|
||||||
|
runner.assertCondition(!reader.eat("float"), "Eat fails on mismatched string 'float'");
|
||||||
|
runner.assertCondition(reader.eat(' '), "Eat single space after failure");
|
||||||
|
runner.assertCondition(reader.eat('x'), "Eat character 'x'");
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_push_pop_rollback(TestRunner& runner) {
|
||||||
|
std::cout << "\n--- Running: Push/Pop Rollback Tests ---\n";
|
||||||
|
std::cout.flush();
|
||||||
|
|
||||||
|
StringTextReader reader("function_name()");
|
||||||
|
|
||||||
|
auto savedPos = reader.push();
|
||||||
|
runner.assertCondition(reader.eat("function_"), "Incomplete parse attempt");
|
||||||
|
|
||||||
|
// Rollback
|
||||||
|
reader.pop(savedPos);
|
||||||
|
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'f', "Rollback restores cursor to 'f'");
|
||||||
|
runner.assertCondition(reader.eat("function_name"), "Subsequent match succeeds after rollback");
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_commit(TestRunner& runner) {
|
||||||
|
std::cout << "\n--- Running: Buffer Commit Tests ---\n";
|
||||||
|
std::cout.flush();
|
||||||
|
|
||||||
|
StringTextReader reader("line1\nline2");
|
||||||
|
|
||||||
|
reader.eat("line1\n");
|
||||||
|
reader.commit(); // Discard historical rollback buffer
|
||||||
|
|
||||||
|
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'l', "Current char after commit is 'l'");
|
||||||
|
runner.assertCondition(reader.eat("line2"), "Reading continues normally after commit");
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_string_mutations(TestRunner& runner) {
|
||||||
|
std::cout << "\n--- Running: String Mutation Tests (set/append) ---\n";
|
||||||
|
std::cout.flush();
|
||||||
|
|
||||||
|
StringTextReader reader("foo");
|
||||||
|
runner.assertCondition(reader.eat("foo"), "Read initial text 'foo'");
|
||||||
|
|
||||||
|
reader.set("reset_text");
|
||||||
|
runner.assertCondition(reader.eat("reset_text"), "Read completely new text after set()");
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_eof_handling(TestRunner& runner) {
|
||||||
|
std::cout << "\n--- Running: EOF & Error State Tests ---\n";
|
||||||
|
std::cout.flush();
|
||||||
|
|
||||||
|
StringTextReader reader("a");
|
||||||
|
|
||||||
|
runner.assertCondition(static_cast<bool>(reader), "Reader is valid initially");
|
||||||
|
runner.assertCondition(!reader.isEOF(), "isEOF is false initially");
|
||||||
|
|
||||||
|
reader.eat('a');
|
||||||
|
std::cout << "Index: " << reader.push().index << std::endl;
|
||||||
|
std::cout << "Value: " << reader.current().value_or(0) << std::endl;
|
||||||
|
runner.assertCondition(!reader.current().has_value(), "current() returns empty optional at EOF");
|
||||||
|
runner.assertCondition(reader.isEOF(), "isEOF is true after consuming all input");
|
||||||
|
runner.assertCondition(!reader.hasError(), "hasError remains false on normal EOF");
|
||||||
|
}
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// MAIN DRIVER
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
int main() {
|
int main() {
|
||||||
|
std::cout << "========================================\n";
|
||||||
|
std::cout << " StringTextReader Unit Test Suite \n";
|
||||||
|
std::cout << "========================================\n";
|
||||||
|
|
||||||
|
TestRunner runner;
|
||||||
|
test_basic_reading(runner);
|
||||||
|
test_eat_operations(runner);
|
||||||
|
test_push_pop_rollback(runner);
|
||||||
|
test_commit(runner);
|
||||||
|
test_string_mutations(runner);
|
||||||
|
test_eof_handling(runner);
|
||||||
|
runner.printSummary();
|
||||||
|
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ namespace spider::asm_ebnf {
|
|||||||
|
|
||||||
// Token Factory
|
// Token Factory
|
||||||
|
|
||||||
TokenFactory tf;
|
//TokenFactory tf;
|
||||||
|
|
||||||
// Char Functions
|
// Char Functions
|
||||||
|
|
||||||
@@ -28,48 +28,120 @@ namespace spider::asm_ebnf {
|
|||||||
return ch != u32('"');
|
return ch != u32('"');
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const Token* letter;
|
||||||
|
const Token* digit;
|
||||||
|
const Token* alpha_num_char;
|
||||||
|
|
||||||
|
const Token* hex_digit;
|
||||||
|
const Token* octal_digit;
|
||||||
|
const Token* binary_digit;
|
||||||
|
|
||||||
|
const Token* ws_char;
|
||||||
|
const Token* ws_optional;
|
||||||
|
const Token* whitespace;
|
||||||
|
const Token* newline;
|
||||||
|
const Token* utf8_char;
|
||||||
|
|
||||||
|
const Token* char_escape;
|
||||||
|
const Token* char_content;
|
||||||
|
const Token* char_lit;
|
||||||
|
|
||||||
|
const Token* string_char;
|
||||||
|
const Token* string_lit;
|
||||||
|
|
||||||
|
const Token* identifier;
|
||||||
|
const Token* comment;
|
||||||
|
|
||||||
|
const Token* sign;
|
||||||
|
const Token* exponent_marker;
|
||||||
|
const Token* exponent;
|
||||||
|
|
||||||
|
const Token* decimal_lit;
|
||||||
|
const Token* float_lit;
|
||||||
|
|
||||||
|
const Token* hex_lit;
|
||||||
|
const Token* octal_lit;
|
||||||
|
const Token* binary_lit;
|
||||||
|
|
||||||
|
const Token* literal;
|
||||||
|
const Token* literal_cast;
|
||||||
|
const Token* literal_decl;
|
||||||
|
|
||||||
|
const Token* register_tok;
|
||||||
|
|
||||||
|
const Token* addrm_ind;
|
||||||
|
const Token* addrm_ptr;
|
||||||
|
const Token* addrm_idx;
|
||||||
|
const Token* addrm_sca;
|
||||||
|
const Token* addrm_dis;
|
||||||
|
|
||||||
|
const Token* addr_modes;
|
||||||
|
const Token* operand;
|
||||||
|
|
||||||
|
const Token* opcode;
|
||||||
|
const Token* operand_list;
|
||||||
|
const Token* instruction;
|
||||||
|
|
||||||
|
const Token* annotation_named;
|
||||||
|
const Token* annotation_arg;
|
||||||
|
const Token* annotation_args;
|
||||||
|
const Token* annotation_pars;
|
||||||
|
const Token* annotation;
|
||||||
|
|
||||||
|
const Token* preprocessor_val;
|
||||||
|
const Token* preprocessor;
|
||||||
|
|
||||||
|
const Token* label;
|
||||||
|
const Token* line_label;
|
||||||
|
const Token* line_annotation;
|
||||||
|
const Token* line_content;
|
||||||
|
const Token* line;
|
||||||
|
const Token* line_last;
|
||||||
|
const Token* program;
|
||||||
|
|
||||||
|
void initTokens(TokenFactory& tf) {
|
||||||
// (* Characters & Basic Predicates *)
|
// (* Characters & Basic Predicates *)
|
||||||
const Token* letter = tf.fn(isUTF8Alpha);
|
letter = tf.fn(isUTF8Alpha);
|
||||||
const Token* digit = tf.choice("0123456789");
|
digit = tf.choice("0123456789");
|
||||||
const Token* alpha_num_char = tf.choice({ letter, digit });
|
alpha_num_char = tf.choice({ letter, digit });
|
||||||
|
|
||||||
const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef");
|
hex_digit = tf.choice("0123456789ABCDEFabcdef");
|
||||||
const Token* octal_digit = tf.choice("01234567");
|
octal_digit = tf.choice("01234567");
|
||||||
const Token* binary_digit = tf.choice("01");
|
binary_digit = tf.choice("01");
|
||||||
|
|
||||||
const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf);
|
ws_char = tf.fn(isWhithespaceCharNotCrLf);
|
||||||
const Token* ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true);
|
ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true);
|
||||||
const Token* whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true);
|
whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true);
|
||||||
const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
|
newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
|
||||||
const Token* utf8_char = tf.fn(isUTF8CharNotCrLf);
|
utf8_char = tf.fn(isUTF8CharNotCrLf);
|
||||||
|
|
||||||
const Token* char_escape = tf.seq({ tf["\\"], utf8_char });
|
char_escape = tf.seq({ tf["\\"], utf8_char });
|
||||||
const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
|
char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
|
||||||
const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
|
char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
|
||||||
|
|
||||||
const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
|
string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
|
||||||
const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
|
string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
|
||||||
|
|
||||||
// (* Literals *)
|
// (* Literals *)
|
||||||
const Token* identifier = tf.tag(tf.seq({
|
identifier = tf.tag(tf.seq({
|
||||||
tf.choice({ letter, tf["_"] }),
|
tf.choice({ letter, tf["_"] }),
|
||||||
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
|
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
|
||||||
}), "identifier", true);
|
}), "identifier", true);
|
||||||
|
|
||||||
const Token* comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true);
|
comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true);
|
||||||
|
|
||||||
const Token* sign = tf.choice("+-");
|
sign = tf.choice("+-");
|
||||||
const Token* exponent_marker = tf.choice("eE");
|
exponent_marker = tf.choice("eE");
|
||||||
const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
|
exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
|
||||||
|
|
||||||
const Token* decimal_lit = tf.tag(tf.seq({
|
decimal_lit = tf.tag(tf.seq({
|
||||||
tf.opt(sign),
|
tf.opt(sign),
|
||||||
digit,
|
digit,
|
||||||
tf.rep(digit),
|
tf.rep(digit),
|
||||||
tf.opt(tf.choice("BSIL"))
|
tf.opt(tf.choice("BSIL"))
|
||||||
}), "decimal_lit", true);
|
}), "decimal_lit", true);
|
||||||
|
|
||||||
const Token* float_lit = tf.tag(tf.seq({
|
float_lit = tf.tag(tf.seq({
|
||||||
tf.opt(sign),
|
tf.opt(sign),
|
||||||
tf.choice({
|
tf.choice({
|
||||||
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
||||||
@@ -79,66 +151,67 @@ namespace spider::asm_ebnf {
|
|||||||
tf.opt(tf.choice("FD"))
|
tf.opt(tf.choice("FD"))
|
||||||
}), "float_lit", true);
|
}), "float_lit", true);
|
||||||
|
|
||||||
const Token* hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true);
|
hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true);
|
||||||
const Token* octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true);
|
octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true);
|
||||||
const Token* binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true);
|
binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true);
|
||||||
|
|
||||||
const Token* literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal");
|
literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal");
|
||||||
const Token* literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast");
|
literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast");
|
||||||
const Token* literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl");
|
literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl");
|
||||||
|
|
||||||
// (* Operands *)
|
// (* Operands *)
|
||||||
const Token* register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true);
|
register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true);
|
||||||
|
|
||||||
const Token* addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true);
|
addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true);
|
||||||
const Token* addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true);
|
addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true);
|
||||||
|
|
||||||
const Token* addrm_idx = tf.tag(tf.seq({
|
addrm_idx = tf.tag(tf.seq({
|
||||||
tf["["], ws_optional, register_tok, ws_optional,
|
tf["["], ws_optional, register_tok, ws_optional,
|
||||||
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||||
}), "addrm_idx", true);
|
}), "addrm_idx", true);
|
||||||
|
|
||||||
const Token* addrm_sca = tf.tag(tf.seq({
|
addrm_sca = tf.tag(tf.seq({
|
||||||
tf["["], ws_optional, register_tok, ws_optional,
|
tf["["], ws_optional, register_tok, ws_optional,
|
||||||
tf["+"], ws_optional, register_tok, ws_optional,
|
tf["+"], ws_optional, register_tok, ws_optional,
|
||||||
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
|
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||||
}), "addrm_sca", true);
|
}), "addrm_sca", true);
|
||||||
|
|
||||||
const Token* addrm_dis = tf.tag(tf.seq({
|
addrm_dis = tf.tag(tf.seq({
|
||||||
tf["["], ws_optional, register_tok, ws_optional,
|
tf["["], ws_optional, register_tok, ws_optional,
|
||||||
tf["+"], ws_optional, register_tok, ws_optional,
|
tf["+"], ws_optional, register_tok, ws_optional,
|
||||||
tf["*"], ws_optional, literal_decl, ws_optional,
|
tf["*"], ws_optional, literal_decl, ws_optional,
|
||||||
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||||
}), "addrm_dis", true);
|
}), "addrm_dis", true);
|
||||||
|
|
||||||
const Token* addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm");
|
addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm");
|
||||||
const Token* operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand");
|
operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand");
|
||||||
|
|
||||||
// (* Generalized Instructions *)
|
// (* Generalized Instructions *)
|
||||||
|
|
||||||
const Token* opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
|
opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
|
||||||
const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
|
operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
|
||||||
const Token* instruction = tf.tag(tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction");
|
instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction");
|
||||||
|
|
||||||
// (* Added Preprocessor, Annotation *)
|
// (* Added Preprocessor, Annotation *)
|
||||||
|
|
||||||
const Token* annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named");
|
annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named");
|
||||||
const Token* annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg");
|
annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg");
|
||||||
const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
|
annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
|
||||||
const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
|
annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
|
||||||
const Token* annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation");
|
annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation");
|
||||||
|
|
||||||
const Token* preprocessor_val = tf.choice({ identifier, literal_decl });
|
preprocessor_val = tf.choice({ identifier, literal_decl });
|
||||||
const Token* preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor");
|
preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor");
|
||||||
|
|
||||||
// (* Line Structure & Program *)
|
// (* Line Structure & Program *)
|
||||||
|
|
||||||
const Token* label = tf.tag(tf.seq({ identifier, tf[":"] }), "label");
|
label = tf.tag(tf.seq({ identifier, tf[":"] }), "label");
|
||||||
const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
|
line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||||
const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
|
line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||||
const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
|
line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
|
||||||
const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
|
line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
|
||||||
const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
|
line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
|
||||||
const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) });
|
program = tf.seq({ tf.rep(line), tf.opt(line_last) });
|
||||||
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -8,11 +8,7 @@ namespace spider {
|
|||||||
|
|
||||||
// Text Reader //
|
// Text Reader //
|
||||||
|
|
||||||
TextReader::TextReader() : err(false), eof(false), bufferIndex(0) {
|
TextReader::TextReader() : err(false), eof(false), bufferIndex(0) {}
|
||||||
// Prime the buffer with the first character
|
|
||||||
// so current() is immediately valid
|
|
||||||
fillBufferTo(0);
|
|
||||||
}
|
|
||||||
|
|
||||||
TextReader::~TextReader() {}
|
TextReader::~TextReader() {}
|
||||||
|
|
||||||
@@ -37,14 +33,25 @@ namespace spider {
|
|||||||
}
|
}
|
||||||
|
|
||||||
bool TextReader::eat(const std::u32string& str) {
|
bool TextReader::eat(const std::u32string& str) {
|
||||||
|
// case 0: no str
|
||||||
|
if(str.empty()) return true;
|
||||||
|
|
||||||
|
// prepare n chars
|
||||||
|
isize index_space = str.size() - 1;
|
||||||
|
fillBufferTo(index_space);
|
||||||
|
|
||||||
|
// fast reject
|
||||||
|
if(!hasBufferTo(index_space)) return false;
|
||||||
|
|
||||||
// compare now
|
// compare now
|
||||||
isize index;
|
for(isize i = 0; i <= index_space; i++) {
|
||||||
for(index = 0; index < str.size(); index++) {
|
if(str[i] != buffer[bufferIndex + i]) {
|
||||||
if(str[index] != peekChar(index)) return false;
|
return false;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// success!
|
// success!
|
||||||
nextChar(index);
|
consumeChars(str.size());
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -68,37 +75,30 @@ namespace spider {
|
|||||||
return char(ch);
|
return char(ch);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Returns the current character.
|
|
||||||
*/
|
|
||||||
optional<u32> TextReader::current() {
|
optional<u32> TextReader::current() {
|
||||||
|
fillBufferTo(0);
|
||||||
if (bufferIndex < buffer.size()) {
|
if (bufferIndex < buffer.size()) {
|
||||||
return buffer[bufferIndex];
|
return buffer[bufferIndex];
|
||||||
}
|
}
|
||||||
return {};
|
return {};
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Reads the next character and advances the position tracker.
|
|
||||||
*/
|
|
||||||
optional<u32> TextReader::nextChar(isize n) {
|
optional<u32> TextReader::nextChar(isize n) {
|
||||||
// Ensure the character we are moving TO exists
|
// Ensure the character we are moving TO exists
|
||||||
if (fillBufferTo(n)) {
|
fillBufferTo(n); // index = n will be accessible
|
||||||
// advance n characters
|
// from [0, n] inclusive, equal to (n + 1) chars
|
||||||
while(n--) {
|
|
||||||
advance(buffer[bufferIndex]);
|
// remember partial success
|
||||||
bufferIndex++;
|
consumeChars(n); // n chars will be removed
|
||||||
}
|
|
||||||
|
// return current char
|
||||||
|
// current char, index = 0
|
||||||
return current();
|
return current();
|
||||||
}
|
}
|
||||||
return {};
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Keeps the next n-th character (n = 0 is current).
|
|
||||||
*/
|
|
||||||
optional<u32> TextReader::peekChar(isize n) {
|
optional<u32> TextReader::peekChar(isize n) {
|
||||||
if (fillBufferTo(n)) return buffer[bufferIndex + n];
|
fillBufferTo(n);
|
||||||
|
if (hasBufferTo(n)) return buffer[bufferIndex + n];
|
||||||
return {};
|
return {};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -110,12 +110,16 @@ namespace spider {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
isize TextReader::push() {
|
TextReader::State TextReader::push() {
|
||||||
return bufferIndex;
|
return { .err = err, .eof = eof, .at = at, .errmsg = errmsg, .index = bufferIndex };
|
||||||
}
|
}
|
||||||
|
|
||||||
void TextReader::pop(isize index) {
|
void TextReader::pop(TextReader::State s) {
|
||||||
bufferIndex = std::min(index, bufferIndex);
|
err = s.err;
|
||||||
|
eof = s.eof;
|
||||||
|
at = s.at;
|
||||||
|
errmsg = s.errmsg;
|
||||||
|
bufferIndex = s.index;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -163,19 +167,27 @@ namespace spider {
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Fills the buffer sequentially until it contains at least up
|
|
||||||
* to (bufferIndex + targetOffset).
|
|
||||||
*/
|
|
||||||
bool TextReader::fillBufferTo(isize targetOffset) {
|
bool TextReader::fillBufferTo(isize targetOffset) {
|
||||||
isize targetSize = bufferIndex + targetOffset + 1;
|
isize targetSize = bufferIndex + targetOffset;
|
||||||
while (buffer.size() < targetSize) {
|
while (targetSize >= buffer.size()) {
|
||||||
if(readChar()) continue;
|
if(readChar()) continue;
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
bool TextReader::hasBufferTo(isize index) {
|
||||||
|
return bufferIndex + index < buffer.size();
|
||||||
|
}
|
||||||
|
|
||||||
|
void TextReader::consumeChars(isize n) {
|
||||||
|
// advance up to specified char.
|
||||||
|
for(isize i = 0; i < n && hasBufferTo(i); i++) {
|
||||||
|
advance(buffer[bufferIndex]);
|
||||||
|
}
|
||||||
|
bufferIndex += n;
|
||||||
|
}
|
||||||
|
|
||||||
pos TextReader::getPosition() const {
|
pos TextReader::getPosition() const {
|
||||||
return at;
|
return at;
|
||||||
}
|
}
|
||||||
@@ -212,24 +224,24 @@ namespace spider {
|
|||||||
// String Reader //
|
// String Reader //
|
||||||
|
|
||||||
StringTextReader::StringTextReader(std::string initialText)
|
StringTextReader::StringTextReader(std::string initialText)
|
||||||
: buffer(std::move(initialText)),
|
: txt_buffer(std::move(initialText)),
|
||||||
stringStream(std::make_unique<std::istringstream>(buffer)) {
|
stringStream(std::make_unique<std::istringstream>(txt_buffer)) { }
|
||||||
}
|
|
||||||
|
|
||||||
std::istream& StringTextReader::getStream() {
|
std::istream& StringTextReader::getStream() {
|
||||||
return *stringStream;
|
return *stringStream;
|
||||||
}
|
}
|
||||||
|
|
||||||
void StringTextReader::set(const std::string& newText) {
|
void StringTextReader::set(const std::string& newText) {
|
||||||
buffer = newText;
|
txt_buffer = newText;
|
||||||
stringStream = std::make_unique<std::istringstream>(buffer);
|
stringStream = std::make_unique<std::istringstream>(txt_buffer);
|
||||||
}
|
|
||||||
|
|
||||||
void StringTextReader::append(const std::string& extraText) {
|
buffer.clear();
|
||||||
std::streampos pos = stringStream->tellg();
|
txt_buffer.clear();
|
||||||
buffer += extraText;
|
|
||||||
stringStream = std::make_unique<std::istringstream>(buffer);
|
bufferIndex = 0;
|
||||||
stringStream->seekg(pos);
|
err = false;
|
||||||
|
eof = false;
|
||||||
|
at = pos();
|
||||||
}
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -34,11 +34,6 @@ namespace spider {
|
|||||||
|
|
||||||
std::string errmsg;
|
std::string errmsg;
|
||||||
|
|
||||||
struct stored_char {
|
|
||||||
u8 byte_count;
|
|
||||||
u32 value;
|
|
||||||
};
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Buffer of extracted characters.
|
* Buffer of extracted characters.
|
||||||
*/
|
*/
|
||||||
@@ -52,6 +47,16 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
isize bufferIndex;
|
isize bufferIndex;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
struct State {
|
||||||
|
bool err;
|
||||||
|
bool eof;
|
||||||
|
pos at;
|
||||||
|
std::string errmsg;
|
||||||
|
isize index;
|
||||||
|
};
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
|
||||||
TextReader();
|
TextReader();
|
||||||
@@ -94,13 +99,16 @@ namespace spider {
|
|||||||
optional<u32> current();
|
optional<u32> current();
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Reads the next n-th character.
|
* Skips n number of characters and returns
|
||||||
|
* the current one in that position.
|
||||||
* n = 0 is a noop, since it's the current one.
|
* n = 0 is a noop, since it's the current one.
|
||||||
|
*
|
||||||
|
* Will advance until the EOF is reached.
|
||||||
*/
|
*/
|
||||||
optional<u32> nextChar(isize n = 1);
|
optional<u32> nextChar(isize n = 1);
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Keeps the next n-th character
|
* Returns the n-th character following the current one.
|
||||||
* n = 0 is the current one.
|
* n = 0 is the current one.
|
||||||
*/
|
*/
|
||||||
optional<u32> peekChar(isize n = 1);
|
optional<u32> peekChar(isize n = 1);
|
||||||
@@ -121,14 +129,14 @@ namespace spider {
|
|||||||
* Inside a parser, this allows to roll
|
* Inside a parser, this allows to roll
|
||||||
* back the index to a specific position.
|
* back the index to a specific position.
|
||||||
*/
|
*/
|
||||||
isize push();
|
State push();
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Sets the current buffer index.
|
* Sets the current buffer index.
|
||||||
* Inside a parser, rolls back to
|
* Inside a parser, rolls back to
|
||||||
* a previous position.
|
* a previous position.
|
||||||
*/
|
*/
|
||||||
void pop(isize index);
|
void pop(State s);
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Returns true if the end of the stream has been reached.
|
* Returns true if the end of the stream has been reached.
|
||||||
@@ -160,8 +168,29 @@ namespace spider {
|
|||||||
|
|
||||||
virtual std::istream& getStream() = 0;
|
virtual std::istream& getStream() = 0;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Fills the buffer sequentially until it the passed
|
||||||
|
* index can be safely accessed, relative to the current
|
||||||
|
* buffer position.
|
||||||
|
*
|
||||||
|
* Returns false if that index could not be reached.
|
||||||
|
* Partial success is possible, check buffer.size()!
|
||||||
|
*/
|
||||||
bool fillBufferTo(isize index);
|
bool fillBufferTo(isize index);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Verifies that the index can be safely accessed,
|
||||||
|
* relative to the current buffer position.
|
||||||
|
*/
|
||||||
|
bool hasBufferTo(isize index);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Triggers the buffer to consume this number
|
||||||
|
* of characters from the buffer. This is a reverseable
|
||||||
|
* operation.
|
||||||
|
*/
|
||||||
|
void consumeChars(isize n);
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -188,7 +217,7 @@ namespace spider {
|
|||||||
class StringTextReader : public TextReader {
|
class StringTextReader : public TextReader {
|
||||||
private:
|
private:
|
||||||
|
|
||||||
std::string buffer;
|
std::string txt_buffer;
|
||||||
std::unique_ptr<std::istringstream> stringStream;
|
std::unique_ptr<std::istringstream> stringStream;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
@@ -199,8 +228,6 @@ namespace spider {
|
|||||||
|
|
||||||
void set(const std::string& newText);
|
void set(const std::string& newText);
|
||||||
|
|
||||||
void append(const std::string& extraText);
|
|
||||||
|
|
||||||
protected:
|
protected:
|
||||||
|
|
||||||
std::istream& getStream() override;
|
std::istream& getStream() override;
|
||||||
|
|||||||
@@ -9,9 +9,13 @@ namespace spider {
|
|||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
Token* TokenFactory::lit(std::string_view text) {
|
Token* TokenFactory::lit(std::string_view text) {
|
||||||
|
auto it = lit_cache.find(std::string(text));
|
||||||
|
if(it != lit_cache.end()) return it->second;
|
||||||
|
|
||||||
auto p = std::make_unique<LitToken>(text);
|
auto p = std::make_unique<LitToken>(text);
|
||||||
auto t = p.get();
|
auto t = p.get();
|
||||||
lit_cache.emplace(text, std::move(p));
|
arena.emplace_back(std::move(p));
|
||||||
|
lit_cache.emplace(text, t);
|
||||||
return t;
|
return t;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -63,7 +67,7 @@ namespace spider {
|
|||||||
return t;
|
return t;
|
||||||
}
|
}
|
||||||
|
|
||||||
Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten = false) {
|
Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten) {
|
||||||
uptr<Token> p = std::make_unique<TagToken>(target, tagname, flatten);
|
uptr<Token> p = std::make_unique<TagToken>(target, tagname, flatten);
|
||||||
auto t = p.get();
|
auto t = p.get();
|
||||||
arena.emplace_back(std::move(p));
|
arena.emplace_back(std::move(p));
|
||||||
@@ -111,12 +115,12 @@ namespace spider {
|
|||||||
// SeqToken Implementation
|
// SeqToken Implementation
|
||||||
// ============================================================================30520370
|
// ============================================================================30520370
|
||||||
|
|
||||||
SeqToken::SeqToken(const vector<const Token*>& tokens) : tokens(tokens) {}
|
SeqToken::SeqToken(const vector<const Token*>& _tokens) : tokens(_tokens) {}
|
||||||
|
|
||||||
TokenResult SeqToken::test(TextReader& ctx) const {
|
TokenResult SeqToken::test(TextReader& ctx) const {
|
||||||
// this is a common branch point
|
// this is a common branch point
|
||||||
TokenResult r;
|
TokenResult r;
|
||||||
isize i = ctx.push();
|
auto i = ctx.push();
|
||||||
|
|
||||||
// All matching steps within a sequence must pass consecutively.
|
// All matching steps within a sequence must pass consecutively.
|
||||||
for (const auto& token_ref : tokens) {
|
for (const auto& token_ref : tokens) {
|
||||||
@@ -143,7 +147,7 @@ namespace spider {
|
|||||||
// OrToken Implementation
|
// OrToken Implementation
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
OrToken::OrToken(const vector<const Token*>& tokens) : tokens(tokens) {}
|
OrToken::OrToken(const vector<const Token*>& _tokens) : tokens(_tokens) {}
|
||||||
|
|
||||||
TokenResult OrToken::test(TextReader& ctx) const {
|
TokenResult OrToken::test(TextReader& ctx) const {
|
||||||
// All matching steps within a sequence must pass consecutively.
|
// All matching steps within a sequence must pass consecutively.
|
||||||
@@ -195,7 +199,7 @@ namespace spider {
|
|||||||
TokenResult res = target->test(ctx);
|
TokenResult res = target->test(ctx);
|
||||||
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
||||||
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
||||||
if (!res.success || i == ctx.push()) {
|
if (!res.success || i.index == ctx.push().index) {
|
||||||
ctx.pop(i);
|
ctx.pop(i);
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -15,15 +15,15 @@ namespace spider {
|
|||||||
/** @brief Indicates if the token composition successfully matched the input boundary. */
|
/** @brief Indicates if the token composition successfully matched the input boundary. */
|
||||||
bool success = false;
|
bool success = false;
|
||||||
|
|
||||||
optional<std::string_view> tag;
|
optional<std::string_view> tag = {};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
|
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
|
||||||
* @note Returns empty when success is false.
|
* @note Returns empty when success is false.
|
||||||
*/
|
*/
|
||||||
std::u32string match;
|
std::u32string match = U"";
|
||||||
|
|
||||||
vector<TokenResult> child;
|
vector<TokenResult> child = {};
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -61,7 +61,7 @@ namespace spider {
|
|||||||
std::vector<std::unique_ptr<Token>> arena;
|
std::vector<std::unique_ptr<Token>> arena;
|
||||||
|
|
||||||
// Deduplication caches
|
// Deduplication caches
|
||||||
std::unordered_map<std::string, const Token*> lit_cache;
|
std::unordered_map<std::string, Token*> lit_cache;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
|
||||||
@@ -238,7 +238,7 @@ namespace spider {
|
|||||||
|
|
||||||
public:
|
public:
|
||||||
|
|
||||||
TagToken(const Token* t, std::string_view tag, bool doflatten = false);
|
TagToken(const Token* t, std::string_view tag, bool doflatten);
|
||||||
|
|
||||||
TokenResult test(TextReader& ctx) const override;
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user