time to test
This commit is contained in:
@@ -2,12 +2,18 @@
|
|||||||
|
|
||||||
namespace spider::asm_ebnf {
|
namespace spider::asm_ebnf {
|
||||||
|
|
||||||
|
// Token Factory
|
||||||
|
|
||||||
|
static TokenFactory tf;
|
||||||
|
|
||||||
|
// Char Functions
|
||||||
|
|
||||||
bool isUTF8Alpha(u32 ch) {
|
bool isUTF8Alpha(u32 ch) {
|
||||||
return false;
|
return (u32('a') <= ch && ch <= u32('z')) || (u32('A') <= ch && ch <= u32('Z'));
|
||||||
}
|
}
|
||||||
|
|
||||||
bool isWhithespaceCharNotCrLf(u32 ch) {
|
bool isWhithespaceCharNotCrLf(u32 ch) {
|
||||||
return false;
|
return ch == u32(' ');
|
||||||
}
|
}
|
||||||
|
|
||||||
bool isUTF8CharNotCrLf(u32 ch) {
|
bool isUTF8CharNotCrLf(u32 ch) {
|
||||||
@@ -22,37 +28,117 @@ namespace spider::asm_ebnf {
|
|||||||
return ch != u32('"');
|
return ch != u32('"');
|
||||||
}
|
}
|
||||||
|
|
||||||
LitToken numbers[] = {
|
// (* Characters & Basic Predicates *)
|
||||||
"0",
|
static const Token* letter = tf.fn(isUTF8Alpha);
|
||||||
"1","2","3",
|
static const Token* digit = tf.choice("0123456789");
|
||||||
"4","5","6",
|
static const Token* alpha_num_char = tf.choice({ letter, digit });
|
||||||
"7","8","9",
|
|
||||||
};
|
|
||||||
|
|
||||||
LitToken hex_digits[][2] = {
|
static const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef");
|
||||||
{"A", "a"},
|
static const Token* octal_digit = tf.choice("01234567");
|
||||||
{"B", "b"},
|
static const Token* binary_digit = tf.choice("01");
|
||||||
{"C", "c"},
|
|
||||||
{"D", "d"},
|
|
||||||
{"E", "e"},
|
|
||||||
{"F", "f"},
|
|
||||||
};
|
|
||||||
|
|
||||||
LitToken new_line[] = { "\r\n", "\r", "\n" };
|
static const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf);
|
||||||
|
static const Token* ws_optional = tf.rep(ws_char);
|
||||||
|
static const Token* whitespace = tf.seq({ ws_char, tf.rep(ws_char) });
|
||||||
|
static const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
|
||||||
|
static const Token* utf8_char = tf.fn(isUTF8CharNotCrLf);
|
||||||
|
|
||||||
LitToken symbols[] = {
|
static const Token* char_escape = tf.seq({ tf["\\"], utf8_char });
|
||||||
"\\", "\'", "\"",
|
static const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
|
||||||
"_" , ";" , "(" ,
|
static const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
|
||||||
"#",
|
|
||||||
"$" , "." , "+" , "-", ",", ")", "@", ":",
|
|
||||||
};
|
|
||||||
|
|
||||||
LitToken lit_letter[] = {
|
static const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
|
||||||
"x", "c", "b"
|
static const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
|
||||||
};
|
|
||||||
|
|
||||||
LitToken type_letter[] = {
|
// (* Literals *)
|
||||||
"B", "S", "I", "L", "F", "D"
|
static const Token* identifier = tf.seq({
|
||||||
};
|
tf.choice({ letter, tf["_"] }),
|
||||||
|
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
|
||||||
|
});
|
||||||
|
|
||||||
|
static const Token* comment = tf.seq({ tf[";"], tf.rep(utf8_char) });
|
||||||
|
|
||||||
|
static const Token* sign = tf.choice("+-");
|
||||||
|
static const Token* exponent_marker = tf.choice("eE");
|
||||||
|
static const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
|
||||||
|
|
||||||
|
static const Token* decimal_lit = tf.seq({
|
||||||
|
tf.opt(sign),
|
||||||
|
digit,
|
||||||
|
tf.rep(digit),
|
||||||
|
tf.opt(tf.choice("BSIL"))
|
||||||
|
});
|
||||||
|
|
||||||
|
static const Token* float_lit = tf.seq({
|
||||||
|
tf.opt(sign),
|
||||||
|
tf.choice({
|
||||||
|
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
||||||
|
tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
||||||
|
tf.seq({ digit, tf.rep(digit), exponent })
|
||||||
|
}),
|
||||||
|
tf.opt(tf.choice("FD"))
|
||||||
|
});
|
||||||
|
|
||||||
|
static const Token* hex_lit = tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) });
|
||||||
|
static const Token* octal_lit = tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) });
|
||||||
|
static const Token* binary_lit = tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) });
|
||||||
|
|
||||||
|
static const Token* literal = tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit });
|
||||||
|
static const Token* literal_cast = tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] });
|
||||||
|
static const Token* literal_decl = tf.choice({ literal, literal_cast });
|
||||||
|
|
||||||
|
// (* Operands *)
|
||||||
|
static const Token* register_tok = tf.seq({ tf["R"], alpha_num_char });
|
||||||
|
|
||||||
|
static const Token* addrm_ind = tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] });
|
||||||
|
static const Token* addrm_ptr = tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] });
|
||||||
|
|
||||||
|
static const Token* addrm_idx = tf.seq({
|
||||||
|
tf["["], ws_optional, register_tok, ws_optional,
|
||||||
|
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||||
|
});
|
||||||
|
|
||||||
|
static const Token* addrm_sca = tf.seq({
|
||||||
|
tf["["], ws_optional, register_tok, ws_optional,
|
||||||
|
tf["+"], ws_optional, register_tok, ws_optional,
|
||||||
|
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||||
|
});
|
||||||
|
|
||||||
|
static const Token* addrm_dis = tf.seq({
|
||||||
|
tf["["], ws_optional, register_tok, ws_optional,
|
||||||
|
tf["+"], ws_optional, register_tok, ws_optional,
|
||||||
|
tf["*"], ws_optional, literal_decl, ws_optional,
|
||||||
|
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||||
|
});
|
||||||
|
|
||||||
|
static const Token* addr_modes = tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind });
|
||||||
|
static const Token* operand = tf.choice({ register_tok, identifier, literal_decl, addr_modes });
|
||||||
|
|
||||||
|
// (* Generalized Instructions *)
|
||||||
|
|
||||||
|
static const Token* opcode = tf.seq({ letter, tf.rep(alpha_num_char) });
|
||||||
|
static const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
|
||||||
|
static const Token* instruction = tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) });
|
||||||
|
|
||||||
|
// (* Added Preprocessor, Annotation *)
|
||||||
|
|
||||||
|
static const Token* annotation_named = tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl });
|
||||||
|
static const Token* annotation_arg = tf.choice({ annotation_named, literal_decl });
|
||||||
|
static const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
|
||||||
|
static const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
|
||||||
|
static const Token* annotation = tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) });
|
||||||
|
|
||||||
|
static const Token* preprocessor_val = tf.choice({ identifier, string_lit });
|
||||||
|
static const Token* preprocessor = tf.seq({ tf["#"], identifier, whitespace, preprocessor_val });
|
||||||
|
|
||||||
|
// (* Line Structure & Program *)
|
||||||
|
|
||||||
|
static const Token* label = tf.seq({ identifier, tf[":"] });
|
||||||
|
static const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||||
|
static const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||||
|
static const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
|
||||||
|
static const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
|
||||||
|
static const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
|
||||||
|
static const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) });
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -6,4 +6,6 @@ namespace spider::asm_ebnf {
|
|||||||
|
|
||||||
extern LitToken letter;
|
extern LitToken letter;
|
||||||
|
|
||||||
|
void createTokens();
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,25 +1,66 @@
|
|||||||
#include "Token.hpp"
|
#include "Token.hpp"
|
||||||
|
|
||||||
|
#include <span>
|
||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// Token Implementation
|
// Token Factory
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
SeqToken Token::operator&(const Token& tok) {
|
Token* TokenFactory::lit(std::string_view text) {
|
||||||
return SeqToken({ tok, *this });
|
auto p = std::make_unique<LitToken>(text);
|
||||||
|
auto t = p.get();
|
||||||
|
lit_cache.emplace(text, std::move(p));
|
||||||
|
return t;
|
||||||
}
|
}
|
||||||
|
|
||||||
OrToken Token::operator|(const Token& tok) {
|
Token* TokenFactory::operator[](std::string_view text) {
|
||||||
return OrToken({ tok, *this });
|
return lit(text);
|
||||||
}
|
}
|
||||||
|
|
||||||
OptToken Token::operator~() {
|
Token* TokenFactory::fn(FnTokenFn predicate) {
|
||||||
return OptToken(*this);
|
uptr<Token> p = std::make_unique<FnToken>(predicate);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
}
|
}
|
||||||
|
|
||||||
RepToken Token::operator*() {
|
Token* TokenFactory::seq(const vector<const Token*>& tokens) {
|
||||||
return RepToken(*this);
|
uptr<Token> p = std::make_unique<SeqToken>(tokens);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
|
}
|
||||||
|
|
||||||
|
Token* TokenFactory::choice(std::string_view opts) {
|
||||||
|
vector<const Token*> toks;
|
||||||
|
for(char c : opts) {
|
||||||
|
std::string s = std::string(1, c);
|
||||||
|
toks.push_back(lit(s));
|
||||||
|
}
|
||||||
|
return choice(toks);
|
||||||
|
}
|
||||||
|
|
||||||
|
Token* TokenFactory::choice(const vector<const Token*>& tokens) {
|
||||||
|
uptr<Token> p = std::make_unique<SeqToken>(tokens);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
|
}
|
||||||
|
|
||||||
|
Token* TokenFactory::opt(const Token* target) {
|
||||||
|
uptr<Token> p = std::make_unique<OptToken>(target);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
|
}
|
||||||
|
|
||||||
|
Token* TokenFactory::rep(const Token* target) {
|
||||||
|
uptr<Token> p = std::make_unique<RepToken>(target);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
}
|
}
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
@@ -30,57 +71,64 @@ namespace spider {
|
|||||||
if (!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
|
if (!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
|
||||||
}
|
}
|
||||||
|
|
||||||
LitToken::LitToken(const char* lit) : LitToken(std::string_view(lit)) {}
|
|
||||||
|
|
||||||
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
|
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
|
||||||
|
|
||||||
TokenResult LitToken::test(TextReader& ctx) const {
|
TokenResult LitToken::test(TextReader& ctx) const {
|
||||||
if(ctx.eat(literal)) return { true, literal };
|
if (ctx.eat(literal)) return { .success = true, .match = literal };
|
||||||
return { false, {} };
|
return { .success = false };
|
||||||
}
|
}
|
||||||
|
|
||||||
FnToken::FnToken(FnTokenFn chfn) : fn(chfn) {}
|
FnToken::FnToken(FnTokenFn chfn) : fn(chfn) {}
|
||||||
|
|
||||||
TokenResult FnToken::test(TextReader& ctx) const {
|
TokenResult FnToken::test(TextReader& ctx) const {
|
||||||
std::u32string acc;
|
TokenResult r;
|
||||||
fn()
|
|
||||||
return { false, {} };
|
for (;;) {
|
||||||
|
auto i = ctx.push();
|
||||||
|
auto c = ctx.current();
|
||||||
|
|
||||||
|
if (!c || !fn(*c)) {
|
||||||
|
ctx.pop(i);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
r.match += *c;
|
||||||
|
ctx.nextChar();
|
||||||
|
}
|
||||||
|
|
||||||
|
r.success = !r.match.empty();
|
||||||
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// SeqToken Implementation
|
// SeqToken Implementation
|
||||||
// ============================================================================
|
// ============================================================================30520370
|
||||||
|
|
||||||
SeqToken::SeqToken(const ilist<ref<const Token>>& list) : tokens(list) {}
|
SeqToken::SeqToken(const vector<const Token*>& tokens) : tokens(tokens) {}
|
||||||
|
|
||||||
TokenResult SeqToken::test(TextReader& ctx) const {
|
TokenResult SeqToken::test(TextReader& ctx) const {
|
||||||
// this is a common branch point
|
// this is a common branch point
|
||||||
std::u32string acc;
|
TokenResult r;
|
||||||
auto tri = ctx.push();
|
isize i = ctx.push();
|
||||||
|
|
||||||
// All matching steps within a sequence must pass consecutively.
|
// All matching steps within a sequence must pass consecutively.
|
||||||
for (const auto& token_ref : tokens) {
|
for (const auto& token_ref : tokens) {
|
||||||
TokenResult res = token_ref.get().test(ctx);
|
TokenResult res = token_ref->test(ctx);
|
||||||
|
|
||||||
if (!res.success) {
|
if (!res.success) {
|
||||||
// Strict ACID Transaction: Roll back context pointer entirely
|
// Strict ACID Transaction: Roll back context pointer entirely
|
||||||
// if any nested condition in the sequence fails.
|
// if any nested condition in the sequence fails.
|
||||||
ctx.pop(tri);
|
ctx.pop(i);
|
||||||
return { false, {} };
|
return { .success = false };
|
||||||
}
|
}
|
||||||
|
|
||||||
// Piecewise accumulation of individual matching sub-tokens
|
// Piecewise accumulation of individual matching sub-tokens
|
||||||
acc += res.match;
|
//r.match += res.match;
|
||||||
|
r.child.push_back(res);
|
||||||
}
|
}
|
||||||
|
|
||||||
return { true, acc };
|
r.success = true;
|
||||||
}
|
return r;
|
||||||
|
|
||||||
SeqToken SeqToken::operator&(const Token& tok) {
|
|
||||||
// Intrusive chaining optimization: Appends the next token directly into the existing
|
|
||||||
// registry vector instead of nesting structures, keeping the layout flattened.
|
|
||||||
tokens.push_back(std::cref(tok));
|
|
||||||
return *this;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -88,42 +136,33 @@ namespace spider {
|
|||||||
// OrToken Implementation
|
// OrToken Implementation
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
OrToken::OrToken(const ilist<ref<const Token>>& list) : tokens(list) {}
|
OrToken::OrToken(const vector<const Token*>& tokens) : tokens(tokens) {}
|
||||||
|
|
||||||
TokenResult OrToken::test(TextReader& ctx) const {
|
TokenResult OrToken::test(TextReader& ctx) const {
|
||||||
// this is a common branch point
|
|
||||||
auto tri = ctx.push();
|
|
||||||
|
|
||||||
// All matching steps within a sequence must pass consecutively.
|
// All matching steps within a sequence must pass consecutively.
|
||||||
|
auto i = ctx.push();
|
||||||
|
|
||||||
for (const auto& token_ref : tokens) {
|
for (const auto& token_ref : tokens) {
|
||||||
// Short-circuit branch: return immediately on first valid choice match
|
// Short-circuit branch: return immediately on first valid choice match
|
||||||
TokenResult res = token_ref.get().test(ctx);
|
TokenResult res = token_ref->test(ctx);
|
||||||
if (res.success) return res;
|
if (res.success) return res;
|
||||||
|
|
||||||
// Backtrack isolation: Reset the cursor position before testing the next alternative path
|
// Backtrack isolation: Reset the cursor position before testing the next alternative path
|
||||||
ctx.pop(tri);
|
ctx.pop(i);
|
||||||
}
|
}
|
||||||
|
|
||||||
return { false, {} };
|
return { .success = false };
|
||||||
}
|
}
|
||||||
|
|
||||||
OrToken OrToken::operator|(const Token& tok) {
|
|
||||||
// Intrusive grouping layout optimization:
|
|
||||||
// flattens alternative tokens at code evaluation time.
|
|
||||||
tokens.push_back(tok);
|
|
||||||
return *this;
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// OptToken Implementation
|
// OptToken Implementation
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
OptToken::OptToken(const Token& t) : target(t) {}
|
OptToken::OptToken(const Token* t) : target(t) {}
|
||||||
|
|
||||||
TokenResult OptToken::test(TextReader& ctx) const {
|
TokenResult OptToken::test(TextReader& ctx) const {
|
||||||
auto tri = ctx.push();
|
auto tri = ctx.push();
|
||||||
TokenResult res = target.test(ctx);
|
TokenResult res = target->test(ctx);
|
||||||
|
|
||||||
if (res.success) {
|
if (res.success) {
|
||||||
return res; // Option matched exactly 1 instance successfully
|
return res; // Option matched exactly 1 instance successfully
|
||||||
@@ -132,46 +171,58 @@ namespace spider {
|
|||||||
// Recovery path: If sub-rule fails, clean up the dirty state mutation
|
// Recovery path: If sub-rule fails, clean up the dirty state mutation
|
||||||
// and successfully return an empty match payload (0 instances).
|
// and successfully return an empty match payload (0 instances).
|
||||||
ctx.pop(tri);
|
ctx.pop(tri);
|
||||||
return { true, U"" };
|
return { .success = false };
|
||||||
}
|
}
|
||||||
|
|
||||||
OptToken OptToken::operator~() {
|
|
||||||
// Redundant layer trap protection: returning self
|
|
||||||
// prevents wrapping an Optional in an Optional
|
|
||||||
return *this;
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// RepToken Implementation
|
// RepToken Implementation
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
RepToken::RepToken(const Token& t) : target(t) {}
|
RepToken::RepToken(const Token* t) : target(t) {}
|
||||||
|
|
||||||
TokenResult RepToken::test(TextReader& ctx) const {
|
TokenResult RepToken::test(TextReader& ctx) const {
|
||||||
std::u32string acc;
|
TokenResult r;
|
||||||
auto tri = ctx.push();
|
|
||||||
|
|
||||||
for (;;) {
|
for (;;) {
|
||||||
TokenResult res = target.test(ctx);
|
auto i = ctx.push();
|
||||||
|
TokenResult res = target->test(ctx);
|
||||||
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
||||||
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
||||||
if(!res.success || tri == ctx.push()) {
|
if (!res.success || i == ctx.push()) {
|
||||||
ctx.pop(tri);
|
ctx.pop(i);
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
acc += res.match;
|
//r.match += res.match;
|
||||||
|
r.child.push_back(res);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Repetition rules (* token) always evaluate to successful
|
// Repetition rules (* token) always evaluate to successful
|
||||||
// completion state, even with 0 matches.
|
// completion state, even with 0 matches.
|
||||||
return { true, acc };
|
r.success = true;
|
||||||
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
RepToken RepToken::operator*() {
|
// Tagged Token
|
||||||
// Redundant layer trap protection: returning self prevents wrapping
|
|
||||||
// a Repetition rule inside a Repetition rule
|
TagToken::TagToken(const Token* t, std::string_view tag, bool doflatten)
|
||||||
return *this;
|
: target(t), tag_name(tag), flatten(doflatten) { }
|
||||||
|
|
||||||
|
TokenResult TagToken::test(TextReader& ctx) const {
|
||||||
|
auto r = target->test(ctx);
|
||||||
|
if (!r.success) return r;
|
||||||
|
|
||||||
|
// Set tag of this result
|
||||||
|
r.tag = tag_name;
|
||||||
|
|
||||||
|
if(flatten) {
|
||||||
|
r.match = U"";
|
||||||
|
for(auto e : r.child) {
|
||||||
|
r.match += e.match;
|
||||||
|
}
|
||||||
|
r.child.clear();
|
||||||
|
}
|
||||||
|
|
||||||
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -13,7 +13,9 @@ namespace spider {
|
|||||||
struct TokenResult {
|
struct TokenResult {
|
||||||
|
|
||||||
/** @brief Indicates if the token composition successfully matched the input boundary. */
|
/** @brief Indicates if the token composition successfully matched the input boundary. */
|
||||||
bool success;
|
bool success = false;
|
||||||
|
|
||||||
|
optional<std::string_view> tag;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
|
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
|
||||||
@@ -21,13 +23,14 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
std::u32string match;
|
std::u32string match;
|
||||||
|
|
||||||
|
vector<TokenResult> child;
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
// Forward declarations required by the abstract interface for operator returns.
|
class Token;
|
||||||
class SeqToken;
|
class TokenFactory;
|
||||||
class OrToken;
|
|
||||||
class OptToken;
|
using FnTokenFn = std::function<bool(u32)>;
|
||||||
class RepToken;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Pure virtual base class defining the EBNF combinator node contract.
|
* @brief Pure virtual base class defining the EBNF combinator node contract.
|
||||||
@@ -50,31 +53,46 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
virtual TokenResult test(TextReader& ctx) const = 0;
|
virtual TokenResult test(TextReader& ctx) const = 0;
|
||||||
|
|
||||||
|
};
|
||||||
|
|
||||||
|
class TokenFactory {
|
||||||
|
private:
|
||||||
|
// Arena owning all created tokens
|
||||||
|
std::vector<std::unique_ptr<Token>> arena;
|
||||||
|
|
||||||
|
// Deduplication caches
|
||||||
|
std::unordered_map<std::string, const Token*> lit_cache;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
|
||||||
/**
|
TokenFactory() = default;
|
||||||
* @brief Chains this token and another sequentially.
|
|
||||||
* @return A temporary structural bridge matching both tokens sequentially.
|
|
||||||
*/
|
|
||||||
virtual SeqToken operator&(const Token& tok);
|
|
||||||
|
|
||||||
/**
|
// Prevent copying to maintain valid internal pointers
|
||||||
* @brief Combines this token and another under alternation.
|
TokenFactory(const TokenFactory&) = delete;
|
||||||
* @return A structural bridge matching either this token or the fallback selection.
|
|
||||||
*/
|
|
||||||
virtual OrToken operator|(const Token& tok);
|
|
||||||
|
|
||||||
/**
|
TokenFactory& operator=(const TokenFactory&) = delete;
|
||||||
* @brief Wraps this node in an optional layout rule.
|
|
||||||
* @return A structure matching zero or one instances of this current node.
|
|
||||||
*/
|
|
||||||
virtual OptToken operator~();
|
|
||||||
|
|
||||||
/**
|
public:
|
||||||
* @brief Wraps this node in a repetitive loop match framework.
|
|
||||||
* @return A structure matching zero or more occurrences of this current node.
|
// --- Primitive Constructors ---
|
||||||
*/
|
|
||||||
virtual RepToken operator*();
|
Token* lit(std::string_view text);
|
||||||
|
|
||||||
|
Token* operator[](std::string_view text);
|
||||||
|
|
||||||
|
Token* fn(FnTokenFn predicate);
|
||||||
|
|
||||||
|
// --- Combinator Constructors ---
|
||||||
|
|
||||||
|
Token* seq(const vector<const Token*>& tokens);
|
||||||
|
|
||||||
|
Token* choice(std::string_view opts);
|
||||||
|
|
||||||
|
Token* choice(const vector<const Token*>& tokens);
|
||||||
|
|
||||||
|
Token* opt(const Token* target);
|
||||||
|
|
||||||
|
Token* rep(const Token* target);
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -93,8 +111,6 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
LitToken(std::string_view lit);
|
LitToken(std::string_view lit);
|
||||||
|
|
||||||
LitToken(const char* lit);
|
|
||||||
|
|
||||||
/** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */
|
/** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */
|
||||||
explicit LitToken(std::u32string lit);
|
explicit LitToken(std::u32string lit);
|
||||||
|
|
||||||
@@ -106,8 +122,6 @@ namespace spider {
|
|||||||
TokenResult test(TextReader& ctx) const override;
|
TokenResult test(TextReader& ctx) const override;
|
||||||
};
|
};
|
||||||
|
|
||||||
using FnTokenFn = std::function<bool(u32) >;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Function based token
|
* @brief Function based token
|
||||||
*/
|
*/
|
||||||
@@ -135,11 +149,10 @@ namespace spider {
|
|||||||
* @brief Internal contiguous layout registry storing lightweight, zero-overhead references.
|
* @brief Internal contiguous layout registry storing lightweight, zero-overhead references.
|
||||||
* @details Avoids heap allocation penalties by referencing static instances immutably.
|
* @details Avoids heap allocation penalties by referencing static instances immutably.
|
||||||
*/
|
*/
|
||||||
vector<ref<const Token>> tokens;
|
vector<const Token*> tokens;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/** @brief Constructs a sequence container directly out of an inline brace-enclosed listing. */
|
SeqToken(const vector<const Token*>& tokens);
|
||||||
SeqToken(const ilist<ref<const Token>>& list);
|
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/**
|
/**
|
||||||
@@ -149,8 +162,6 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
TokenResult test(TextReader& ctx) const override;
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
/** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */
|
|
||||||
SeqToken operator&(const Token& tok) override;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -159,11 +170,10 @@ namespace spider {
|
|||||||
class OrToken : public Token {
|
class OrToken : public Token {
|
||||||
private:
|
private:
|
||||||
/** @brief Ordered registry of possible alternate structural paths. */
|
/** @brief Ordered registry of possible alternate structural paths. */
|
||||||
vector<ref<const Token>> tokens;
|
vector<const Token*> tokens;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/** @brief Constructs an alternation choice layout from brace-enclosed tokens. */
|
OrToken(const vector<const Token*>& tokens);
|
||||||
OrToken(const ilist<ref<const Token>>& list);
|
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/**
|
/**
|
||||||
@@ -172,8 +182,6 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
TokenResult test(TextReader& ctx) const override;
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
/** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */
|
|
||||||
OrToken operator|(const Token& tok) override;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -182,11 +190,11 @@ namespace spider {
|
|||||||
class OptToken : public Token {
|
class OptToken : public Token {
|
||||||
private:
|
private:
|
||||||
/** @brief Read-only target node reference to test optional status against. */
|
/** @brief Read-only target node reference to test optional status against. */
|
||||||
const Token& target;
|
const Token* target;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/** @brief Binds the target node rule structural dependency layout wrapper. */
|
/** @brief Binds the target node rule structural dependency layout wrapper. */
|
||||||
explicit OptToken(const Token& t);
|
explicit OptToken(const Token* t);
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/**
|
/**
|
||||||
@@ -195,8 +203,6 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
TokenResult test(TextReader& ctx) const override;
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
/** @brief Stub override providing standard compliance with the base Token interface signature. */
|
|
||||||
OptToken operator~() override;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -205,11 +211,11 @@ namespace spider {
|
|||||||
class RepToken : public Token {
|
class RepToken : public Token {
|
||||||
private:
|
private:
|
||||||
/** @brief The base token node sequence layer evaluated in loops. */
|
/** @brief The base token node sequence layer evaluated in loops. */
|
||||||
const Token& target;
|
const Token* target;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/** @brief Binds the repeated structural blueprint node wrapper configuration. */
|
/** @brief Binds the repeated structural blueprint node wrapper configuration. */
|
||||||
explicit RepToken(const Token& t);
|
explicit RepToken(const Token* t);
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/**
|
/**
|
||||||
@@ -219,8 +225,21 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
TokenResult test(TextReader& ctx) const override;
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
/** @brief Stub override providing standard compliance with the base Token interface signature. */
|
};
|
||||||
RepToken operator*() override;
|
|
||||||
|
class TagToken : public Token {
|
||||||
|
private:
|
||||||
|
|
||||||
|
const Token* target;
|
||||||
|
std::string tag_name;
|
||||||
|
bool flatten;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
TagToken(const Token* t, std::string_view tag, bool doflatten = false);
|
||||||
|
|
||||||
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user