time to test

This commit is contained in:
2026-08-02 19:04:52 -06:00
parent a228c0c59c
commit 3779848355
5 changed files with 310 additions and 152 deletions
+115 -29
View File
@@ -2,12 +2,18 @@
namespace spider::asm_ebnf {
// Token Factory
static TokenFactory tf;
// Char Functions
bool isUTF8Alpha(u32 ch) {
return false;
return (u32('a') <= ch && ch <= u32('z')) || (u32('A') <= ch && ch <= u32('Z'));
}
bool isWhithespaceCharNotCrLf(u32 ch) {
return false;
return ch == u32(' ');
}
bool isUTF8CharNotCrLf(u32 ch) {
@@ -22,37 +28,117 @@ namespace spider::asm_ebnf {
return ch != u32('"');
}
LitToken numbers[] = {
"0",
"1","2","3",
"4","5","6",
"7","8","9",
};
// (* Characters & Basic Predicates *)
static const Token* letter = tf.fn(isUTF8Alpha);
static const Token* digit = tf.choice("0123456789");
static const Token* alpha_num_char = tf.choice({ letter, digit });
LitToken hex_digits[][2] = {
{"A", "a"},
{"B", "b"},
{"C", "c"},
{"D", "d"},
{"E", "e"},
{"F", "f"},
};
static const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef");
static const Token* octal_digit = tf.choice("01234567");
static const Token* binary_digit = tf.choice("01");
LitToken new_line[] = { "\r\n", "\r", "\n" };
static const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf);
static const Token* ws_optional = tf.rep(ws_char);
static const Token* whitespace = tf.seq({ ws_char, tf.rep(ws_char) });
static const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
static const Token* utf8_char = tf.fn(isUTF8CharNotCrLf);
LitToken symbols[] = {
"\\", "\'", "\"",
"_" , ";" , "(" ,
"#",
"$" , "." , "+" , "-", ",", ")", "@", ":",
};
static const Token* char_escape = tf.seq({ tf["\\"], utf8_char });
static const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
static const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
LitToken lit_letter[] = {
"x", "c", "b"
};
static const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
static const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
LitToken type_letter[] = {
"B", "S", "I", "L", "F", "D"
};
// (* Literals *)
static const Token* identifier = tf.seq({
tf.choice({ letter, tf["_"] }),
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
});
static const Token* comment = tf.seq({ tf[";"], tf.rep(utf8_char) });
static const Token* sign = tf.choice("+-");
static const Token* exponent_marker = tf.choice("eE");
static const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
static const Token* decimal_lit = tf.seq({
tf.opt(sign),
digit,
tf.rep(digit),
tf.opt(tf.choice("BSIL"))
});
static const Token* float_lit = tf.seq({
tf.opt(sign),
tf.choice({
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ digit, tf.rep(digit), exponent })
}),
tf.opt(tf.choice("FD"))
});
static const Token* hex_lit = tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) });
static const Token* octal_lit = tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) });
static const Token* binary_lit = tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) });
static const Token* literal = tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit });
static const Token* literal_cast = tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] });
static const Token* literal_decl = tf.choice({ literal, literal_cast });
// (* Operands *)
static const Token* register_tok = tf.seq({ tf["R"], alpha_num_char });
static const Token* addrm_ind = tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] });
static const Token* addrm_ptr = tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] });
static const Token* addrm_idx = tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
});
static const Token* addrm_sca = tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
});
static const Token* addrm_dis = tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
});
static const Token* addr_modes = tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind });
static const Token* operand = tf.choice({ register_tok, identifier, literal_decl, addr_modes });
// (* Generalized Instructions *)
static const Token* opcode = tf.seq({ letter, tf.rep(alpha_num_char) });
static const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
static const Token* instruction = tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) });
// (* Added Preprocessor, Annotation *)
static const Token* annotation_named = tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl });
static const Token* annotation_arg = tf.choice({ annotation_named, literal_decl });
static const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
static const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
static const Token* annotation = tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) });
static const Token* preprocessor_val = tf.choice({ identifier, string_lit });
static const Token* preprocessor = tf.seq({ tf["#"], identifier, whitespace, preprocessor_val });
// (* Line Structure & Program *)
static const Token* label = tf.seq({ identifier, tf[":"] });
static const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
static const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
static const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
static const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
static const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
static const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) });
}
@@ -6,4 +6,6 @@ namespace spider::asm_ebnf {
extern LitToken letter;
void createTokens();
}
+120 -69
View File
@@ -1,25 +1,66 @@
#include "Token.hpp"
#include <span>
namespace spider {
// ============================================================================
// Token Implementation
// Token Factory
// ============================================================================
SeqToken Token::operator&(const Token& tok) {
return SeqToken({ tok, *this });
Token* TokenFactory::lit(std::string_view text) {
auto p = std::make_unique<LitToken>(text);
auto t = p.get();
lit_cache.emplace(text, std::move(p));
return t;
}
OrToken Token::operator|(const Token& tok) {
return OrToken({ tok, *this });
Token* TokenFactory::operator[](std::string_view text) {
return lit(text);
}
OptToken Token::operator~() {
return OptToken(*this);
Token* TokenFactory::fn(FnTokenFn predicate) {
uptr<Token> p = std::make_unique<FnToken>(predicate);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
RepToken Token::operator*() {
return RepToken(*this);
Token* TokenFactory::seq(const vector<const Token*>& tokens) {
uptr<Token> p = std::make_unique<SeqToken>(tokens);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
Token* TokenFactory::choice(std::string_view opts) {
vector<const Token*> toks;
for(char c : opts) {
std::string s = std::string(1, c);
toks.push_back(lit(s));
}
return choice(toks);
}
Token* TokenFactory::choice(const vector<const Token*>& tokens) {
uptr<Token> p = std::make_unique<SeqToken>(tokens);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
Token* TokenFactory::opt(const Token* target) {
uptr<Token> p = std::make_unique<OptToken>(target);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
Token* TokenFactory::rep(const Token* target) {
uptr<Token> p = std::make_unique<RepToken>(target);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
// ============================================================================
@@ -30,57 +71,64 @@ namespace spider {
if (!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
}
LitToken::LitToken(const char* lit) : LitToken(std::string_view(lit)) {}
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
TokenResult LitToken::test(TextReader& ctx) const {
if(ctx.eat(literal)) return { true, literal };
return { false, {} };
if (ctx.eat(literal)) return { .success = true, .match = literal };
return { .success = false };
}
FnToken::FnToken(FnTokenFn chfn) : fn(chfn) {}
TokenResult FnToken::test(TextReader& ctx) const {
std::u32string acc;
fn()
return { false, {} };
TokenResult r;
for (;;) {
auto i = ctx.push();
auto c = ctx.current();
if (!c || !fn(*c)) {
ctx.pop(i);
break;
}
r.match += *c;
ctx.nextChar();
}
r.success = !r.match.empty();
return r;
}
// ============================================================================
// SeqToken Implementation
// ============================================================================
// ============================================================================30520370
SeqToken::SeqToken(const ilist<ref<const Token>>& list) : tokens(list) {}
SeqToken::SeqToken(const vector<const Token*>& tokens) : tokens(tokens) {}
TokenResult SeqToken::test(TextReader& ctx) const {
// this is a common branch point
std::u32string acc;
auto tri = ctx.push();
TokenResult r;
isize i = ctx.push();
// All matching steps within a sequence must pass consecutively.
for (const auto& token_ref : tokens) {
TokenResult res = token_ref.get().test(ctx);
TokenResult res = token_ref->test(ctx);
if (!res.success) {
// Strict ACID Transaction: Roll back context pointer entirely
// if any nested condition in the sequence fails.
ctx.pop(tri);
return { false, {} };
ctx.pop(i);
return { .success = false };
}
// Piecewise accumulation of individual matching sub-tokens
acc += res.match;
//r.match += res.match;
r.child.push_back(res);
}
return { true, acc };
}
SeqToken SeqToken::operator&(const Token& tok) {
// Intrusive chaining optimization: Appends the next token directly into the existing
// registry vector instead of nesting structures, keeping the layout flattened.
tokens.push_back(std::cref(tok));
return *this;
r.success = true;
return r;
}
@@ -88,42 +136,33 @@ namespace spider {
// OrToken Implementation
// ============================================================================
OrToken::OrToken(const ilist<ref<const Token>>& list) : tokens(list) {}
OrToken::OrToken(const vector<const Token*>& tokens) : tokens(tokens) {}
TokenResult OrToken::test(TextReader& ctx) const {
// this is a common branch point
auto tri = ctx.push();
// All matching steps within a sequence must pass consecutively.
auto i = ctx.push();
for (const auto& token_ref : tokens) {
// Short-circuit branch: return immediately on first valid choice match
TokenResult res = token_ref.get().test(ctx);
TokenResult res = token_ref->test(ctx);
if (res.success) return res;
// Backtrack isolation: Reset the cursor position before testing the next alternative path
ctx.pop(tri);
ctx.pop(i);
}
return { false, {} };
return { .success = false };
}
OrToken OrToken::operator|(const Token& tok) {
// Intrusive grouping layout optimization:
// flattens alternative tokens at code evaluation time.
tokens.push_back(tok);
return *this;
}
// ============================================================================
// OptToken Implementation
// ============================================================================
OptToken::OptToken(const Token& t) : target(t) {}
OptToken::OptToken(const Token* t) : target(t) {}
TokenResult OptToken::test(TextReader& ctx) const {
auto tri = ctx.push();
TokenResult res = target.test(ctx);
TokenResult res = target->test(ctx);
if (res.success) {
return res; // Option matched exactly 1 instance successfully
@@ -132,46 +171,58 @@ namespace spider {
// Recovery path: If sub-rule fails, clean up the dirty state mutation
// and successfully return an empty match payload (0 instances).
ctx.pop(tri);
return { true, U"" };
return { .success = false };
}
OptToken OptToken::operator~() {
// Redundant layer trap protection: returning self
// prevents wrapping an Optional in an Optional
return *this;
}
// ============================================================================
// RepToken Implementation
// ============================================================================
RepToken::RepToken(const Token& t) : target(t) {}
RepToken::RepToken(const Token* t) : target(t) {}
TokenResult RepToken::test(TextReader& ctx) const {
std::u32string acc;
auto tri = ctx.push();
TokenResult r;
for (;;) {
TokenResult res = target.test(ctx);
auto i = ctx.push();
TokenResult res = target->test(ctx);
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
if(!res.success || tri == ctx.push()) {
ctx.pop(tri);
if (!res.success || i == ctx.push()) {
ctx.pop(i);
break;
}
acc += res.match;
//r.match += res.match;
r.child.push_back(res);
}
// Repetition rules (* token) always evaluate to successful
// completion state, even with 0 matches.
return { true, acc };
r.success = true;
return r;
}
RepToken RepToken::operator*() {
// Redundant layer trap protection: returning self prevents wrapping
// a Repetition rule inside a Repetition rule
return *this;
// Tagged Token
TagToken::TagToken(const Token* t, std::string_view tag, bool doflatten)
: target(t), tag_name(tag), flatten(doflatten) { }
TokenResult TagToken::test(TextReader& ctx) const {
auto r = target->test(ctx);
if (!r.success) return r;
// Set tag of this result
r.tag = tag_name;
if(flatten) {
r.match = U"";
for(auto e : r.child) {
r.match += e.match;
}
r.child.clear();
}
return r;
}
}
+67 -48
View File
@@ -13,7 +13,9 @@ namespace spider {
struct TokenResult {
/** @brief Indicates if the token composition successfully matched the input boundary. */
bool success;
bool success = false;
optional<std::string_view> tag;
/**
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
@@ -21,13 +23,14 @@ namespace spider {
*/
std::u32string match;
vector<TokenResult> child;
};
// Forward declarations required by the abstract interface for operator returns.
class SeqToken;
class OrToken;
class OptToken;
class RepToken;
class Token;
class TokenFactory;
using FnTokenFn = std::function<bool(u32)>;
/**
* @brief Pure virtual base class defining the EBNF combinator node contract.
@@ -50,31 +53,46 @@ namespace spider {
*/
virtual TokenResult test(TextReader& ctx) const = 0;
};
class TokenFactory {
private:
// Arena owning all created tokens
std::vector<std::unique_ptr<Token>> arena;
// Deduplication caches
std::unordered_map<std::string, const Token*> lit_cache;
public:
/**
* @brief Chains this token and another sequentially.
* @return A temporary structural bridge matching both tokens sequentially.
*/
virtual SeqToken operator&(const Token& tok);
TokenFactory() = default;
/**
* @brief Combines this token and another under alternation.
* @return A structural bridge matching either this token or the fallback selection.
*/
virtual OrToken operator|(const Token& tok);
// Prevent copying to maintain valid internal pointers
TokenFactory(const TokenFactory&) = delete;
/**
* @brief Wraps this node in an optional layout rule.
* @return A structure matching zero or one instances of this current node.
*/
virtual OptToken operator~();
TokenFactory& operator=(const TokenFactory&) = delete;
/**
* @brief Wraps this node in a repetitive loop match framework.
* @return A structure matching zero or more occurrences of this current node.
*/
virtual RepToken operator*();
public:
// --- Primitive Constructors ---
Token* lit(std::string_view text);
Token* operator[](std::string_view text);
Token* fn(FnTokenFn predicate);
// --- Combinator Constructors ---
Token* seq(const vector<const Token*>& tokens);
Token* choice(std::string_view opts);
Token* choice(const vector<const Token*>& tokens);
Token* opt(const Token* target);
Token* rep(const Token* target);
};
@@ -93,8 +111,6 @@ namespace spider {
*/
LitToken(std::string_view lit);
LitToken(const char* lit);
/** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */
explicit LitToken(std::u32string lit);
@@ -106,8 +122,6 @@ namespace spider {
TokenResult test(TextReader& ctx) const override;
};
using FnTokenFn = std::function<bool(u32) >;
/**
* @brief Function based token
*/
@@ -135,11 +149,10 @@ namespace spider {
* @brief Internal contiguous layout registry storing lightweight, zero-overhead references.
* @details Avoids heap allocation penalties by referencing static instances immutably.
*/
vector<ref<const Token>> tokens;
vector<const Token*> tokens;
public:
/** @brief Constructs a sequence container directly out of an inline brace-enclosed listing. */
SeqToken(const ilist<ref<const Token>>& list);
SeqToken(const vector<const Token*>& tokens);
public:
/**
@@ -149,8 +162,6 @@ namespace spider {
*/
TokenResult test(TextReader& ctx) const override;
/** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */
SeqToken operator&(const Token& tok) override;
};
/**
@@ -159,11 +170,10 @@ namespace spider {
class OrToken : public Token {
private:
/** @brief Ordered registry of possible alternate structural paths. */
vector<ref<const Token>> tokens;
vector<const Token*> tokens;
public:
/** @brief Constructs an alternation choice layout from brace-enclosed tokens. */
OrToken(const ilist<ref<const Token>>& list);
OrToken(const vector<const Token*>& tokens);
public:
/**
@@ -172,8 +182,6 @@ namespace spider {
*/
TokenResult test(TextReader& ctx) const override;
/** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */
OrToken operator|(const Token& tok) override;
};
/**
@@ -182,11 +190,11 @@ namespace spider {
class OptToken : public Token {
private:
/** @brief Read-only target node reference to test optional status against. */
const Token& target;
const Token* target;
public:
/** @brief Binds the target node rule structural dependency layout wrapper. */
explicit OptToken(const Token& t);
explicit OptToken(const Token* t);
public:
/**
@@ -195,8 +203,6 @@ namespace spider {
*/
TokenResult test(TextReader& ctx) const override;
/** @brief Stub override providing standard compliance with the base Token interface signature. */
OptToken operator~() override;
};
/**
@@ -205,11 +211,11 @@ namespace spider {
class RepToken : public Token {
private:
/** @brief The base token node sequence layer evaluated in loops. */
const Token& target;
const Token* target;
public:
/** @brief Binds the repeated structural blueprint node wrapper configuration. */
explicit RepToken(const Token& t);
explicit RepToken(const Token* t);
public:
/**
@@ -219,8 +225,21 @@ namespace spider {
*/
TokenResult test(TextReader& ctx) const override;
/** @brief Stub override providing standard compliance with the base Token interface signature. */
RepToken operator*() override;
};
class TagToken : public Token {
private:
const Token* target;
std::string tag_name;
bool flatten;
public:
TagToken(const Token* t, std::string_view tag, bool doflatten = false);
TokenResult test(TextReader& ctx) const override;
};
}