beggining to stablish the EBNF of spider assembly

This commit is contained in:
2026-07-19 22:43:23 -06:00
parent b30a59accd
commit 3a44ec594a
7 changed files with 114 additions and 121 deletions
+11 -8
View File
@@ -60,17 +60,20 @@ opcode = letter , { alpha_num_char } ;
operand_list = operand , { "," , ws_optional , operand } ; operand_list = operand , { "," , ws_optional , operand } ;
instruction = opcode , [ whitespace , operand_list ] ; instruction = opcode , [ whitespace , operand_list ] ;
(* Added Preprocessor, Sections, and Metadata Syntaxes *) (* Added Preprocessor, Annotation *)
include_decl = "include", whitespace, string_lit ; annotation_named = identifier, ws_optional, "=", ws_optional, literal_decl;
annotation_oper = identifier, [ ws_optional, "=", ws_optional, literal_decl ] ; annotation_arg = literal_decl | annotation_named;
annotation_ops = annotation_oper , { ws_optional, "," , ws_optional , annotation_oper } ; annotation_args = annotation_arg, { ws_optional, "," , ws_optional , annotation_arg } ;
annotation_args = "(", ws_optional, annotation_ops, ws_optional, ")" ; annotation_pars = "(", ws_optional, annotation_args, ws_optional, ")" ;
annotation = "@", identifier, [ annotation_args ] ; annotation = "@", identifier, [ annotation_pars ] ;
section_decl = "section", whitespace, ".", identifier ; preprocessor_val = identifier | string_lit;
preprocessor = "#", identifier, whitespace, preprocessor_val;
(* Line Structure *) (* Line Structure *)
label = identifier, ":" ; label = identifier, ":" ;
line_content = include_decl | section_decl | ( [ annotation, whitespace ], [ label, ws_optional ], [ instruction ] ) ; line_label = label, [ whitespace, instruction ];
line_annotation = annotation, [ whitespace, instruction ];
line_content = preprocessor | line_annotation | line_label | instruction;
line = ws_optional, [ line_content ], ws_optional , [ comment ] , newline ; line = ws_optional, [ line_content ], ws_optional , [ comment ] , newline ;
line_last = ws_optional, [ line_content ], ws_optional , [ comment ] ; line_last = ws_optional, [ line_content ], ws_optional , [ comment ] ;
program = { line }, [ line_last ] ; program = { line }, [ line_last ] ;
@@ -0,0 +1,9 @@
#pragma once
#include <spider/compiler/text/Token.hpp>
namespace spider {
}
+25 -36
View File
@@ -33,7 +33,10 @@ namespace spider {
// and then do an easy compare! // and then do an easy compare!
std::u32string str; std::u32string str;
if(!utf8::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!"); if(!utf8::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!");
return eat(str);
}
bool TextReader::eat(const std::u32string& str) {
// compare now // compare now
isize index; isize index;
for(index = 0; index < str.size(); index++) { for(index = 0; index < str.size(); index++) {
@@ -68,46 +71,37 @@ namespace spider {
/** /**
* Returns the current character. * Returns the current character.
*/ */
u32 TextReader::current() { optional<u32> TextReader::current() {
if (bufferIndex < buffer.size()) { if (bufferIndex < buffer.size()) {
return buffer[bufferIndex]; return buffer[bufferIndex];
} }
return 0; return {};
} }
/** /**
* Reads the next character and advances the position tracker. * Reads the next character and advances the position tracker.
*/ */
u32 TextReader::nextChar(isize n) { optional<u32> TextReader::nextChar(isize n) {
if (err) return 0;
// Ensure the character we are moving TO exists // Ensure the character we are moving TO exists
if (fillBufferTo(n)) { if (fillBufferTo(n)) {
// advance n characters // advance n characters
while(n--) { while(n--) {
advance(current()); advance(buffer[bufferIndex]);
bufferIndex++; bufferIndex++;
} }
return current(); return current();
} }
return {};
// If we couldn't fill the buffer, we hit EOF
eof = true;
return 0;
} }
/** /**
* Keeps the next n-th character (n = 0 is current). * Keeps the next n-th character (n = 0 is current).
*/ */
u32 TextReader::peekChar(isize n) { optional<u32> TextReader::peekChar(isize n) {
if (err) return 0;
if (fillBufferTo(n)) return buffer[bufferIndex + n]; if (fillBufferTo(n)) return buffer[bufferIndex + n];
return 0; return {};
} }
/**
* Clears the buffer from previous characters, keeping current and future ones.
*/
void TextReader::commit() { void TextReader::commit() {
if (bufferIndex > 0) { if (bufferIndex > 0) {
// Erase everything before the current buffer index // Erase everything before the current buffer index
@@ -116,21 +110,12 @@ namespace spider {
} }
} }
/** isize TextReader::push() {
* Rolls back any previous characters within the limits of the uncommitted buffer. return bufferIndex;
*/
void TextReader::rollback(isize n) {
// Prevent rolling back past the start of our committed buffer
if (n > bufferIndex) n = bufferIndex;
// We must track positions backward or recalculate if exact column match is needed.
// Assuming simple rollback of the pointer here per definition.
bufferIndex -= n;
eof = false;
} }
TextReader::operator bool() const { void TextReader::pop(isize index) {
return !err; bufferIndex = std::min(index, bufferIndex);
} }
/** /**
@@ -191,16 +176,20 @@ namespace spider {
return true; return true;
} }
/** pos TextReader::getPosition() const {
* Returns true if the stream is consumed and no elements remain in the read buffer. return at;
*/ }
bool TextReader::isEOF() {
if (err) return false; bool TextReader::isEOF() const{
return eof && bufferIndex >= buffer.size(); return eof && bufferIndex >= buffer.size();
} }
pos TextReader::getPosition() const { bool TextReader::hasError() const{
return at; return err;
}
TextReader::operator bool() const {
return !isEOF() && !hasError();
} }
std::string TextReader::getError() const { std::string TextReader::getError() const {
+27 -10
View File
@@ -84,53 +84,70 @@ namespace spider {
*/ */
bool eat(const std::string& chars); bool eat(const std::string& chars);
bool eat(const std::u32string& chars);
public: public:
/** /**
* Returns the current character. * Returns the current character.
*/ */
u32 current(); optional<u32> current();
/** /**
* Reads the next n-th character. * Reads the next n-th character.
* n = 0 is a noop, since it's the current one. * n = 0 is a noop, since it's the current one.
*/ */
u32 nextChar(isize n = 1); optional<u32> nextChar(isize n = 1);
/** /**
* Keeps the next n-th character * Keeps the next n-th character
* n = 0 is the current one. * n = 0 is the current one.
*/ */
u32 peekChar(isize n = 1); optional<u32> peekChar(isize n = 1);
/** /**
* Clears the buffer from previous characters, * Clears the buffer from previous characters,
* removing the ability for rolling back * removing the ability for rolling back
* any previous characters from this point on. * any previous characters from this point on.
*
* Inside a parser, make sure to call this once
* no other previous syntaxes are possible. For
* example, after every line.
*/ */
void commit(); void commit();
/** /**
* Rolls back any previous characters, * Returns the current buffer index.
* so long as the state hasn't commited. * Inside a parser, this allows to roll
* n = 0 is a no op, since it's the current char. * back the index to a specific position.
*/ */
void rollback(isize n = isize(-1)); isize push();
/**
* Sets the current buffer index.
* Inside a parser, rolls back to
* a previous position.
*/
void pop(isize index);
/** /**
* Returns true if the end of the stream has been reached. * Returns true if the end of the stream has been reached.
* Returns false if the EOS hasn't been reached but
* an error has occurred
*/ */
bool isEOF(); bool isEOF() const;
/** /**
* Returns the position of the cursor. * Returns the position of the cursor.
*/ */
pos getPosition() const; pos getPosition() const;
/**
* Returns true if this isn't EOF and there
* is no error.
*/
operator bool() const; operator bool() const;
bool hasError() const;
std::string getError() const; std::string getError() const;
protected: protected:
+31 -46
View File
@@ -32,22 +32,8 @@ namespace spider {
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {} LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
TokenResult LitToken::test(TokenContext& ctx) const { TokenResult LitToken::test(TextReader& ctx) const {
// Safety check: Prevent out-of-bounds pointer slicing if the remaining if(ctx.eat(literal)) return { true, literal };
// input is smaller than the target literal.
if (ctx.cursor + literal.size() > ctx.input.size()) {
return { false, {} };
}
// Window extract optimization: Acquire a zero-copy view over the input segment
std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size());
// Direct lexicographical verification of the UTF-32 code-point sequence
if (sub == literal) {
ctx.advance(literal.size());
return { true, std::u32string(sub) }; // Deep copy payload returned per requirement
}
return { false, {} }; return { false, {} };
} }
@@ -58,9 +44,10 @@ namespace spider {
SeqToken::SeqToken(const ilist<ref<const Token>>& list) : tokens(list) {} SeqToken::SeqToken(const ilist<ref<const Token>>& list) : tokens(list) {}
TokenResult SeqToken::test(TokenContext& ctx) const { TokenResult SeqToken::test(TextReader& ctx) const {
const size_t transactional_fallback_pos = ctx.cursor; // this is a common branch point
std::u32string accumulated_match; std::u32string acc;
auto tri = ctx.push();
// All matching steps within a sequence must pass consecutively. // All matching steps within a sequence must pass consecutively.
for (const auto& token_ref : tokens) { for (const auto& token_ref : tokens) {
@@ -69,15 +56,15 @@ namespace spider {
if (!res.success) { if (!res.success) {
// Strict ACID Transaction: Roll back context pointer entirely // Strict ACID Transaction: Roll back context pointer entirely
// if any nested condition in the sequence fails. // if any nested condition in the sequence fails.
ctx.cursor = transactional_fallback_pos; ctx.pop(tri);
return { false, {} }; return { false, {} };
} }
// Piecewise accumulation of individual matching sub-tokens // Piecewise accumulation of individual matching sub-tokens
accumulated_match += res.match; acc += res.match;
} }
return { true, accumulated_match }; return { true, acc };
} }
SeqToken SeqToken::operator&(const Token& tok) { SeqToken SeqToken::operator&(const Token& tok) {
@@ -94,19 +81,18 @@ namespace spider {
OrToken::OrToken(const ilist<ref<const Token>>& list) : tokens(list) {} OrToken::OrToken(const ilist<ref<const Token>>& list) : tokens(list) {}
TokenResult OrToken::test(TokenContext& ctx) const { TokenResult OrToken::test(TextReader& ctx) const {
const size_t local_fallback_pos = ctx.cursor; // this is a common branch point
auto tri = ctx.push();
// Ordered choice evaluation: Evaluate variants sequentially. // All matching steps within a sequence must pass consecutively.
for (const auto& token_ref : tokens) { for (const auto& token_ref : tokens) {
// Short-circuit branch: return immediately on first valid choice match
TokenResult res = token_ref.get().test(ctx); TokenResult res = token_ref.get().test(ctx);
if (res.success) return res;
if (res.success) {
return res; // Short-circuit branch: return immediately on first valid choice match
}
// Backtrack isolation: Reset the cursor position before testing the next alternative path // Backtrack isolation: Reset the cursor position before testing the next alternative path
ctx.cursor = local_fallback_pos; ctx.pop(tri);
} }
return { false, {} }; return { false, {} };
@@ -126,8 +112,8 @@ namespace spider {
OptToken::OptToken(const Token& t) : target(t) {} OptToken::OptToken(const Token& t) : target(t) {}
TokenResult OptToken::test(TokenContext& ctx) const { TokenResult OptToken::test(TextReader& ctx) const {
const size_t local_fallback_pos = ctx.cursor; auto tri = ctx.push();
TokenResult res = target.test(ctx); TokenResult res = target.test(ctx);
if (res.success) { if (res.success) {
@@ -136,7 +122,7 @@ namespace spider {
// Recovery path: If sub-rule fails, clean up the dirty state mutation // Recovery path: If sub-rule fails, clean up the dirty state mutation
// and successfully return an empty match payload (0 instances). // and successfully return an empty match payload (0 instances).
ctx.cursor = local_fallback_pos; ctx.pop(tri);
return { true, U"" }; return { true, U"" };
} }
@@ -153,30 +139,29 @@ namespace spider {
RepToken::RepToken(const Token& t) : target(t) {} RepToken::RepToken(const Token& t) : target(t) {}
TokenResult RepToken::test(TokenContext& ctx) const { TokenResult RepToken::test(TextReader& ctx) const {
std::u32string accumulated_match; std::u32string acc;
auto tri = ctx.push();
// Greedily consume matches while input stream headroom remains for(;;) {
while (ctx.has_more()) {
const size_t pre_loop_cursor = ctx.cursor;
TokenResult res = target.test(ctx); TokenResult res = target.test(ctx);
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching // Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups). // rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
if (!res.success || ctx.cursor == pre_loop_cursor) { if(!res.success || tri == ctx.push()) {
ctx.cursor = pre_loop_cursor; // Revert cursor to last healthy match checkpoint ctx.pop(tri);
break; break;
} }
acc += res.match;
accumulated_match += res.match;
} }
// Repetition rules (* token) always evaluate to successful completion state, even with 0 matches. // Repetition rules (* token) always evaluate to successful
return { true, accumulated_match }; // completion state, even with 0 matches.
return { true, acc };
} }
RepToken RepToken::operator*() { RepToken RepToken::operator*() {
// Redundant layer trap protection: returning self prevents wrapping a Repetition rule inside a Repetition rule // Redundant layer trap protection: returning self prevents wrapping
// a Repetition rule inside a Repetition rule
return *this; return *this;
} }
+7 -17
View File
@@ -3,20 +3,10 @@
#include <spider/compiler/common.hpp> #include <spider/compiler/common.hpp>
#include <spider/compiler/text/utf8.hpp> #include <spider/compiler/text/utf8.hpp>
#include <spider/compiler/text/TextReader.hpp>
namespace spider { namespace spider {
struct TokenContext {
std::u32string_view input;
size_t cursor = 0;
bool has_more() const { return cursor < input.size(); }
char32_t peek() const { return input[cursor]; }
void advance(size_t n = 1) { cursor += n; }
};
/** /**
* @brief The structural payload returned by every parsing component execution. * @brief The structural payload returned by every parsing component execution.
*/ */
@@ -58,7 +48,7 @@ namespace spider {
* @return TokenResult Containing verification state and the parsed copy of matching data. * @return TokenResult Containing verification state and the parsed copy of matching data.
* @note Pure virtual; implementation details handle node-specific combinator semantics. * @note Pure virtual; implementation details handle node-specific combinator semantics.
*/ */
virtual TokenResult test(TokenContext& ctx) const = 0; virtual TokenResult test(TextReader& ctx) const = 0;
public: public:
@@ -111,7 +101,7 @@ namespace spider {
* @brief Validates match of the backing u32string exactly at the context cursor pointer. * @brief Validates match of the backing u32string exactly at the context cursor pointer.
* @details Advances the context cursor precisely by literal length on success; zero state mutation on failure. * @details Advances the context cursor precisely by literal length on success; zero state mutation on failure.
*/ */
TokenResult test(TokenContext& ctx) const override; TokenResult test(TextReader& ctx) const override;
}; };
/** /**
@@ -135,7 +125,7 @@ namespace spider {
* @details Implements a strict transaction boundary: if any internal element fails, the index * @details Implements a strict transaction boundary: if any internal element fails, the index
* backtracks entirely to its starting cursor value before returning failure. * backtracks entirely to its starting cursor value before returning failure.
*/ */
TokenResult test(TokenContext& ctx) const override; TokenResult test(TextReader& ctx) const override;
/** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */ /** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */
SeqToken operator&(const Token& tok) override; SeqToken operator&(const Token& tok) override;
@@ -158,7 +148,7 @@ namespace spider {
* @brief Scans through alternatives, resolving immediately on the first candidate that passes. * @brief Scans through alternatives, resolving immediately on the first candidate that passes.
* @details Safely rolls back changes to the context cursor point between failed alternative attempts. * @details Safely rolls back changes to the context cursor point between failed alternative attempts.
*/ */
TokenResult test(TokenContext& ctx) const override; TokenResult test(TextReader& ctx) const override;
/** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */ /** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */
OrToken operator|(const Token& tok) override; OrToken operator|(const Token& tok) override;
@@ -181,7 +171,7 @@ namespace spider {
* @brief Evaluates target presence. Returns success true regardless of sub-rule evaluation outcome. * @brief Evaluates target presence. Returns success true regardless of sub-rule evaluation outcome.
* @details If the nested rule fails, the context cursor rolls back to initial state, returning empty matches. * @details If the nested rule fails, the context cursor rolls back to initial state, returning empty matches.
*/ */
TokenResult test(TokenContext& ctx) const override; TokenResult test(TextReader& ctx) const override;
/** @brief Stub override providing standard compliance with the base Token interface signature. */ /** @brief Stub override providing standard compliance with the base Token interface signature. */
OptToken operator~() override; OptToken operator~() override;
@@ -205,7 +195,7 @@ namespace spider {
* @details Includes internal safety loops checking cursor delta advancement to guarantee infinite * @details Includes internal safety loops checking cursor delta advancement to guarantee infinite
* empty-matching sub-loops do not cause thread lockups. * empty-matching sub-loops do not cause thread lockups.
*/ */
TokenResult test(TokenContext& ctx) const override; TokenResult test(TextReader& ctx) const override;
/** @brief Stub override providing standard compliance with the base Token interface signature. */ /** @brief Stub override providing standard compliance with the base Token interface signature. */
RepToken operator*() override; RepToken operator*() override;