beggining to stablish the EBNF of spider assembly

This commit is contained in:
2026-07-19 22:43:23 -06:00
parent b30a59accd
commit 3a44ec594a
7 changed files with 114 additions and 121 deletions
@@ -0,0 +1,9 @@
#pragma once
#include <spider/compiler/text/Token.hpp>
namespace spider {
}
+25 -36
View File
@@ -33,7 +33,10 @@ namespace spider {
// and then do an easy compare!
std::u32string str;
if(!utf8::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!");
return eat(str);
}
bool TextReader::eat(const std::u32string& str) {
// compare now
isize index;
for(index = 0; index < str.size(); index++) {
@@ -68,46 +71,37 @@ namespace spider {
/**
* Returns the current character.
*/
u32 TextReader::current() {
optional<u32> TextReader::current() {
if (bufferIndex < buffer.size()) {
return buffer[bufferIndex];
}
return 0;
return {};
}
/**
* Reads the next character and advances the position tracker.
*/
u32 TextReader::nextChar(isize n) {
if (err) return 0;
optional<u32> TextReader::nextChar(isize n) {
// Ensure the character we are moving TO exists
if (fillBufferTo(n)) {
// advance n characters
while(n--) {
advance(current());
advance(buffer[bufferIndex]);
bufferIndex++;
}
return current();
}
// If we couldn't fill the buffer, we hit EOF
eof = true;
return 0;
return {};
}
/**
* Keeps the next n-th character (n = 0 is current).
*/
u32 TextReader::peekChar(isize n) {
if (err) return 0;
optional<u32> TextReader::peekChar(isize n) {
if (fillBufferTo(n)) return buffer[bufferIndex + n];
return 0;
return {};
}
/**
* Clears the buffer from previous characters, keeping current and future ones.
*/
void TextReader::commit() {
if (bufferIndex > 0) {
// Erase everything before the current buffer index
@@ -116,21 +110,12 @@ namespace spider {
}
}
/**
* Rolls back any previous characters within the limits of the uncommitted buffer.
*/
void TextReader::rollback(isize n) {
// Prevent rolling back past the start of our committed buffer
if (n > bufferIndex) n = bufferIndex;
// We must track positions backward or recalculate if exact column match is needed.
// Assuming simple rollback of the pointer here per definition.
bufferIndex -= n;
eof = false;
isize TextReader::push() {
return bufferIndex;
}
TextReader::operator bool() const {
return !err;
void TextReader::pop(isize index) {
bufferIndex = std::min(index, bufferIndex);
}
/**
@@ -191,16 +176,20 @@ namespace spider {
return true;
}
/**
* Returns true if the stream is consumed and no elements remain in the read buffer.
*/
bool TextReader::isEOF() {
if (err) return false;
pos TextReader::getPosition() const {
return at;
}
bool TextReader::isEOF() const{
return eof && bufferIndex >= buffer.size();
}
pos TextReader::getPosition() const {
return at;
bool TextReader::hasError() const{
return err;
}
TextReader::operator bool() const {
return !isEOF() && !hasError();
}
std::string TextReader::getError() const {
+27 -10
View File
@@ -84,53 +84,70 @@ namespace spider {
*/
bool eat(const std::string& chars);
bool eat(const std::u32string& chars);
public:
/**
* Returns the current character.
*/
u32 current();
optional<u32> current();
/**
* Reads the next n-th character.
* n = 0 is a noop, since it's the current one.
*/
u32 nextChar(isize n = 1);
optional<u32> nextChar(isize n = 1);
/**
* Keeps the next n-th character
* n = 0 is the current one.
*/
u32 peekChar(isize n = 1);
optional<u32> peekChar(isize n = 1);
/**
* Clears the buffer from previous characters,
* removing the ability for rolling back
* any previous characters from this point on.
*
* Inside a parser, make sure to call this once
* no other previous syntaxes are possible. For
* example, after every line.
*/
void commit();
/**
* Rolls back any previous characters,
* so long as the state hasn't commited.
* n = 0 is a no op, since it's the current char.
* Returns the current buffer index.
* Inside a parser, this allows to roll
* back the index to a specific position.
*/
void rollback(isize n = isize(-1));
isize push();
/**
* Sets the current buffer index.
* Inside a parser, rolls back to
* a previous position.
*/
void pop(isize index);
/**
* Returns true if the end of the stream has been reached.
* Returns false if the EOS hasn't been reached but
* an error has occurred
*/
bool isEOF();
bool isEOF() const;
/**
* Returns the position of the cursor.
*/
pos getPosition() const;
/**
* Returns true if this isn't EOF and there
* is no error.
*/
operator bool() const;
bool hasError() const;
std::string getError() const;
protected:
+32 -47
View File
@@ -32,22 +32,8 @@ namespace spider {
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
TokenResult LitToken::test(TokenContext& ctx) const {
// Safety check: Prevent out-of-bounds pointer slicing if the remaining
// input is smaller than the target literal.
if (ctx.cursor + literal.size() > ctx.input.size()) {
return { false, {} };
}
// Window extract optimization: Acquire a zero-copy view over the input segment
std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size());
// Direct lexicographical verification of the UTF-32 code-point sequence
if (sub == literal) {
ctx.advance(literal.size());
return { true, std::u32string(sub) }; // Deep copy payload returned per requirement
}
TokenResult LitToken::test(TextReader& ctx) const {
if(ctx.eat(literal)) return { true, literal };
return { false, {} };
}
@@ -58,9 +44,10 @@ namespace spider {
SeqToken::SeqToken(const ilist<ref<const Token>>& list) : tokens(list) {}
TokenResult SeqToken::test(TokenContext& ctx) const {
const size_t transactional_fallback_pos = ctx.cursor;
std::u32string accumulated_match;
TokenResult SeqToken::test(TextReader& ctx) const {
// this is a common branch point
std::u32string acc;
auto tri = ctx.push();
// All matching steps within a sequence must pass consecutively.
for (const auto& token_ref : tokens) {
@@ -69,15 +56,15 @@ namespace spider {
if (!res.success) {
// Strict ACID Transaction: Roll back context pointer entirely
// if any nested condition in the sequence fails.
ctx.cursor = transactional_fallback_pos;
ctx.pop(tri);
return { false, {} };
}
// Piecewise accumulation of individual matching sub-tokens
accumulated_match += res.match;
acc += res.match;
}
return { true, accumulated_match };
return { true, acc };
}
SeqToken SeqToken::operator&(const Token& tok) {
@@ -94,19 +81,18 @@ namespace spider {
OrToken::OrToken(const ilist<ref<const Token>>& list) : tokens(list) {}
TokenResult OrToken::test(TokenContext& ctx) const {
const size_t local_fallback_pos = ctx.cursor;
TokenResult OrToken::test(TextReader& ctx) const {
// this is a common branch point
auto tri = ctx.push();
// Ordered choice evaluation: Evaluate variants sequentially.
// All matching steps within a sequence must pass consecutively.
for (const auto& token_ref : tokens) {
// Short-circuit branch: return immediately on first valid choice match
TokenResult res = token_ref.get().test(ctx);
if (res.success) {
return res; // Short-circuit branch: return immediately on first valid choice match
}
if (res.success) return res;
// Backtrack isolation: Reset the cursor position before testing the next alternative path
ctx.cursor = local_fallback_pos;
ctx.pop(tri);
}
return { false, {} };
@@ -126,8 +112,8 @@ namespace spider {
OptToken::OptToken(const Token& t) : target(t) {}
TokenResult OptToken::test(TokenContext& ctx) const {
const size_t local_fallback_pos = ctx.cursor;
TokenResult OptToken::test(TextReader& ctx) const {
auto tri = ctx.push();
TokenResult res = target.test(ctx);
if (res.success) {
@@ -136,7 +122,7 @@ namespace spider {
// Recovery path: If sub-rule fails, clean up the dirty state mutation
// and successfully return an empty match payload (0 instances).
ctx.cursor = local_fallback_pos;
ctx.pop(tri);
return { true, U"" };
}
@@ -153,30 +139,29 @@ namespace spider {
RepToken::RepToken(const Token& t) : target(t) {}
TokenResult RepToken::test(TokenContext& ctx) const {
std::u32string accumulated_match;
TokenResult RepToken::test(TextReader& ctx) const {
std::u32string acc;
auto tri = ctx.push();
// Greedily consume matches while input stream headroom remains
while (ctx.has_more()) {
const size_t pre_loop_cursor = ctx.cursor;
for(;;) {
TokenResult res = target.test(ctx);
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
if (!res.success || ctx.cursor == pre_loop_cursor) {
ctx.cursor = pre_loop_cursor; // Revert cursor to last healthy match checkpoint
if(!res.success || tri == ctx.push()) {
ctx.pop(tri);
break;
}
accumulated_match += res.match;
acc += res.match;
}
// Repetition rules (* token) always evaluate to successful completion state, even with 0 matches.
return { true, accumulated_match };
// Repetition rules (* token) always evaluate to successful
// completion state, even with 0 matches.
return { true, acc };
}
RepToken RepToken::operator*() {
// Redundant layer trap protection: returning self prevents wrapping a Repetition rule inside a Repetition rule
// Redundant layer trap protection: returning self prevents wrapping
// a Repetition rule inside a Repetition rule
return *this;
}
+9 -19
View File
@@ -3,23 +3,13 @@
#include <spider/compiler/common.hpp>
#include <spider/compiler/text/utf8.hpp>
#include <spider/compiler/text/TextReader.hpp>
namespace spider {
struct TokenContext {
std::u32string_view input;
size_t cursor = 0;
bool has_more() const { return cursor < input.size(); }
char32_t peek() const { return input[cursor]; }
void advance(size_t n = 1) { cursor += n; }
};
/**
* @brief The structural payload returned by every parsing component execution.
*/
* @brief The structural payload returned by every parsing component execution.
*/
struct TokenResult {
/** @brief Indicates if the token composition successfully matched the input boundary. */
@@ -58,7 +48,7 @@ namespace spider {
* @return TokenResult Containing verification state and the parsed copy of matching data.
* @note Pure virtual; implementation details handle node-specific combinator semantics.
*/
virtual TokenResult test(TokenContext& ctx) const = 0;
virtual TokenResult test(TextReader& ctx) const = 0;
public:
@@ -111,7 +101,7 @@ namespace spider {
* @brief Validates match of the backing u32string exactly at the context cursor pointer.
* @details Advances the context cursor precisely by literal length on success; zero state mutation on failure.
*/
TokenResult test(TokenContext& ctx) const override;
TokenResult test(TextReader& ctx) const override;
};
/**
@@ -135,7 +125,7 @@ namespace spider {
* @details Implements a strict transaction boundary: if any internal element fails, the index
* backtracks entirely to its starting cursor value before returning failure.
*/
TokenResult test(TokenContext& ctx) const override;
TokenResult test(TextReader& ctx) const override;
/** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */
SeqToken operator&(const Token& tok) override;
@@ -158,7 +148,7 @@ namespace spider {
* @brief Scans through alternatives, resolving immediately on the first candidate that passes.
* @details Safely rolls back changes to the context cursor point between failed alternative attempts.
*/
TokenResult test(TokenContext& ctx) const override;
TokenResult test(TextReader& ctx) const override;
/** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */
OrToken operator|(const Token& tok) override;
@@ -181,7 +171,7 @@ namespace spider {
* @brief Evaluates target presence. Returns success true regardless of sub-rule evaluation outcome.
* @details If the nested rule fails, the context cursor rolls back to initial state, returning empty matches.
*/
TokenResult test(TokenContext& ctx) const override;
TokenResult test(TextReader& ctx) const override;
/** @brief Stub override providing standard compliance with the base Token interface signature. */
OptToken operator~() override;
@@ -205,7 +195,7 @@ namespace spider {
* @details Includes internal safety loops checking cursor delta advancement to guarantee infinite
* empty-matching sub-loops do not cause thread lockups.
*/
TokenResult test(TokenContext& ctx) const override;
TokenResult test(TextReader& ctx) const override;
/** @brief Stub override providing standard compliance with the base Token interface signature. */
RepToken operator*() override;