beggining to stablish the EBNF of spider assembly
This commit is contained in:
+12
-9
@@ -35,7 +35,7 @@ float_lit = [ sign ] , (
|
||||
( digit , { digit } , "." , digit , { digit } , [ exponent ] ) |
|
||||
( "." , digit , { digit } , [ exponent ] ) |
|
||||
( digit , { digit } , exponent )
|
||||
) , [ "F" | "D" ] ;
|
||||
) , [ "F" | "D" ] ;
|
||||
|
||||
hex_lit = [ sign ] , "0x" , hex_digit , { hex_digit } ;
|
||||
octal_lit = [ sign ] , "0c" , octal_digit , { octal_digit } ;
|
||||
@@ -60,17 +60,20 @@ opcode = letter , { alpha_num_char } ;
|
||||
operand_list = operand , { "," , ws_optional , operand } ;
|
||||
instruction = opcode , [ whitespace , operand_list ] ;
|
||||
|
||||
(* Added Preprocessor, Sections, and Metadata Syntaxes *)
|
||||
include_decl = "include", whitespace, string_lit ;
|
||||
annotation_oper = identifier, [ ws_optional, "=", ws_optional, literal_decl ] ;
|
||||
annotation_ops = annotation_oper , { ws_optional, "," , ws_optional , annotation_oper } ;
|
||||
annotation_args = "(", ws_optional, annotation_ops, ws_optional, ")" ;
|
||||
annotation = "@", identifier, [ annotation_args ] ;
|
||||
section_decl = "section", whitespace, ".", identifier ;
|
||||
(* Added Preprocessor, Annotation *)
|
||||
annotation_named = identifier, ws_optional, "=", ws_optional, literal_decl;
|
||||
annotation_arg = literal_decl | annotation_named;
|
||||
annotation_args = annotation_arg, { ws_optional, "," , ws_optional , annotation_arg } ;
|
||||
annotation_pars = "(", ws_optional, annotation_args, ws_optional, ")" ;
|
||||
annotation = "@", identifier, [ annotation_pars ] ;
|
||||
preprocessor_val = identifier | string_lit;
|
||||
preprocessor = "#", identifier, whitespace, preprocessor_val;
|
||||
|
||||
(* Line Structure *)
|
||||
label = identifier, ":" ;
|
||||
line_content = include_decl | section_decl | ( [ annotation, whitespace ], [ label, ws_optional ], [ instruction ] ) ;
|
||||
line_label = label, [ whitespace, instruction ];
|
||||
line_annotation = annotation, [ whitespace, instruction ];
|
||||
line_content = preprocessor | line_annotation | line_label | instruction;
|
||||
line = ws_optional, [ line_content ], ws_optional , [ comment ] , newline ;
|
||||
line_last = ws_optional, [ line_content ], ws_optional , [ comment ] ;
|
||||
program = { line }, [ line_last ] ;
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
#pragma once
|
||||
|
||||
#include <spider/compiler/text/Token.hpp>
|
||||
|
||||
namespace spider {
|
||||
|
||||
|
||||
|
||||
}
|
||||
@@ -33,7 +33,10 @@ namespace spider {
|
||||
// and then do an easy compare!
|
||||
std::u32string str;
|
||||
if(!utf8::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!");
|
||||
return eat(str);
|
||||
}
|
||||
|
||||
bool TextReader::eat(const std::u32string& str) {
|
||||
// compare now
|
||||
isize index;
|
||||
for(index = 0; index < str.size(); index++) {
|
||||
@@ -68,46 +71,37 @@ namespace spider {
|
||||
/**
|
||||
* Returns the current character.
|
||||
*/
|
||||
u32 TextReader::current() {
|
||||
optional<u32> TextReader::current() {
|
||||
if (bufferIndex < buffer.size()) {
|
||||
return buffer[bufferIndex];
|
||||
}
|
||||
return 0;
|
||||
return {};
|
||||
}
|
||||
|
||||
/**
|
||||
* Reads the next character and advances the position tracker.
|
||||
*/
|
||||
u32 TextReader::nextChar(isize n) {
|
||||
if (err) return 0;
|
||||
|
||||
optional<u32> TextReader::nextChar(isize n) {
|
||||
// Ensure the character we are moving TO exists
|
||||
if (fillBufferTo(n)) {
|
||||
// advance n characters
|
||||
while(n--) {
|
||||
advance(current());
|
||||
advance(buffer[bufferIndex]);
|
||||
bufferIndex++;
|
||||
}
|
||||
return current();
|
||||
}
|
||||
|
||||
// If we couldn't fill the buffer, we hit EOF
|
||||
eof = true;
|
||||
return 0;
|
||||
return {};
|
||||
}
|
||||
|
||||
/**
|
||||
* Keeps the next n-th character (n = 0 is current).
|
||||
*/
|
||||
u32 TextReader::peekChar(isize n) {
|
||||
if (err) return 0;
|
||||
optional<u32> TextReader::peekChar(isize n) {
|
||||
if (fillBufferTo(n)) return buffer[bufferIndex + n];
|
||||
return 0;
|
||||
return {};
|
||||
}
|
||||
|
||||
/**
|
||||
* Clears the buffer from previous characters, keeping current and future ones.
|
||||
*/
|
||||
void TextReader::commit() {
|
||||
if (bufferIndex > 0) {
|
||||
// Erase everything before the current buffer index
|
||||
@@ -116,21 +110,12 @@ namespace spider {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Rolls back any previous characters within the limits of the uncommitted buffer.
|
||||
*/
|
||||
void TextReader::rollback(isize n) {
|
||||
// Prevent rolling back past the start of our committed buffer
|
||||
if (n > bufferIndex) n = bufferIndex;
|
||||
|
||||
// We must track positions backward or recalculate if exact column match is needed.
|
||||
// Assuming simple rollback of the pointer here per definition.
|
||||
bufferIndex -= n;
|
||||
eof = false;
|
||||
isize TextReader::push() {
|
||||
return bufferIndex;
|
||||
}
|
||||
|
||||
TextReader::operator bool() const {
|
||||
return !err;
|
||||
void TextReader::pop(isize index) {
|
||||
bufferIndex = std::min(index, bufferIndex);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -191,16 +176,20 @@ namespace spider {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns true if the stream is consumed and no elements remain in the read buffer.
|
||||
*/
|
||||
bool TextReader::isEOF() {
|
||||
if (err) return false;
|
||||
pos TextReader::getPosition() const {
|
||||
return at;
|
||||
}
|
||||
|
||||
bool TextReader::isEOF() const{
|
||||
return eof && bufferIndex >= buffer.size();
|
||||
}
|
||||
|
||||
pos TextReader::getPosition() const {
|
||||
return at;
|
||||
bool TextReader::hasError() const{
|
||||
return err;
|
||||
}
|
||||
|
||||
TextReader::operator bool() const {
|
||||
return !isEOF() && !hasError();
|
||||
}
|
||||
|
||||
std::string TextReader::getError() const {
|
||||
|
||||
@@ -84,53 +84,70 @@ namespace spider {
|
||||
*/
|
||||
bool eat(const std::string& chars);
|
||||
|
||||
bool eat(const std::u32string& chars);
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
* Returns the current character.
|
||||
*/
|
||||
u32 current();
|
||||
optional<u32> current();
|
||||
|
||||
/**
|
||||
* Reads the next n-th character.
|
||||
* n = 0 is a noop, since it's the current one.
|
||||
*/
|
||||
u32 nextChar(isize n = 1);
|
||||
optional<u32> nextChar(isize n = 1);
|
||||
|
||||
/**
|
||||
* Keeps the next n-th character
|
||||
* n = 0 is the current one.
|
||||
*/
|
||||
u32 peekChar(isize n = 1);
|
||||
optional<u32> peekChar(isize n = 1);
|
||||
|
||||
/**
|
||||
* Clears the buffer from previous characters,
|
||||
* removing the ability for rolling back
|
||||
* any previous characters from this point on.
|
||||
*
|
||||
* Inside a parser, make sure to call this once
|
||||
* no other previous syntaxes are possible. For
|
||||
* example, after every line.
|
||||
*/
|
||||
void commit();
|
||||
|
||||
/**
|
||||
* Rolls back any previous characters,
|
||||
* so long as the state hasn't commited.
|
||||
* n = 0 is a no op, since it's the current char.
|
||||
* Returns the current buffer index.
|
||||
* Inside a parser, this allows to roll
|
||||
* back the index to a specific position.
|
||||
*/
|
||||
void rollback(isize n = isize(-1));
|
||||
isize push();
|
||||
|
||||
/**
|
||||
* Sets the current buffer index.
|
||||
* Inside a parser, rolls back to
|
||||
* a previous position.
|
||||
*/
|
||||
void pop(isize index);
|
||||
|
||||
/**
|
||||
* Returns true if the end of the stream has been reached.
|
||||
* Returns false if the EOS hasn't been reached but
|
||||
* an error has occurred
|
||||
*/
|
||||
bool isEOF();
|
||||
bool isEOF() const;
|
||||
|
||||
/**
|
||||
* Returns the position of the cursor.
|
||||
*/
|
||||
pos getPosition() const;
|
||||
|
||||
/**
|
||||
* Returns true if this isn't EOF and there
|
||||
* is no error.
|
||||
*/
|
||||
operator bool() const;
|
||||
|
||||
bool hasError() const;
|
||||
|
||||
std::string getError() const;
|
||||
|
||||
protected:
|
||||
|
||||
@@ -32,22 +32,8 @@ namespace spider {
|
||||
|
||||
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
|
||||
|
||||
TokenResult LitToken::test(TokenContext& ctx) const {
|
||||
// Safety check: Prevent out-of-bounds pointer slicing if the remaining
|
||||
// input is smaller than the target literal.
|
||||
if (ctx.cursor + literal.size() > ctx.input.size()) {
|
||||
return { false, {} };
|
||||
}
|
||||
|
||||
// Window extract optimization: Acquire a zero-copy view over the input segment
|
||||
std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size());
|
||||
|
||||
// Direct lexicographical verification of the UTF-32 code-point sequence
|
||||
if (sub == literal) {
|
||||
ctx.advance(literal.size());
|
||||
return { true, std::u32string(sub) }; // Deep copy payload returned per requirement
|
||||
}
|
||||
|
||||
TokenResult LitToken::test(TextReader& ctx) const {
|
||||
if(ctx.eat(literal)) return { true, literal };
|
||||
return { false, {} };
|
||||
}
|
||||
|
||||
@@ -58,9 +44,10 @@ namespace spider {
|
||||
|
||||
SeqToken::SeqToken(const ilist<ref<const Token>>& list) : tokens(list) {}
|
||||
|
||||
TokenResult SeqToken::test(TokenContext& ctx) const {
|
||||
const size_t transactional_fallback_pos = ctx.cursor;
|
||||
std::u32string accumulated_match;
|
||||
TokenResult SeqToken::test(TextReader& ctx) const {
|
||||
// this is a common branch point
|
||||
std::u32string acc;
|
||||
auto tri = ctx.push();
|
||||
|
||||
// All matching steps within a sequence must pass consecutively.
|
||||
for (const auto& token_ref : tokens) {
|
||||
@@ -69,15 +56,15 @@ namespace spider {
|
||||
if (!res.success) {
|
||||
// Strict ACID Transaction: Roll back context pointer entirely
|
||||
// if any nested condition in the sequence fails.
|
||||
ctx.cursor = transactional_fallback_pos;
|
||||
ctx.pop(tri);
|
||||
return { false, {} };
|
||||
}
|
||||
|
||||
// Piecewise accumulation of individual matching sub-tokens
|
||||
accumulated_match += res.match;
|
||||
acc += res.match;
|
||||
}
|
||||
|
||||
return { true, accumulated_match };
|
||||
return { true, acc };
|
||||
}
|
||||
|
||||
SeqToken SeqToken::operator&(const Token& tok) {
|
||||
@@ -94,19 +81,18 @@ namespace spider {
|
||||
|
||||
OrToken::OrToken(const ilist<ref<const Token>>& list) : tokens(list) {}
|
||||
|
||||
TokenResult OrToken::test(TokenContext& ctx) const {
|
||||
const size_t local_fallback_pos = ctx.cursor;
|
||||
TokenResult OrToken::test(TextReader& ctx) const {
|
||||
// this is a common branch point
|
||||
auto tri = ctx.push();
|
||||
|
||||
// Ordered choice evaluation: Evaluate variants sequentially.
|
||||
// All matching steps within a sequence must pass consecutively.
|
||||
for (const auto& token_ref : tokens) {
|
||||
// Short-circuit branch: return immediately on first valid choice match
|
||||
TokenResult res = token_ref.get().test(ctx);
|
||||
|
||||
if (res.success) {
|
||||
return res; // Short-circuit branch: return immediately on first valid choice match
|
||||
}
|
||||
if (res.success) return res;
|
||||
|
||||
// Backtrack isolation: Reset the cursor position before testing the next alternative path
|
||||
ctx.cursor = local_fallback_pos;
|
||||
ctx.pop(tri);
|
||||
}
|
||||
|
||||
return { false, {} };
|
||||
@@ -126,8 +112,8 @@ namespace spider {
|
||||
|
||||
OptToken::OptToken(const Token& t) : target(t) {}
|
||||
|
||||
TokenResult OptToken::test(TokenContext& ctx) const {
|
||||
const size_t local_fallback_pos = ctx.cursor;
|
||||
TokenResult OptToken::test(TextReader& ctx) const {
|
||||
auto tri = ctx.push();
|
||||
TokenResult res = target.test(ctx);
|
||||
|
||||
if (res.success) {
|
||||
@@ -136,7 +122,7 @@ namespace spider {
|
||||
|
||||
// Recovery path: If sub-rule fails, clean up the dirty state mutation
|
||||
// and successfully return an empty match payload (0 instances).
|
||||
ctx.cursor = local_fallback_pos;
|
||||
ctx.pop(tri);
|
||||
return { true, U"" };
|
||||
}
|
||||
|
||||
@@ -153,30 +139,29 @@ namespace spider {
|
||||
|
||||
RepToken::RepToken(const Token& t) : target(t) {}
|
||||
|
||||
TokenResult RepToken::test(TokenContext& ctx) const {
|
||||
std::u32string accumulated_match;
|
||||
TokenResult RepToken::test(TextReader& ctx) const {
|
||||
std::u32string acc;
|
||||
auto tri = ctx.push();
|
||||
|
||||
// Greedily consume matches while input stream headroom remains
|
||||
while (ctx.has_more()) {
|
||||
const size_t pre_loop_cursor = ctx.cursor;
|
||||
for(;;) {
|
||||
TokenResult res = target.test(ctx);
|
||||
|
||||
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
||||
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
||||
if (!res.success || ctx.cursor == pre_loop_cursor) {
|
||||
ctx.cursor = pre_loop_cursor; // Revert cursor to last healthy match checkpoint
|
||||
if(!res.success || tri == ctx.push()) {
|
||||
ctx.pop(tri);
|
||||
break;
|
||||
}
|
||||
|
||||
accumulated_match += res.match;
|
||||
acc += res.match;
|
||||
}
|
||||
|
||||
// Repetition rules (* token) always evaluate to successful completion state, even with 0 matches.
|
||||
return { true, accumulated_match };
|
||||
// Repetition rules (* token) always evaluate to successful
|
||||
// completion state, even with 0 matches.
|
||||
return { true, acc };
|
||||
}
|
||||
|
||||
RepToken RepToken::operator*() {
|
||||
// Redundant layer trap protection: returning self prevents wrapping a Repetition rule inside a Repetition rule
|
||||
// Redundant layer trap protection: returning self prevents wrapping
|
||||
// a Repetition rule inside a Repetition rule
|
||||
return *this;
|
||||
}
|
||||
|
||||
|
||||
@@ -3,23 +3,13 @@
|
||||
#include <spider/compiler/common.hpp>
|
||||
|
||||
#include <spider/compiler/text/utf8.hpp>
|
||||
#include <spider/compiler/text/TextReader.hpp>
|
||||
|
||||
namespace spider {
|
||||
|
||||
struct TokenContext {
|
||||
|
||||
std::u32string_view input;
|
||||
size_t cursor = 0;
|
||||
|
||||
bool has_more() const { return cursor < input.size(); }
|
||||
char32_t peek() const { return input[cursor]; }
|
||||
void advance(size_t n = 1) { cursor += n; }
|
||||
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief The structural payload returned by every parsing component execution.
|
||||
*/
|
||||
* @brief The structural payload returned by every parsing component execution.
|
||||
*/
|
||||
struct TokenResult {
|
||||
|
||||
/** @brief Indicates if the token composition successfully matched the input boundary. */
|
||||
@@ -58,7 +48,7 @@ namespace spider {
|
||||
* @return TokenResult Containing verification state and the parsed copy of matching data.
|
||||
* @note Pure virtual; implementation details handle node-specific combinator semantics.
|
||||
*/
|
||||
virtual TokenResult test(TokenContext& ctx) const = 0;
|
||||
virtual TokenResult test(TextReader& ctx) const = 0;
|
||||
|
||||
public:
|
||||
|
||||
@@ -111,7 +101,7 @@ namespace spider {
|
||||
* @brief Validates match of the backing u32string exactly at the context cursor pointer.
|
||||
* @details Advances the context cursor precisely by literal length on success; zero state mutation on failure.
|
||||
*/
|
||||
TokenResult test(TokenContext& ctx) const override;
|
||||
TokenResult test(TextReader& ctx) const override;
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -135,7 +125,7 @@ namespace spider {
|
||||
* @details Implements a strict transaction boundary: if any internal element fails, the index
|
||||
* backtracks entirely to its starting cursor value before returning failure.
|
||||
*/
|
||||
TokenResult test(TokenContext& ctx) const override;
|
||||
TokenResult test(TextReader& ctx) const override;
|
||||
|
||||
/** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */
|
||||
SeqToken operator&(const Token& tok) override;
|
||||
@@ -158,7 +148,7 @@ namespace spider {
|
||||
* @brief Scans through alternatives, resolving immediately on the first candidate that passes.
|
||||
* @details Safely rolls back changes to the context cursor point between failed alternative attempts.
|
||||
*/
|
||||
TokenResult test(TokenContext& ctx) const override;
|
||||
TokenResult test(TextReader& ctx) const override;
|
||||
|
||||
/** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */
|
||||
OrToken operator|(const Token& tok) override;
|
||||
@@ -181,7 +171,7 @@ namespace spider {
|
||||
* @brief Evaluates target presence. Returns success true regardless of sub-rule evaluation outcome.
|
||||
* @details If the nested rule fails, the context cursor rolls back to initial state, returning empty matches.
|
||||
*/
|
||||
TokenResult test(TokenContext& ctx) const override;
|
||||
TokenResult test(TextReader& ctx) const override;
|
||||
|
||||
/** @brief Stub override providing standard compliance with the base Token interface signature. */
|
||||
OptToken operator~() override;
|
||||
@@ -205,7 +195,7 @@ namespace spider {
|
||||
* @details Includes internal safety loops checking cursor delta advancement to guarantee infinite
|
||||
* empty-matching sub-loops do not cause thread lockups.
|
||||
*/
|
||||
TokenResult test(TokenContext& ctx) const override;
|
||||
TokenResult test(TextReader& ctx) const override;
|
||||
|
||||
/** @brief Stub override providing standard compliance with the base Token interface signature. */
|
||||
RepToken operator*() override;
|
||||
|
||||
Reference in New Issue
Block a user