diff --git a/samples/assembly.ebnf b/samples/assembly.ebnf index af8feec..38d97a4 100644 --- a/samples/assembly.ebnf +++ b/samples/assembly.ebnf @@ -35,7 +35,7 @@ float_lit = [ sign ] , ( ( digit , { digit } , "." , digit , { digit } , [ exponent ] ) | ( "." , digit , { digit } , [ exponent ] ) | ( digit , { digit } , exponent ) - ) , [ "F" | "D" ] ; + ) , [ "F" | "D" ] ; hex_lit = [ sign ] , "0x" , hex_digit , { hex_digit } ; octal_lit = [ sign ] , "0c" , octal_digit , { octal_digit } ; @@ -60,17 +60,20 @@ opcode = letter , { alpha_num_char } ; operand_list = operand , { "," , ws_optional , operand } ; instruction = opcode , [ whitespace , operand_list ] ; -(* Added Preprocessor, Sections, and Metadata Syntaxes *) -include_decl = "include", whitespace, string_lit ; -annotation_oper = identifier, [ ws_optional, "=", ws_optional, literal_decl ] ; -annotation_ops = annotation_oper , { ws_optional, "," , ws_optional , annotation_oper } ; -annotation_args = "(", ws_optional, annotation_ops, ws_optional, ")" ; -annotation = "@", identifier, [ annotation_args ] ; -section_decl = "section", whitespace, ".", identifier ; +(* Added Preprocessor, Annotation *) +annotation_named = identifier, ws_optional, "=", ws_optional, literal_decl; +annotation_arg = literal_decl | annotation_named; +annotation_args = annotation_arg, { ws_optional, "," , ws_optional , annotation_arg } ; +annotation_pars = "(", ws_optional, annotation_args, ws_optional, ")" ; +annotation = "@", identifier, [ annotation_pars ] ; +preprocessor_val = identifier | string_lit; +preprocessor = "#", identifier, whitespace, preprocessor_val; (* Line Structure *) label = identifier, ":" ; -line_content = include_decl | section_decl | ( [ annotation, whitespace ], [ label, ws_optional ], [ instruction ] ) ; +line_label = label, [ whitespace, instruction ]; +line_annotation = annotation, [ whitespace, instruction ]; +line_content = preprocessor | line_annotation | line_label | instruction; line = ws_optional, [ line_content ], ws_optional , [ comment ] , newline ; line_last = ws_optional, [ line_content ], ws_optional , [ comment ] ; program = { line }, [ line_last ] ; diff --git a/src/spider/compiler/assembler/AsmEBNF.cpp b/src/spider/compiler/assembler/AsmEBNF.cpp new file mode 100644 index 0000000..e69de29 diff --git a/src/spider/compiler/assembler/AsmEBNF.hpp b/src/spider/compiler/assembler/AsmEBNF.hpp new file mode 100644 index 0000000..eda512e --- /dev/null +++ b/src/spider/compiler/assembler/AsmEBNF.hpp @@ -0,0 +1,9 @@ +#pragma once + +#include + +namespace spider { + + + +} diff --git a/src/spider/compiler/text/TextReader.cpp b/src/spider/compiler/text/TextReader.cpp index 8c70eec..75accde 100644 --- a/src/spider/compiler/text/TextReader.cpp +++ b/src/spider/compiler/text/TextReader.cpp @@ -33,7 +33,10 @@ namespace spider { // and then do an easy compare! std::u32string str; if(!utf8::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!"); + return eat(str); + } + bool TextReader::eat(const std::u32string& str) { // compare now isize index; for(index = 0; index < str.size(); index++) { @@ -68,46 +71,37 @@ namespace spider { /** * Returns the current character. */ - u32 TextReader::current() { + optional TextReader::current() { if (bufferIndex < buffer.size()) { return buffer[bufferIndex]; } - return 0; + return {}; } /** * Reads the next character and advances the position tracker. */ - u32 TextReader::nextChar(isize n) { - if (err) return 0; - + optional TextReader::nextChar(isize n) { // Ensure the character we are moving TO exists if (fillBufferTo(n)) { // advance n characters while(n--) { - advance(current()); + advance(buffer[bufferIndex]); bufferIndex++; } return current(); } - - // If we couldn't fill the buffer, we hit EOF - eof = true; - return 0; + return {}; } /** * Keeps the next n-th character (n = 0 is current). */ - u32 TextReader::peekChar(isize n) { - if (err) return 0; + optional TextReader::peekChar(isize n) { if (fillBufferTo(n)) return buffer[bufferIndex + n]; - return 0; + return {}; } - /** - * Clears the buffer from previous characters, keeping current and future ones. - */ void TextReader::commit() { if (bufferIndex > 0) { // Erase everything before the current buffer index @@ -116,21 +110,12 @@ namespace spider { } } - /** - * Rolls back any previous characters within the limits of the uncommitted buffer. - */ - void TextReader::rollback(isize n) { - // Prevent rolling back past the start of our committed buffer - if (n > bufferIndex) n = bufferIndex; - - // We must track positions backward or recalculate if exact column match is needed. - // Assuming simple rollback of the pointer here per definition. - bufferIndex -= n; - eof = false; + isize TextReader::push() { + return bufferIndex; } - TextReader::operator bool() const { - return !err; + void TextReader::pop(isize index) { + bufferIndex = std::min(index, bufferIndex); } /** @@ -191,16 +176,20 @@ namespace spider { return true; } - /** - * Returns true if the stream is consumed and no elements remain in the read buffer. - */ - bool TextReader::isEOF() { - if (err) return false; + pos TextReader::getPosition() const { + return at; + } + + bool TextReader::isEOF() const{ return eof && bufferIndex >= buffer.size(); } - pos TextReader::getPosition() const { - return at; + bool TextReader::hasError() const{ + return err; + } + + TextReader::operator bool() const { + return !isEOF() && !hasError(); } std::string TextReader::getError() const { diff --git a/src/spider/compiler/text/TextReader.hpp b/src/spider/compiler/text/TextReader.hpp index 3f5c882..4bf1bb0 100644 --- a/src/spider/compiler/text/TextReader.hpp +++ b/src/spider/compiler/text/TextReader.hpp @@ -84,53 +84,70 @@ namespace spider { */ bool eat(const std::string& chars); + bool eat(const std::u32string& chars); + public: /** * Returns the current character. */ - u32 current(); + optional current(); /** * Reads the next n-th character. * n = 0 is a noop, since it's the current one. */ - u32 nextChar(isize n = 1); + optional nextChar(isize n = 1); /** * Keeps the next n-th character * n = 0 is the current one. */ - u32 peekChar(isize n = 1); + optional peekChar(isize n = 1); /** * Clears the buffer from previous characters, * removing the ability for rolling back * any previous characters from this point on. + * + * Inside a parser, make sure to call this once + * no other previous syntaxes are possible. For + * example, after every line. */ void commit(); /** - * Rolls back any previous characters, - * so long as the state hasn't commited. - * n = 0 is a no op, since it's the current char. + * Returns the current buffer index. + * Inside a parser, this allows to roll + * back the index to a specific position. */ - void rollback(isize n = isize(-1)); + isize push(); + + /** + * Sets the current buffer index. + * Inside a parser, rolls back to + * a previous position. + */ + void pop(isize index); /** * Returns true if the end of the stream has been reached. - * Returns false if the EOS hasn't been reached but - * an error has occurred */ - bool isEOF(); + bool isEOF() const; /** * Returns the position of the cursor. */ pos getPosition() const; + /** + * Returns true if this isn't EOF and there + * is no error. + */ operator bool() const; + bool hasError() const; + std::string getError() const; protected: diff --git a/src/spider/compiler/text/Token.cpp b/src/spider/compiler/text/Token.cpp index 638ed2f..03bdf7a 100644 --- a/src/spider/compiler/text/Token.cpp +++ b/src/spider/compiler/text/Token.cpp @@ -32,22 +32,8 @@ namespace spider { LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {} - TokenResult LitToken::test(TokenContext& ctx) const { - // Safety check: Prevent out-of-bounds pointer slicing if the remaining - // input is smaller than the target literal. - if (ctx.cursor + literal.size() > ctx.input.size()) { - return { false, {} }; - } - - // Window extract optimization: Acquire a zero-copy view over the input segment - std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size()); - - // Direct lexicographical verification of the UTF-32 code-point sequence - if (sub == literal) { - ctx.advance(literal.size()); - return { true, std::u32string(sub) }; // Deep copy payload returned per requirement - } - + TokenResult LitToken::test(TextReader& ctx) const { + if(ctx.eat(literal)) return { true, literal }; return { false, {} }; } @@ -58,9 +44,10 @@ namespace spider { SeqToken::SeqToken(const ilist>& list) : tokens(list) {} - TokenResult SeqToken::test(TokenContext& ctx) const { - const size_t transactional_fallback_pos = ctx.cursor; - std::u32string accumulated_match; + TokenResult SeqToken::test(TextReader& ctx) const { + // this is a common branch point + std::u32string acc; + auto tri = ctx.push(); // All matching steps within a sequence must pass consecutively. for (const auto& token_ref : tokens) { @@ -69,15 +56,15 @@ namespace spider { if (!res.success) { // Strict ACID Transaction: Roll back context pointer entirely // if any nested condition in the sequence fails. - ctx.cursor = transactional_fallback_pos; + ctx.pop(tri); return { false, {} }; } // Piecewise accumulation of individual matching sub-tokens - accumulated_match += res.match; + acc += res.match; } - return { true, accumulated_match }; + return { true, acc }; } SeqToken SeqToken::operator&(const Token& tok) { @@ -94,19 +81,18 @@ namespace spider { OrToken::OrToken(const ilist>& list) : tokens(list) {} - TokenResult OrToken::test(TokenContext& ctx) const { - const size_t local_fallback_pos = ctx.cursor; + TokenResult OrToken::test(TextReader& ctx) const { + // this is a common branch point + auto tri = ctx.push(); - // Ordered choice evaluation: Evaluate variants sequentially. + // All matching steps within a sequence must pass consecutively. for (const auto& token_ref : tokens) { + // Short-circuit branch: return immediately on first valid choice match TokenResult res = token_ref.get().test(ctx); - - if (res.success) { - return res; // Short-circuit branch: return immediately on first valid choice match - } + if (res.success) return res; // Backtrack isolation: Reset the cursor position before testing the next alternative path - ctx.cursor = local_fallback_pos; + ctx.pop(tri); } return { false, {} }; @@ -126,8 +112,8 @@ namespace spider { OptToken::OptToken(const Token& t) : target(t) {} - TokenResult OptToken::test(TokenContext& ctx) const { - const size_t local_fallback_pos = ctx.cursor; + TokenResult OptToken::test(TextReader& ctx) const { + auto tri = ctx.push(); TokenResult res = target.test(ctx); if (res.success) { @@ -136,7 +122,7 @@ namespace spider { // Recovery path: If sub-rule fails, clean up the dirty state mutation // and successfully return an empty match payload (0 instances). - ctx.cursor = local_fallback_pos; + ctx.pop(tri); return { true, U"" }; } @@ -153,30 +139,29 @@ namespace spider { RepToken::RepToken(const Token& t) : target(t) {} - TokenResult RepToken::test(TokenContext& ctx) const { - std::u32string accumulated_match; + TokenResult RepToken::test(TextReader& ctx) const { + std::u32string acc; + auto tri = ctx.push(); - // Greedily consume matches while input stream headroom remains - while (ctx.has_more()) { - const size_t pre_loop_cursor = ctx.cursor; + for(;;) { TokenResult res = target.test(ctx); - - // Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching + // Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching // rule succeeded without advancing the buffer index (prevents dynamic parsing lockups). - if (!res.success || ctx.cursor == pre_loop_cursor) { - ctx.cursor = pre_loop_cursor; // Revert cursor to last healthy match checkpoint + if(!res.success || tri == ctx.push()) { + ctx.pop(tri); break; } - - accumulated_match += res.match; + acc += res.match; } - // Repetition rules (* token) always evaluate to successful completion state, even with 0 matches. - return { true, accumulated_match }; + // Repetition rules (* token) always evaluate to successful + // completion state, even with 0 matches. + return { true, acc }; } RepToken RepToken::operator*() { - // Redundant layer trap protection: returning self prevents wrapping a Repetition rule inside a Repetition rule + // Redundant layer trap protection: returning self prevents wrapping + // a Repetition rule inside a Repetition rule return *this; } diff --git a/src/spider/compiler/text/Token.hpp b/src/spider/compiler/text/Token.hpp index 76a4d50..db2ac16 100644 --- a/src/spider/compiler/text/Token.hpp +++ b/src/spider/compiler/text/Token.hpp @@ -3,23 +3,13 @@ #include #include +#include namespace spider { - struct TokenContext { - - std::u32string_view input; - size_t cursor = 0; - - bool has_more() const { return cursor < input.size(); } - char32_t peek() const { return input[cursor]; } - void advance(size_t n = 1) { cursor += n; } - - }; - /** - * @brief The structural payload returned by every parsing component execution. - */ + * @brief The structural payload returned by every parsing component execution. + */ struct TokenResult { /** @brief Indicates if the token composition successfully matched the input boundary. */ @@ -58,7 +48,7 @@ namespace spider { * @return TokenResult Containing verification state and the parsed copy of matching data. * @note Pure virtual; implementation details handle node-specific combinator semantics. */ - virtual TokenResult test(TokenContext& ctx) const = 0; + virtual TokenResult test(TextReader& ctx) const = 0; public: @@ -111,7 +101,7 @@ namespace spider { * @brief Validates match of the backing u32string exactly at the context cursor pointer. * @details Advances the context cursor precisely by literal length on success; zero state mutation on failure. */ - TokenResult test(TokenContext& ctx) const override; + TokenResult test(TextReader& ctx) const override; }; /** @@ -135,7 +125,7 @@ namespace spider { * @details Implements a strict transaction boundary: if any internal element fails, the index * backtracks entirely to its starting cursor value before returning failure. */ - TokenResult test(TokenContext& ctx) const override; + TokenResult test(TextReader& ctx) const override; /** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */ SeqToken operator&(const Token& tok) override; @@ -158,7 +148,7 @@ namespace spider { * @brief Scans through alternatives, resolving immediately on the first candidate that passes. * @details Safely rolls back changes to the context cursor point between failed alternative attempts. */ - TokenResult test(TokenContext& ctx) const override; + TokenResult test(TextReader& ctx) const override; /** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */ OrToken operator|(const Token& tok) override; @@ -181,7 +171,7 @@ namespace spider { * @brief Evaluates target presence. Returns success true regardless of sub-rule evaluation outcome. * @details If the nested rule fails, the context cursor rolls back to initial state, returning empty matches. */ - TokenResult test(TokenContext& ctx) const override; + TokenResult test(TextReader& ctx) const override; /** @brief Stub override providing standard compliance with the base Token interface signature. */ OptToken operator~() override; @@ -205,7 +195,7 @@ namespace spider { * @details Includes internal safety loops checking cursor delta advancement to guarantee infinite * empty-matching sub-loops do not cause thread lockups. */ - TokenResult test(TokenContext& ctx) const override; + TokenResult test(TextReader& ctx) const override; /** @brief Stub override providing standard compliance with the base Token interface signature. */ RepToken operator*() override;