diff --git a/src/spider/compiler/assembler/AsmParser.cpp b/src/spider/compiler/assembler/AsmParser.cpp new file mode 100644 index 0000000..86be82b --- /dev/null +++ b/src/spider/compiler/assembler/AsmParser.cpp @@ -0,0 +1,383 @@ +#include "AsmParser.hpp" + +namespace spider { + + AsmParser::AsmParser(uptr srcReader) + : reader(std::move(srcReader)) { + } + + bool AsmParser::isDigit(u32 ch) const { + return ch >= '0' && ch <= '9'; + } + + bool AsmParser::isOctalDigit(u32 ch) const { + return ch >= '0' && ch <= '7'; + } + + bool AsmParser::isBinaryDigit(u32 ch) const { + return ch == '0' || ch == '1'; + } + + bool AsmParser::isHexDigit(u32 ch) const { + return (ch >= '0' && ch <= '9') || + (ch >= 'A' && ch <= 'F') || + (ch >= 'a' && ch <= 'f'); + } + + bool AsmParser::isLetter(u32 ch) const { + return (ch >= 'a' && ch <= 'z') || (ch >= 'A' && ch <= 'Z'); + } + + bool AsmParser::isAlphaNum(u32 ch) const { + return isLetter(ch) || isDigit(ch); + } + + bool AsmParser::isWhitespaceChar(u32 ch) const { + return ch == ' ' || ch == '\t'; + } + + void AsmParser::parse_ws_optional() { + while (isWhitespaceChar(reader->current())) { + reader->nextChar(); + } + } + + bool AsmParser::parse_whitespace() { + if (!isWhitespaceChar(reader->current())) return false; + while (isWhitespaceChar(reader->current())) { + reader->nextChar(); + } + return true; + } + + bool AsmParser::parse_newline() { + u32 cur = reader->current(); + if (cur == '\n') { + reader->nextChar(); + return true; + } + if (cur == '\r') { + reader->nextChar(); + if (reader->current() == '\n') { + reader->nextChar(); + } + return true; + } + return false; + } + + bool AsmParser::parse_comment() { + if (!reader->eat(';')) return false; + while (!reader->isEOF() && reader->current() != '\n' && reader->current() != '\r') { + reader->nextChar(); + } + return true; + } + + bool AsmParser::parse_identifier(std::string& out_id) { + u32 cur = reader->current(); + if (!isLetter(cur) && cur != '_') return false; + + out_id.clear(); + out_id += static_cast(cur); + reader->nextChar(); + + while (isAlphaNum(reader->current()) || reader->current() == '_') { + out_id += static_cast(reader->current()); + reader->nextChar(); + } + return true; + } + + bool AsmParser::parse_string_lit(std::string& out_str) { + if (!reader->eat('"')) return false; + out_str.clear(); + while (!reader->isEOF() && reader->current() != '"') { + if (reader->current() == '\\') { // Escape rules + reader->nextChar(); + } + out_str += static_cast(reader->current()); + reader->nextChar(); + } + return reader->eat('"'); + } + + bool AsmParser::parse_char_lit(u32& out_char) { + if (!reader->eat('\'')) return false; + if (reader->current() == '\\') { + reader->nextChar(); // Handle escapes + } + out_char = reader->current(); + reader->nextChar(); + return reader->eat('\''); + } + + bool AsmParser::parse_literal() { + reader->commit(); + + // Try complex prefixes first: 0x (Hex), 0c (Octal), 0b (Binary) + bool has_sign = reader->eat('+') || reader->eat('-'); + if (reader->current() == '0') { + u32 prefix = reader->peekChar(1); + if (prefix == 'x' || prefix == 'X') { + reader->nextChar(); reader->nextChar(); // Consume 0x + if (!isHexDigit(reader->current())) { reader->rollback(); return false; } + while (isHexDigit(reader->current())) reader->nextChar(); + return true; + } + if (prefix == 'c' || prefix == 'C') { + reader->nextChar(); reader->nextChar(); + if (!isOctalDigit(reader->current())) { reader->rollback(); return false; } + while (isOctalDigit(reader->current())) reader->nextChar(); + return true; + } + if (prefix == 'b' || prefix == 'B') { + reader->nextChar(); reader->nextChar(); + if (!isBinaryDigit(reader->current())) { reader->rollback(); return false; } + while (isBinaryDigit(reader->current())) reader->nextChar(); + return true; + } + } + + // Reset if pure prefix didn't match to try floats/decimals cleanly + reader->rollback(); + reader->commit(); + + reader->eat('+'); + reader->eat('-'); // optional sign + + // Float or Decimal + if (isDigit(reader->current()) || reader->current() == '.') { + bool saw_dot = false; + if (reader->eat('.')) saw_dot = true; + + if (!isDigit(reader->current()) && saw_dot) { reader->rollback(); return false; } + while (isDigit(reader->current())) reader->nextChar(); + + if (!saw_dot && reader->eat('.')) { + while (isDigit(reader->current())) reader->nextChar(); + saw_dot = true; + } + + // Exponent marker + if (reader->current() == 'e' || reader->current() == 'E') { + reader->nextChar(); + if (reader->current() == '+' || reader->current() == '-') reader->nextChar(); + if (!isDigit(reader->current())) { reader->rollback(); return false; } + while (isDigit(reader->current())) reader->nextChar(); + saw_dot = true; // Forcing it to evaluate as float behavior if needed + } + + // Width Suffixes + u32 suffix = reader->current(); + if (saw_dot) { + if (suffix == 'F' || suffix == 'D') reader->nextChar(); + } else { + if (suffix == 'B' || suffix == 'S' || suffix == 'I' || suffix == 'L') reader->nextChar(); + } + return true; + } + + // Check strings or chars + std::string dummy_str; u32 dummy_ch; + if (parse_string_lit(dummy_str) || parse_char_lit(dummy_ch)) return true; + + reader->rollback(); + return false; + } + + bool AsmParser::parse_literal_decl() { + reader->commit(); + u32 cast = reader->current(); + if (cast == 'B' || cast == 'S' || cast == 'I' || cast == 'L' || cast == 'F' || cast == 'D') { + if (reader->peekChar(1) == '(' || (isWhitespaceChar(reader->peekChar(1)) && reader->peekChar(2) == '(')) { + reader->nextChar(); // cast width char + parse_ws_optional(); + reader->eat('('); + parse_ws_optional(); + if (!parse_literal()) { reader->rollback(); return false; } + parse_ws_optional(); + if (reader->eat(')')) return true; + reader->rollback(); + return false; + } + } + return parse_literal(); + } + + bool AsmParser::parse_register() { + if (reader->eat(u32('R'))) { + if (isAlphaNum(reader->current())) { + reader->nextChar(); + return true; + } + } + return false; + } + + bool AsmParser::parse_addr_modes() { + if (!reader->eat(u32('['))) return false; + parse_ws_optional(); + + reader->commit(); + + // This parses all the permutations of nested elements inside `[...]` safely via back-tracking + // permutation 1: addrm_ind -> [ literal_decl ] + if (parse_literal_decl()) { + parse_ws_optional(); + if (reader->eat(u32(']'))) return true; + } + + reader->rollback(); + reader->commit(); + + // Base register is required for remaining components + if (parse_register()) { + parse_ws_optional(); + if (reader->eat(u32(']'))) return true; // permutation 2: addrm_ptr -> [ register ] + + if (reader->eat('+')) { + parse_ws_optional(); + + // Could be another register or a literal offset + reader->commit(); + if (parse_register()) { // Scale/Displacement modes + parse_ws_optional(); + if (reader->eat('*')) { + parse_ws_optional(); + if (parse_literal_decl()) { + parse_ws_optional(); + if (reader->eat(']')) return true; // permutation 4: addrm_sca + if (reader->eat('+')) { + parse_ws_optional(); + if (parse_literal_decl()) { + parse_ws_optional(); + if (reader->eat(']')) return true; // permutation 5: addrm_dis + } + } + } + } + } + reader->rollback(); + + // Fall back into standard index offset: permutation 3: addrm_idx -> [ reg + literal ] + if (parse_literal_decl()) { + parse_ws_optional(); + if (reader->eat(']')) return true; + } + } + } + + reader->rollback(); + return false; + } + + bool AsmParser::parse_operand() { + if (parse_register()) return true; + if (parse_addr_modes()) return true; + if (parse_literal_decl()) return true; + std::string dummy_id; + if (parse_identifier(dummy_id)) return true; + return false; + } + + // --- Higher Level Statements --- + + bool AsmParser::parse_instruction() { + u32 first = reader->current(); + if (!isLetter(first)) return false; + + // opcode name extraction + while (isAlphaNum(reader->current())) reader->nextChar(); + + reader->commit(); + if (parse_whitespace()) { + if (parse_operand()) { + while (reader->eat(',')) { + parse_ws_optional(); + if (!parse_operand()) { reader->rollback(); return false; } + } + return true; + } + reader->rollback(); // No valid operand list followed whitespace + } + return true; // Simple parameterless opcode + } + + bool AsmParser::parse_annotation() { + if (!reader->eat('@')) return false; + std::string tag; + if (!parse_identifier(tag)) return false; + + if (reader->eat('(')) { + parse_ws_optional(); + do { + std::string arg; + if (!parse_identifier(arg)) return false; + parse_ws_optional(); + if (reader->eat('=')) { + parse_ws_optional(); + if (!parse_literal_decl()) return false; + parse_ws_optional(); + } + } while (reader->eat(',')); + parse_ws_optional(); + if (!reader->eat(')')) return false; + } + return true; + } + + bool AsmParser::parse_line_content() { + if (reader->eat("include")) { + if (!parse_whitespace()) return false; + std::string path; + return parse_string_lit(path); + } + + if (reader->eat("section")) { + if (!parse_whitespace()) return false; + if (!reader->eat('.')) return false; + std::string sec_name; + return parse_identifier(sec_name); + } + + // Main structural execution flow: [annotation] [label] [instruction] + reader->commit(); + if (parse_annotation()) { + if (!parse_whitespace()) { reader->rollback(); return false; } + } + + std::string lbl; + reader->commit(); + if (parse_identifier(lbl)) { + if (reader->eat(':')) { + parse_ws_optional(); + } else { + reader->rollback(); // Wasn't a label statement layout + } + } + + // Optional structural tailing statement instruction + parse_instruction(); + return true; + } + + bool AsmParser::parse_program() { + while (!reader->isEOF()) { + parse_ws_optional(); + + parse_line_content(); + + parse_ws_optional(); + if (reader->eat(';')) { + parse_comment(); + } + + if (!parse_newline() && !reader->isEOF()) { + // Handle compilation/lex error layout safely + reader->nextChar(); + } + } + } + +} diff --git a/src/spider/compiler/assembler/AsmParser.hpp b/src/spider/compiler/assembler/AsmParser.hpp index 0f3db2b..3f06c67 100644 --- a/src/spider/compiler/assembler/AsmParser.hpp +++ b/src/spider/compiler/assembler/AsmParser.hpp @@ -10,17 +10,65 @@ namespace spider { class AsmParser { private: - public: - - AsmParser(); - - ~AsmParser(); + uptr reader; public: + AsmParser(uptr srcReader); + + private: + + bool isDigit(u32 ch) const; + + bool isOctalDigit(u32 ch) const; + + bool isBinaryDigit(u32 ch) const; + + bool isHexDigit(u32 ch) const; + + bool isLetter(u32 ch) const; + + bool isAlphaNum(u32 ch) const; + + bool isWhitespaceChar(u32 ch) const; + public: - void ebnf_(); + void parse_ws_optional(); + + bool parse_whitespace(); + + bool parse_newline(); + + bool parse_comment(); + + bool parse_identifier(std::string& out_id); + + bool parse_string_lit(std::string& out_str); + + bool parse_char_lit(u32& out_char); + + bool parse_literal(); + + bool parse_literal_decl(); + + // --- Operands & Registers --- + + bool parse_register(); + + bool parse_addr_modes(); + + bool parse_operand(); + + // --- Higher Level Statements --- + + bool parse_instruction(); + + bool parse_annotation(); + + bool parse_line_content(); + + bool parse_program(); }; diff --git a/src/spider/compiler/text/TextReader.cpp b/src/spider/compiler/text/TextReader.cpp index 9286af2..939d5f4 100644 --- a/src/spider/compiler/text/TextReader.cpp +++ b/src/spider/compiler/text/TextReader.cpp @@ -16,6 +16,33 @@ namespace spider { TextReader::~TextReader() {} + bool TextReader::eat(u32 _char) { + if(current() == _char) { + nextChar(); + return true; + } + return false; + } + + bool TextReader::eat(char _char) { + return this->eat(u32(_char)); + } + + bool TextReader::eat(const std::string& chars) { + isize index = 0, count = 0; + u32 _char; + while(index < chars.length()) { + if(!utf8::charAt(chars, index, _char)) return false; + if(_char != peekChar(count)) return false; + count++; + } + if(index == chars.length()) { + nextChar(count); + return true; + } + return false; + } + char TextReader::readByte() { if (err) return 0; auto& s = getStream(); @@ -49,14 +76,16 @@ namespace spider { /** * Reads the next character and advances the position tracker. */ - u32 TextReader::nextChar() { + u32 TextReader::nextChar(isize n) { if (err) return 0; // Ensure the character we are moving TO exists - if (fillBufferTo(1)) { - // Track the cursor position using the character we are leaving behind - advance(current()); - bufferIndex++; + if (fillBufferTo(n)) { + // advance n characters + while(n--) { + advance(current()); + bufferIndex++; + } return current(); } diff --git a/src/spider/compiler/text/TextReader.hpp b/src/spider/compiler/text/TextReader.hpp index 336a795..3f5c882 100644 --- a/src/spider/compiler/text/TextReader.hpp +++ b/src/spider/compiler/text/TextReader.hpp @@ -58,6 +58,32 @@ namespace spider { virtual ~TextReader(); + public: + + /** + * Checks if the current character is + * the one specified. If so, advances + * the index and returns true. Returns + * false otherwise. + * + * Spent character is left in rollback + * buffer. + */ + bool eat(u32 _char); + + bool eat(char _char); + + /** + * Checks if the current characters are + * the ones specified. If so, advances + * the index and returns true. Returns + * false otherwise. + * + * Spent characters are left in rollback + * buffer. + */ + bool eat(const std::string& chars); + public: /** @@ -66,9 +92,10 @@ namespace spider { u32 current(); /** - * Reads the next character. + * Reads the next n-th character. + * n = 0 is a noop, since it's the current one. */ - u32 nextChar(); + u32 nextChar(isize n = 1); /** * Keeps the next n-th character diff --git a/src/spider/compiler/text/utf8.hpp b/src/spider/compiler/text/utf8.hpp index dbabed8..e535016 100644 --- a/src/spider/compiler/text/utf8.hpp +++ b/src/spider/compiler/text/utf8.hpp @@ -86,6 +86,14 @@ namespace spider { return out; } + inline bool charAt(const std::string& str, isize& index, u32& out) { + u8 ch0 = u8(str[index]); + isize chlen = isValidSeq(str.c_str(), str.size()); + if(chlen == 0) return false; + out = decodeArr(str.c_str(), chlen); + return true; + } + inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {} }