more changes, working on the asm parser

This commit is contained in:
2026-07-07 23:13:11 -06:00
parent ff492fa00f
commit 1fce18460e
5 changed files with 508 additions and 13 deletions
+383
View File
@@ -0,0 +1,383 @@
#include "AsmParser.hpp"
namespace spider {
AsmParser::AsmParser(uptr<TextReader> srcReader)
: reader(std::move(srcReader)) {
}
bool AsmParser::isDigit(u32 ch) const {
return ch >= '0' && ch <= '9';
}
bool AsmParser::isOctalDigit(u32 ch) const {
return ch >= '0' && ch <= '7';
}
bool AsmParser::isBinaryDigit(u32 ch) const {
return ch == '0' || ch == '1';
}
bool AsmParser::isHexDigit(u32 ch) const {
return (ch >= '0' && ch <= '9') ||
(ch >= 'A' && ch <= 'F') ||
(ch >= 'a' && ch <= 'f');
}
bool AsmParser::isLetter(u32 ch) const {
return (ch >= 'a' && ch <= 'z') || (ch >= 'A' && ch <= 'Z');
}
bool AsmParser::isAlphaNum(u32 ch) const {
return isLetter(ch) || isDigit(ch);
}
bool AsmParser::isWhitespaceChar(u32 ch) const {
return ch == ' ' || ch == '\t';
}
void AsmParser::parse_ws_optional() {
while (isWhitespaceChar(reader->current())) {
reader->nextChar();
}
}
bool AsmParser::parse_whitespace() {
if (!isWhitespaceChar(reader->current())) return false;
while (isWhitespaceChar(reader->current())) {
reader->nextChar();
}
return true;
}
bool AsmParser::parse_newline() {
u32 cur = reader->current();
if (cur == '\n') {
reader->nextChar();
return true;
}
if (cur == '\r') {
reader->nextChar();
if (reader->current() == '\n') {
reader->nextChar();
}
return true;
}
return false;
}
bool AsmParser::parse_comment() {
if (!reader->eat(';')) return false;
while (!reader->isEOF() && reader->current() != '\n' && reader->current() != '\r') {
reader->nextChar();
}
return true;
}
bool AsmParser::parse_identifier(std::string& out_id) {
u32 cur = reader->current();
if (!isLetter(cur) && cur != '_') return false;
out_id.clear();
out_id += static_cast<char>(cur);
reader->nextChar();
while (isAlphaNum(reader->current()) || reader->current() == '_') {
out_id += static_cast<char>(reader->current());
reader->nextChar();
}
return true;
}
bool AsmParser::parse_string_lit(std::string& out_str) {
if (!reader->eat('"')) return false;
out_str.clear();
while (!reader->isEOF() && reader->current() != '"') {
if (reader->current() == '\\') { // Escape rules
reader->nextChar();
}
out_str += static_cast<char>(reader->current());
reader->nextChar();
}
return reader->eat('"');
}
bool AsmParser::parse_char_lit(u32& out_char) {
if (!reader->eat('\'')) return false;
if (reader->current() == '\\') {
reader->nextChar(); // Handle escapes
}
out_char = reader->current();
reader->nextChar();
return reader->eat('\'');
}
bool AsmParser::parse_literal() {
reader->commit();
// Try complex prefixes first: 0x (Hex), 0c (Octal), 0b (Binary)
bool has_sign = reader->eat('+') || reader->eat('-');
if (reader->current() == '0') {
u32 prefix = reader->peekChar(1);
if (prefix == 'x' || prefix == 'X') {
reader->nextChar(); reader->nextChar(); // Consume 0x
if (!isHexDigit(reader->current())) { reader->rollback(); return false; }
while (isHexDigit(reader->current())) reader->nextChar();
return true;
}
if (prefix == 'c' || prefix == 'C') {
reader->nextChar(); reader->nextChar();
if (!isOctalDigit(reader->current())) { reader->rollback(); return false; }
while (isOctalDigit(reader->current())) reader->nextChar();
return true;
}
if (prefix == 'b' || prefix == 'B') {
reader->nextChar(); reader->nextChar();
if (!isBinaryDigit(reader->current())) { reader->rollback(); return false; }
while (isBinaryDigit(reader->current())) reader->nextChar();
return true;
}
}
// Reset if pure prefix didn't match to try floats/decimals cleanly
reader->rollback();
reader->commit();
reader->eat('+');
reader->eat('-'); // optional sign
// Float or Decimal
if (isDigit(reader->current()) || reader->current() == '.') {
bool saw_dot = false;
if (reader->eat('.')) saw_dot = true;
if (!isDigit(reader->current()) && saw_dot) { reader->rollback(); return false; }
while (isDigit(reader->current())) reader->nextChar();
if (!saw_dot && reader->eat('.')) {
while (isDigit(reader->current())) reader->nextChar();
saw_dot = true;
}
// Exponent marker
if (reader->current() == 'e' || reader->current() == 'E') {
reader->nextChar();
if (reader->current() == '+' || reader->current() == '-') reader->nextChar();
if (!isDigit(reader->current())) { reader->rollback(); return false; }
while (isDigit(reader->current())) reader->nextChar();
saw_dot = true; // Forcing it to evaluate as float behavior if needed
}
// Width Suffixes
u32 suffix = reader->current();
if (saw_dot) {
if (suffix == 'F' || suffix == 'D') reader->nextChar();
} else {
if (suffix == 'B' || suffix == 'S' || suffix == 'I' || suffix == 'L') reader->nextChar();
}
return true;
}
// Check strings or chars
std::string dummy_str; u32 dummy_ch;
if (parse_string_lit(dummy_str) || parse_char_lit(dummy_ch)) return true;
reader->rollback();
return false;
}
bool AsmParser::parse_literal_decl() {
reader->commit();
u32 cast = reader->current();
if (cast == 'B' || cast == 'S' || cast == 'I' || cast == 'L' || cast == 'F' || cast == 'D') {
if (reader->peekChar(1) == '(' || (isWhitespaceChar(reader->peekChar(1)) && reader->peekChar(2) == '(')) {
reader->nextChar(); // cast width char
parse_ws_optional();
reader->eat('(');
parse_ws_optional();
if (!parse_literal()) { reader->rollback(); return false; }
parse_ws_optional();
if (reader->eat(')')) return true;
reader->rollback();
return false;
}
}
return parse_literal();
}
bool AsmParser::parse_register() {
if (reader->eat(u32('R'))) {
if (isAlphaNum(reader->current())) {
reader->nextChar();
return true;
}
}
return false;
}
bool AsmParser::parse_addr_modes() {
if (!reader->eat(u32('['))) return false;
parse_ws_optional();
reader->commit();
// This parses all the permutations of nested elements inside `[...]` safely via back-tracking
// permutation 1: addrm_ind -> [ literal_decl ]
if (parse_literal_decl()) {
parse_ws_optional();
if (reader->eat(u32(']'))) return true;
}
reader->rollback();
reader->commit();
// Base register is required for remaining components
if (parse_register()) {
parse_ws_optional();
if (reader->eat(u32(']'))) return true; // permutation 2: addrm_ptr -> [ register ]
if (reader->eat('+')) {
parse_ws_optional();
// Could be another register or a literal offset
reader->commit();
if (parse_register()) { // Scale/Displacement modes
parse_ws_optional();
if (reader->eat('*')) {
parse_ws_optional();
if (parse_literal_decl()) {
parse_ws_optional();
if (reader->eat(']')) return true; // permutation 4: addrm_sca
if (reader->eat('+')) {
parse_ws_optional();
if (parse_literal_decl()) {
parse_ws_optional();
if (reader->eat(']')) return true; // permutation 5: addrm_dis
}
}
}
}
}
reader->rollback();
// Fall back into standard index offset: permutation 3: addrm_idx -> [ reg + literal ]
if (parse_literal_decl()) {
parse_ws_optional();
if (reader->eat(']')) return true;
}
}
}
reader->rollback();
return false;
}
bool AsmParser::parse_operand() {
if (parse_register()) return true;
if (parse_addr_modes()) return true;
if (parse_literal_decl()) return true;
std::string dummy_id;
if (parse_identifier(dummy_id)) return true;
return false;
}
// --- Higher Level Statements ---
bool AsmParser::parse_instruction() {
u32 first = reader->current();
if (!isLetter(first)) return false;
// opcode name extraction
while (isAlphaNum(reader->current())) reader->nextChar();
reader->commit();
if (parse_whitespace()) {
if (parse_operand()) {
while (reader->eat(',')) {
parse_ws_optional();
if (!parse_operand()) { reader->rollback(); return false; }
}
return true;
}
reader->rollback(); // No valid operand list followed whitespace
}
return true; // Simple parameterless opcode
}
bool AsmParser::parse_annotation() {
if (!reader->eat('@')) return false;
std::string tag;
if (!parse_identifier(tag)) return false;
if (reader->eat('(')) {
parse_ws_optional();
do {
std::string arg;
if (!parse_identifier(arg)) return false;
parse_ws_optional();
if (reader->eat('=')) {
parse_ws_optional();
if (!parse_literal_decl()) return false;
parse_ws_optional();
}
} while (reader->eat(','));
parse_ws_optional();
if (!reader->eat(')')) return false;
}
return true;
}
bool AsmParser::parse_line_content() {
if (reader->eat("include")) {
if (!parse_whitespace()) return false;
std::string path;
return parse_string_lit(path);
}
if (reader->eat("section")) {
if (!parse_whitespace()) return false;
if (!reader->eat('.')) return false;
std::string sec_name;
return parse_identifier(sec_name);
}
// Main structural execution flow: [annotation] [label] [instruction]
reader->commit();
if (parse_annotation()) {
if (!parse_whitespace()) { reader->rollback(); return false; }
}
std::string lbl;
reader->commit();
if (parse_identifier(lbl)) {
if (reader->eat(':')) {
parse_ws_optional();
} else {
reader->rollback(); // Wasn't a label statement layout
}
}
// Optional structural tailing statement instruction
parse_instruction();
return true;
}
bool AsmParser::parse_program() {
while (!reader->isEOF()) {
parse_ws_optional();
parse_line_content();
parse_ws_optional();
if (reader->eat(';')) {
parse_comment();
}
if (!parse_newline() && !reader->isEOF()) {
// Handle compilation/lex error layout safely
reader->nextChar();
}
}
}
}
+54 -6
View File
@@ -10,17 +10,65 @@ namespace spider {
class AsmParser { class AsmParser {
private: private:
public: uptr<TextReader> reader;
AsmParser();
~AsmParser();
public: public:
AsmParser(uptr<TextReader> srcReader);
private:
bool isDigit(u32 ch) const;
bool isOctalDigit(u32 ch) const;
bool isBinaryDigit(u32 ch) const;
bool isHexDigit(u32 ch) const;
bool isLetter(u32 ch) const;
bool isAlphaNum(u32 ch) const;
bool isWhitespaceChar(u32 ch) const;
public: public:
void ebnf_(); void parse_ws_optional();
bool parse_whitespace();
bool parse_newline();
bool parse_comment();
bool parse_identifier(std::string& out_id);
bool parse_string_lit(std::string& out_str);
bool parse_char_lit(u32& out_char);
bool parse_literal();
bool parse_literal_decl();
// --- Operands & Registers ---
bool parse_register();
bool parse_addr_modes();
bool parse_operand();
// --- Higher Level Statements ---
bool parse_instruction();
bool parse_annotation();
bool parse_line_content();
bool parse_program();
}; };
+32 -3
View File
@@ -16,6 +16,33 @@ namespace spider {
TextReader::~TextReader() {} TextReader::~TextReader() {}
bool TextReader::eat(u32 _char) {
if(current() == _char) {
nextChar();
return true;
}
return false;
}
bool TextReader::eat(char _char) {
return this->eat(u32(_char));
}
bool TextReader::eat(const std::string& chars) {
isize index = 0, count = 0;
u32 _char;
while(index < chars.length()) {
if(!utf8::charAt(chars, index, _char)) return false;
if(_char != peekChar(count)) return false;
count++;
}
if(index == chars.length()) {
nextChar(count);
return true;
}
return false;
}
char TextReader::readByte() { char TextReader::readByte() {
if (err) return 0; if (err) return 0;
auto& s = getStream(); auto& s = getStream();
@@ -49,14 +76,16 @@ namespace spider {
/** /**
* Reads the next character and advances the position tracker. * Reads the next character and advances the position tracker.
*/ */
u32 TextReader::nextChar() { u32 TextReader::nextChar(isize n) {
if (err) return 0; if (err) return 0;
// Ensure the character we are moving TO exists // Ensure the character we are moving TO exists
if (fillBufferTo(1)) { if (fillBufferTo(n)) {
// Track the cursor position using the character we are leaving behind // advance n characters
while(n--) {
advance(current()); advance(current());
bufferIndex++; bufferIndex++;
}
return current(); return current();
} }
+29 -2
View File
@@ -58,6 +58,32 @@ namespace spider {
virtual ~TextReader(); virtual ~TextReader();
public:
/**
* Checks if the current character is
* the one specified. If so, advances
* the index and returns true. Returns
* false otherwise.
*
* Spent character is left in rollback
* buffer.
*/
bool eat(u32 _char);
bool eat(char _char);
/**
* Checks if the current characters are
* the ones specified. If so, advances
* the index and returns true. Returns
* false otherwise.
*
* Spent characters are left in rollback
* buffer.
*/
bool eat(const std::string& chars);
public: public:
/** /**
@@ -66,9 +92,10 @@ namespace spider {
u32 current(); u32 current();
/** /**
* Reads the next character. * Reads the next n-th character.
* n = 0 is a noop, since it's the current one.
*/ */
u32 nextChar(); u32 nextChar(isize n = 1);
/** /**
* Keeps the next n-th character * Keeps the next n-th character
+8
View File
@@ -86,6 +86,14 @@ namespace spider {
return out; return out;
} }
inline bool charAt(const std::string& str, isize& index, u32& out) {
u8 ch0 = u8(str[index]);
isize chlen = isValidSeq(str.c_str(), str.size());
if(chlen == 0) return false;
out = decodeArr(str.c_str(), chlen);
return true;
}
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {} inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {}
} }