From 29784ea152700864365d387299d67378328d89c9 Mon Sep 17 00:00:00 2001 From: Kittycannon Date: Sun, 19 Jul 2026 13:45:05 -0600 Subject: [PATCH] compiles! Still working on tokens --- .vscode/settings.json | 3 + run.sh | 35 ++ src/spider/compiler/Compiler.cpp | 3 +- src/spider/compiler/assembler/AsmParser.cpp | 383 -------------------- src/spider/compiler/assembler/AsmParser.hpp | 75 ---- src/spider/compiler/assembler/Assembler.cpp | 10 +- src/spider/compiler/assembler/Assembler.hpp | 2 - src/spider/compiler/text/TextReader.cpp | 2 +- src/spider/compiler/text/Token.cpp | 0 src/spider/compiler/text/Token.hpp | 162 ++++++++- src/spider/compiler/text/utf8.hpp | 16 +- 11 files changed, 204 insertions(+), 487 deletions(-) create mode 100644 .vscode/settings.json create mode 100644 run.sh delete mode 100644 src/spider/compiler/assembler/AsmParser.cpp delete mode 100644 src/spider/compiler/assembler/AsmParser.hpp create mode 100644 src/spider/compiler/text/Token.cpp diff --git a/.vscode/settings.json b/.vscode/settings.json new file mode 100644 index 0000000..02928bb --- /dev/null +++ b/.vscode/settings.json @@ -0,0 +1,3 @@ +{ + "C_Cpp.default.compilerPath": "C:/msys64/ucrt64/bin/g++.exe" +} \ No newline at end of file diff --git a/run.sh b/run.sh new file mode 100644 index 0000000..4cd8284 --- /dev/null +++ b/run.sh @@ -0,0 +1,35 @@ +#!/usr/bin/env bash + +BOLD="\033[1m" +RESET="\033[0m" + +SUCCESS="\033[38;2;80;220;100m" +FAIL="\033[38;2;255;90;90m" +INFO="\033[38;2;100;200;255m" +DIM="\033[2m" + +LOG="./out/out.log" + +# Start +printf "${DIM}────────────────────────────────────────${RESET}\n" +printf "${INFO}${BOLD}> Running...${RESET}\n" +printf "${DIM}────────────────────────────────────────${RESET}\n\n" + +# Send both stdout and stderr to the console and the log file. +start=$(date +%s.%N) +./out/out "$@" 2>&1 | tee "$LOG" +status=${PIPESTATUS[0]} +end=$(date +%s.%N) +elapsed=$(awk "BEGIN { printf \"%.3f\", $end - $start }") + +# Status! +printf "\n${DIM}────────────────────────────────────────${RESET}\n" +if (( status == 0 )); then + printf "${SUCCESS}${BOLD}✓ Success${RESET} (exit %d)\n" "$status" +else + printf "${FAIL}${BOLD}✗ Failed${RESET} (exit %d)\n" "$status" +fi +printf "${INFO}${BOLD}> Time${RESET} ${elapsed}s\n" +printf "${INFO}${BOLD}> Log${RESET} %s\n" "$LOG" +printf "${DIM}────────────────────────────────────────${RESET}" +exit "$status" diff --git a/src/spider/compiler/Compiler.cpp b/src/spider/compiler/Compiler.cpp index 3fc1059..6654941 100644 --- a/src/spider/compiler/Compiler.cpp +++ b/src/spider/compiler/Compiler.cpp @@ -1,4 +1,4 @@ -#include "spider/compiler/assembler/AsmParser.hpp" +#include namespace spider { @@ -7,5 +7,6 @@ namespace spider { } int main() { + std::cout << "Happy Day!" << std::endl; return 0; } diff --git a/src/spider/compiler/assembler/AsmParser.cpp b/src/spider/compiler/assembler/AsmParser.cpp deleted file mode 100644 index 86be82b..0000000 --- a/src/spider/compiler/assembler/AsmParser.cpp +++ /dev/null @@ -1,383 +0,0 @@ -#include "AsmParser.hpp" - -namespace spider { - - AsmParser::AsmParser(uptr srcReader) - : reader(std::move(srcReader)) { - } - - bool AsmParser::isDigit(u32 ch) const { - return ch >= '0' && ch <= '9'; - } - - bool AsmParser::isOctalDigit(u32 ch) const { - return ch >= '0' && ch <= '7'; - } - - bool AsmParser::isBinaryDigit(u32 ch) const { - return ch == '0' || ch == '1'; - } - - bool AsmParser::isHexDigit(u32 ch) const { - return (ch >= '0' && ch <= '9') || - (ch >= 'A' && ch <= 'F') || - (ch >= 'a' && ch <= 'f'); - } - - bool AsmParser::isLetter(u32 ch) const { - return (ch >= 'a' && ch <= 'z') || (ch >= 'A' && ch <= 'Z'); - } - - bool AsmParser::isAlphaNum(u32 ch) const { - return isLetter(ch) || isDigit(ch); - } - - bool AsmParser::isWhitespaceChar(u32 ch) const { - return ch == ' ' || ch == '\t'; - } - - void AsmParser::parse_ws_optional() { - while (isWhitespaceChar(reader->current())) { - reader->nextChar(); - } - } - - bool AsmParser::parse_whitespace() { - if (!isWhitespaceChar(reader->current())) return false; - while (isWhitespaceChar(reader->current())) { - reader->nextChar(); - } - return true; - } - - bool AsmParser::parse_newline() { - u32 cur = reader->current(); - if (cur == '\n') { - reader->nextChar(); - return true; - } - if (cur == '\r') { - reader->nextChar(); - if (reader->current() == '\n') { - reader->nextChar(); - } - return true; - } - return false; - } - - bool AsmParser::parse_comment() { - if (!reader->eat(';')) return false; - while (!reader->isEOF() && reader->current() != '\n' && reader->current() != '\r') { - reader->nextChar(); - } - return true; - } - - bool AsmParser::parse_identifier(std::string& out_id) { - u32 cur = reader->current(); - if (!isLetter(cur) && cur != '_') return false; - - out_id.clear(); - out_id += static_cast(cur); - reader->nextChar(); - - while (isAlphaNum(reader->current()) || reader->current() == '_') { - out_id += static_cast(reader->current()); - reader->nextChar(); - } - return true; - } - - bool AsmParser::parse_string_lit(std::string& out_str) { - if (!reader->eat('"')) return false; - out_str.clear(); - while (!reader->isEOF() && reader->current() != '"') { - if (reader->current() == '\\') { // Escape rules - reader->nextChar(); - } - out_str += static_cast(reader->current()); - reader->nextChar(); - } - return reader->eat('"'); - } - - bool AsmParser::parse_char_lit(u32& out_char) { - if (!reader->eat('\'')) return false; - if (reader->current() == '\\') { - reader->nextChar(); // Handle escapes - } - out_char = reader->current(); - reader->nextChar(); - return reader->eat('\''); - } - - bool AsmParser::parse_literal() { - reader->commit(); - - // Try complex prefixes first: 0x (Hex), 0c (Octal), 0b (Binary) - bool has_sign = reader->eat('+') || reader->eat('-'); - if (reader->current() == '0') { - u32 prefix = reader->peekChar(1); - if (prefix == 'x' || prefix == 'X') { - reader->nextChar(); reader->nextChar(); // Consume 0x - if (!isHexDigit(reader->current())) { reader->rollback(); return false; } - while (isHexDigit(reader->current())) reader->nextChar(); - return true; - } - if (prefix == 'c' || prefix == 'C') { - reader->nextChar(); reader->nextChar(); - if (!isOctalDigit(reader->current())) { reader->rollback(); return false; } - while (isOctalDigit(reader->current())) reader->nextChar(); - return true; - } - if (prefix == 'b' || prefix == 'B') { - reader->nextChar(); reader->nextChar(); - if (!isBinaryDigit(reader->current())) { reader->rollback(); return false; } - while (isBinaryDigit(reader->current())) reader->nextChar(); - return true; - } - } - - // Reset if pure prefix didn't match to try floats/decimals cleanly - reader->rollback(); - reader->commit(); - - reader->eat('+'); - reader->eat('-'); // optional sign - - // Float or Decimal - if (isDigit(reader->current()) || reader->current() == '.') { - bool saw_dot = false; - if (reader->eat('.')) saw_dot = true; - - if (!isDigit(reader->current()) && saw_dot) { reader->rollback(); return false; } - while (isDigit(reader->current())) reader->nextChar(); - - if (!saw_dot && reader->eat('.')) { - while (isDigit(reader->current())) reader->nextChar(); - saw_dot = true; - } - - // Exponent marker - if (reader->current() == 'e' || reader->current() == 'E') { - reader->nextChar(); - if (reader->current() == '+' || reader->current() == '-') reader->nextChar(); - if (!isDigit(reader->current())) { reader->rollback(); return false; } - while (isDigit(reader->current())) reader->nextChar(); - saw_dot = true; // Forcing it to evaluate as float behavior if needed - } - - // Width Suffixes - u32 suffix = reader->current(); - if (saw_dot) { - if (suffix == 'F' || suffix == 'D') reader->nextChar(); - } else { - if (suffix == 'B' || suffix == 'S' || suffix == 'I' || suffix == 'L') reader->nextChar(); - } - return true; - } - - // Check strings or chars - std::string dummy_str; u32 dummy_ch; - if (parse_string_lit(dummy_str) || parse_char_lit(dummy_ch)) return true; - - reader->rollback(); - return false; - } - - bool AsmParser::parse_literal_decl() { - reader->commit(); - u32 cast = reader->current(); - if (cast == 'B' || cast == 'S' || cast == 'I' || cast == 'L' || cast == 'F' || cast == 'D') { - if (reader->peekChar(1) == '(' || (isWhitespaceChar(reader->peekChar(1)) && reader->peekChar(2) == '(')) { - reader->nextChar(); // cast width char - parse_ws_optional(); - reader->eat('('); - parse_ws_optional(); - if (!parse_literal()) { reader->rollback(); return false; } - parse_ws_optional(); - if (reader->eat(')')) return true; - reader->rollback(); - return false; - } - } - return parse_literal(); - } - - bool AsmParser::parse_register() { - if (reader->eat(u32('R'))) { - if (isAlphaNum(reader->current())) { - reader->nextChar(); - return true; - } - } - return false; - } - - bool AsmParser::parse_addr_modes() { - if (!reader->eat(u32('['))) return false; - parse_ws_optional(); - - reader->commit(); - - // This parses all the permutations of nested elements inside `[...]` safely via back-tracking - // permutation 1: addrm_ind -> [ literal_decl ] - if (parse_literal_decl()) { - parse_ws_optional(); - if (reader->eat(u32(']'))) return true; - } - - reader->rollback(); - reader->commit(); - - // Base register is required for remaining components - if (parse_register()) { - parse_ws_optional(); - if (reader->eat(u32(']'))) return true; // permutation 2: addrm_ptr -> [ register ] - - if (reader->eat('+')) { - parse_ws_optional(); - - // Could be another register or a literal offset - reader->commit(); - if (parse_register()) { // Scale/Displacement modes - parse_ws_optional(); - if (reader->eat('*')) { - parse_ws_optional(); - if (parse_literal_decl()) { - parse_ws_optional(); - if (reader->eat(']')) return true; // permutation 4: addrm_sca - if (reader->eat('+')) { - parse_ws_optional(); - if (parse_literal_decl()) { - parse_ws_optional(); - if (reader->eat(']')) return true; // permutation 5: addrm_dis - } - } - } - } - } - reader->rollback(); - - // Fall back into standard index offset: permutation 3: addrm_idx -> [ reg + literal ] - if (parse_literal_decl()) { - parse_ws_optional(); - if (reader->eat(']')) return true; - } - } - } - - reader->rollback(); - return false; - } - - bool AsmParser::parse_operand() { - if (parse_register()) return true; - if (parse_addr_modes()) return true; - if (parse_literal_decl()) return true; - std::string dummy_id; - if (parse_identifier(dummy_id)) return true; - return false; - } - - // --- Higher Level Statements --- - - bool AsmParser::parse_instruction() { - u32 first = reader->current(); - if (!isLetter(first)) return false; - - // opcode name extraction - while (isAlphaNum(reader->current())) reader->nextChar(); - - reader->commit(); - if (parse_whitespace()) { - if (parse_operand()) { - while (reader->eat(',')) { - parse_ws_optional(); - if (!parse_operand()) { reader->rollback(); return false; } - } - return true; - } - reader->rollback(); // No valid operand list followed whitespace - } - return true; // Simple parameterless opcode - } - - bool AsmParser::parse_annotation() { - if (!reader->eat('@')) return false; - std::string tag; - if (!parse_identifier(tag)) return false; - - if (reader->eat('(')) { - parse_ws_optional(); - do { - std::string arg; - if (!parse_identifier(arg)) return false; - parse_ws_optional(); - if (reader->eat('=')) { - parse_ws_optional(); - if (!parse_literal_decl()) return false; - parse_ws_optional(); - } - } while (reader->eat(',')); - parse_ws_optional(); - if (!reader->eat(')')) return false; - } - return true; - } - - bool AsmParser::parse_line_content() { - if (reader->eat("include")) { - if (!parse_whitespace()) return false; - std::string path; - return parse_string_lit(path); - } - - if (reader->eat("section")) { - if (!parse_whitespace()) return false; - if (!reader->eat('.')) return false; - std::string sec_name; - return parse_identifier(sec_name); - } - - // Main structural execution flow: [annotation] [label] [instruction] - reader->commit(); - if (parse_annotation()) { - if (!parse_whitespace()) { reader->rollback(); return false; } - } - - std::string lbl; - reader->commit(); - if (parse_identifier(lbl)) { - if (reader->eat(':')) { - parse_ws_optional(); - } else { - reader->rollback(); // Wasn't a label statement layout - } - } - - // Optional structural tailing statement instruction - parse_instruction(); - return true; - } - - bool AsmParser::parse_program() { - while (!reader->isEOF()) { - parse_ws_optional(); - - parse_line_content(); - - parse_ws_optional(); - if (reader->eat(';')) { - parse_comment(); - } - - if (!parse_newline() && !reader->isEOF()) { - // Handle compilation/lex error layout safely - reader->nextChar(); - } - } - } - -} diff --git a/src/spider/compiler/assembler/AsmParser.hpp b/src/spider/compiler/assembler/AsmParser.hpp deleted file mode 100644 index b5b07a9..0000000 --- a/src/spider/compiler/assembler/AsmParser.hpp +++ /dev/null @@ -1,75 +0,0 @@ -#pragma once - -#include - -namespace spider { - - /** - * EBNF Parser for the Spider Assembly. - */ - class AsmParser { - private: - - uptr reader; - - public: - - AsmParser(uptr srcReader); - - private: - - bool isDigit(u32 ch) const; - - bool isOctalDigit(u32 ch) const; - - bool isBinaryDigit(u32 ch) const; - - bool isHexDigit(u32 ch) const; - - bool isLetter(u32 ch) const; - - bool isAlphaNum(u32 ch) const; - - bool isWhitespaceChar(u32 ch) const; - - public: - - void parse_ws_optional(); - - bool parse_whitespace(); - - bool parse_newline(); - - bool parse_comment(); - - bool parse_identifier(std::u32string& out_id); - - bool parse_string_lit(std::u32string& out_str); - - bool parse_char_lit(u32& out_char); - - bool parse_literal(); - - bool parse_literal_decl(); - - // --- Operands & Registers --- - - bool parse_register(); - - bool parse_addr_modes(); - - bool parse_operand(); - - // --- Higher Level Statements --- - - bool parse_instruction(); - - bool parse_annotation(); - - bool parse_line_content(); - - bool parse_program(); - - }; - -} diff --git a/src/spider/compiler/assembler/Assembler.cpp b/src/spider/compiler/assembler/Assembler.cpp index 7fa45fc..e6b4c25 100644 --- a/src/spider/compiler/assembler/Assembler.cpp +++ b/src/spider/compiler/assembler/Assembler.cpp @@ -16,10 +16,10 @@ namespace spider { auto ir = fstack.insert(abs_path); // Actually load! - levels.emplace_back(Level { - .reader = std::make_unique(new FileTextReader(abs_path.string())), - .source = abs_path.string(), - }); + //levels.emplace_back(Level { + // .reader = std::make_unique(new FileTextReader(abs_path.string())), + // .source = abs_path.string(), + //}); parseCurrentLevel(); // alright! @@ -28,7 +28,7 @@ namespace spider { } void Assembler::parseCurrentLevel() { - auto& lvl = levels.back(); + //auto& lvl = levels.back(); } diff --git a/src/spider/compiler/assembler/Assembler.hpp b/src/spider/compiler/assembler/Assembler.hpp index 6534f29..5ffa591 100644 --- a/src/spider/compiler/assembler/Assembler.hpp +++ b/src/spider/compiler/assembler/Assembler.hpp @@ -2,7 +2,6 @@ #include -#include #include namespace spider { @@ -21,7 +20,6 @@ namespace spider { public: set fstack; - CompToken root; public: diff --git a/src/spider/compiler/text/TextReader.cpp b/src/spider/compiler/text/TextReader.cpp index 939d5f4..599b735 100644 --- a/src/spider/compiler/text/TextReader.cpp +++ b/src/spider/compiler/text/TextReader.cpp @@ -109,7 +109,7 @@ namespace spider { void TextReader::commit() { if (bufferIndex > 0) { // Erase everything before the current buffer index - buffer.erase(buffer.begin(), buffer.begin() + bufferIndex); + buffer.erase(buffer.begin(), buffer.begin() + std::ptrdiff_t(bufferIndex)); bufferIndex = 0; } } diff --git a/src/spider/compiler/text/Token.cpp b/src/spider/compiler/text/Token.cpp new file mode 100644 index 0000000..e69de29 diff --git a/src/spider/compiler/text/Token.hpp b/src/spider/compiler/text/Token.hpp index d6f9aee..c7ce813 100644 --- a/src/spider/compiler/text/Token.hpp +++ b/src/spider/compiler/text/Token.hpp @@ -1,33 +1,157 @@ #pragma once -#include +#include "spider/compiler/common.hpp" -#include +#include "spider/compiler/text/utf8.hpp" namespace spider { - struct LitToken { - public: - const range at; - const std::u32string type; - const std::u32string value; + struct TokenContext { + + std::u32string_view input; + size_t cursor = 0; + + bool has_more() const { return cursor < input.size(); } + + char32_t peek() const { return input[cursor]; } + + void advance(size_t n = 1) { cursor += n; } + }; - struct SimpleToken { - public: - const range at; - const std::u32string type; + struct TokenResult { + bool success; + std::u32string match; }; - struct CompToken; - - using Token = std::variant; - - struct CompToken { + class Token { public: - const range at; - const std::u32string type; - vector inners; + + virtual ~Token() = default; + + virtual TokenResult parse(TokenContext& ctx) const = 0; + + }; + + class LitToken : public Token { + private: + + std::u32string literal; + + public: + + explicit LitToken(std::string_view lit) { + if(!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!"); + } + + explicit LitToken(std::u32string lit) : literal(std::move(lit)) {} + + public: + + TokenResult parse(TokenContext& ctx) const override { + if (ctx.cursor + literal.size() > ctx.input.size()) return { false, {} }; + std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size()); + if (sub == literal) { + ctx.advance(literal.size()); + return { true, std::u32string(sub) }; + } + return { false, {} }; + } + + }; + + // AndToken references existing static tokens rather than managing their lifecycles + class AndToken : public Token { + private: + + const Token& lhs; + const Token& rhs; + + public: + + AndToken(const Token& l, const Token& r) : lhs(l), rhs(r) {} + + public: + + TokenResult parse(TokenContext& ctx) const override { + size_t start_pos = ctx.cursor; + if (!lhs.parse(ctx).success) return { false, {} }; + if (!rhs.parse(ctx).success) { + ctx.cursor = start_pos; // Backtrack + return { false, {} }; + } + return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) }; + } + }; + + class OrToken : public Token { + private: + + const Token& lhs; + const Token& rhs; + + public: + + OrToken(const Token& l, const Token& r) : lhs(l), rhs(r) {} + + public: + + TokenResult parse(TokenContext& ctx) const override { + size_t start_pos = ctx.cursor; + auto res = lhs.parse(ctx); + if (res.success) return res; + ctx.cursor = start_pos; // Backtrack + return rhs.parse(ctx); + } + + }; + + class OptToken : public Token { + private: + + const Token& target; + + public: + + explicit OptToken(const Token& t) : target(t) {} + + public: + + TokenResult parse(TokenContext& ctx) const override { + size_t start_pos = ctx.cursor; + if (target.parse(ctx).success) return { + true, + std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) + }; + ctx.cursor = start_pos; + return { true, std::u32string(ctx.input.substr(start_pos, 0)) }; + } + + }; + + class RepToken : public Token { + private: + + const Token& target; + + public: + + explicit RepToken(const Token& t) : target(t) {} + + public: + + TokenResult parse(TokenContext& ctx) const override { + size_t start_pos = ctx.cursor; + while (ctx.has_more()) { + size_t loop_start = ctx.cursor; + if (!target.parse(ctx).success || ctx.cursor == loop_start) { + ctx.cursor = loop_start; + break; + } + } + return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) }; + } + }; } diff --git a/src/spider/compiler/text/utf8.hpp b/src/spider/compiler/text/utf8.hpp index e535016..7e1a896 100644 --- a/src/spider/compiler/text/utf8.hpp +++ b/src/spider/compiler/text/utf8.hpp @@ -87,13 +87,27 @@ namespace spider { } inline bool charAt(const std::string& str, isize& index, u32& out) { - u8 ch0 = u8(str[index]); isize chlen = isValidSeq(str.c_str(), str.size()); if(chlen == 0) return false; out = decodeArr(str.c_str(), chlen); return true; } + inline bool toUTF32(std::string_view str, std::u32string& out) { + isize _i = 0, _size; + + auto cptr = str.cbegin(); + auto csize = str.size(); + + while(_i < csize) { + _size = isValidSeq(cptr + _i, csize - _i); + if(_size == 0) return false; + out += decodeArr(cptr + _i, _size); + } + + return _i == csize; + } + inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {} }