From fb87813cbfdd96efd4222e120a6e2225b1fa668f Mon Sep 17 00:00:00 2001 From: Kittycannon Date: Sun, 5 Jul 2026 13:43:09 -0600 Subject: [PATCH] still not ready --- makefile | 78 ++ spider/compiler/assembly/AssemblyParser.hpp | 812 -------------------- src/spider/compiler/Compiler.cpp | 11 + src/spider/compiler/assembler/AsmParser.hpp | 27 + src/spider/compiler/assembler/Assembler.hpp | 13 +- src/spider/compiler/common.hpp | 5 +- src/spider/compiler/text/TextReader.cpp | 191 ++++- src/spider/compiler/text/TextReader.hpp | 91 ++- src/spider/compiler/text/utf8.hpp | 2 + 9 files changed, 362 insertions(+), 868 deletions(-) create mode 100644 makefile delete mode 100644 spider/compiler/assembly/AssemblyParser.hpp create mode 100644 src/spider/compiler/assembler/AsmParser.hpp diff --git a/makefile b/makefile new file mode 100644 index 0000000..c42f542 --- /dev/null +++ b/makefile @@ -0,0 +1,78 @@ +# Include variables from user defined +# makevars.mak for paths & other stuff. +#include makevars.mk + +#Compiler and Linker +CC := g++ + +#The Target Binary Program +TARGET := out.exe + +#The Directories, Source, Includes, Objects, Binary and Resources +SRCDIR := src +BUILDDIR := bin +TARGETDIR := out +SRCEXT := cpp +DEPEXT := d +OBJEXT := o + +#Flags, Libraries and Includes +ROOT := ./ +CFLAGS := -std=c++20 -O2 \ + -Wall -Werror -Wextra \ + -Wshadow -Wnon-virtual-dtor -Wold-style-cast -Wcast-align \ + -Wunused -Woverloaded-virtual -Wconversion \ + -Wsign-conversion -Wnull-dereference -Wdouble-promotion \ + -Wformat=2 -Wimplicit-fallthrough -Wsuggest-override \ + -Wextra-semi -Wduplicated-cond -Wduplicated-branches \ + -Wlogical-op -Wuseless-cast -Wno-unused-parameter +LFLAGS := -std=c++20 -static -static-libstdc++ -static-libgcc \ + -Wl,--fatal-warnings -Wl,--warn-common +LIBDIRS := +LIB := +INC := -I./src/ + +#--------------------------------------------------------------------------------- +# DO NOT EDIT BELOW THIS LINE +#--------------------------------------------------------------------------------- +SOURCES := $(shell find $(SRCDIR) -type f -name *.$(SRCEXT)) +OBJECTS := $(patsubst $(SRCDIR)/%,$(BUILDDIR)/%,$(SOURCES:.$(SRCEXT)=.$(OBJEXT))) + +#Defauilt Make +all: directories $(TARGET) + +#Remake +remake: cleaner all + +#Make the Directories +directories: + @mkdir -p $(TARGETDIR) + @mkdir -p $(BUILDDIR) + +#Clean only Objecst +clean: + @$(RM) -rf $(BUILDDIR) + +#Full Clean, Objects and Binaries +cleaner: clean + @$(RM) -rf $(TARGETDIR) + +#Pull in dependency info for *existing* .o files +-include $(OBJECTS:.$(OBJEXT)=.$(DEPEXT)) + +#Link +$(TARGET): $(OBJECTS) + $(CC) $(LFLAGS) -o $(TARGETDIR)/$(TARGET) $^ $(LIBDIRS) $(LIB) + +#Compile +$(BUILDDIR)/%.$(OBJEXT): $(SRCDIR)/%.$(SRCEXT) + @mkdir -p $(dir $@) + $(CC) $(CFLAGS) $(INC) -c -o $@ $< + @$(CC) $(CFLAGS) -MM $(SRCDIR)/$*.$(SRCEXT) > $(BUILDDIR)/$*.$(DEPEXT) + @cp -f $(BUILDDIR)/$*.$(DEPEXT) $(BUILDDIR)/$*.$(DEPEXT).tmp + @sed -e 's|.*:|$(BUILDDIR)/$*.$(OBJEXT):|' < $(BUILDDIR)/$*.$(DEPEXT).tmp > $(BUILDDIR)/$*.$(DEPEXT) + @sed -e 's/.*://' -e 's/\\$$//' < $(BUILDDIR)/$*.$(DEPEXT).tmp | fmt -1 | sed -e 's/^ *//' -e 's/$$/:/' >> $(BUILDDIR)/$*.$(DEPEXT) + @rm -f $(BUILDDIR)/$*.$(DEPEXT).tmp + +#Non-File Targets +.PHONY: all remake clean cleaner resources diff --git a/spider/compiler/assembly/AssemblyParser.hpp b/spider/compiler/assembly/AssemblyParser.hpp deleted file mode 100644 index 09fd1e0..0000000 --- a/spider/compiler/assembly/AssemblyParser.hpp +++ /dev/null @@ -1,812 +0,0 @@ -#pragma once - -#include -#include -#include -#include - -class AssemblyParser { -private: - std::string src; - size_t pos = 0; - - std::string peek_str(size_t len) { - if (pos + len <= src.length()) return src.substr(pos, len); - return src.substr(pos); - } - - char peek() { return pos < src.length() ? src[pos] : '\0'; } - - void match_char(char expected) { - if (peek() == expected) pos++; - else throw std::runtime_error("Unexpected token matching character"); - } - - void match_string(std::string expected) { - if (peek_str(expected.length()) == expected) pos += expected.length(); - else throw std::runtime_error("Unexpected token matching string: " + expected); - } - - bool isUTF8Alpha() { return isalpha(peek()); } - bool isWhithespaceCharNotCrLf() { return peek() == ' ' || peek() == '\t'; } - bool isUTF8CharNotCrLf() { return peek() != '\r' && peek() != '\n' && peek() != '\0'; } - bool isUTF8CharLitCont() { return peek() != '\'' && peek() != '\\'; } - bool isUTF8StringLitCont() { return peek() != '"' && peek() != '\\'; } - -public: - AssemblyParser(std::string input) : src(input) {} - - void parse() { - parse_program(); - if (pos < src.length()) throw std::runtime_error("Trailing characters left unparsed."); - std::cout << "Assembly source compiled cleanly!" << std::endl; - } - - void parse_letter() { - if (/* option 1 */ true) { - if (isUTF8Alpha()) { pos++; } else { throw std::runtime_error("Failed validation for isUTF8Alpha"); } - } - } - - void parse_digit() { - if (/* option 1 */ true) { - match_char('0'); - } else if (/* option 2 */ true) { - match_char('1'); - } else if (/* option 3 */ true) { - match_char('2'); - } else if (/* option 4 */ true) { - match_char('3'); - } else if (/* option 5 */ true) { - match_char('4'); - } else if (/* option 6 */ true) { - match_char('5'); - } else if (/* option 7 */ true) { - match_char('6'); - } else if (/* option 8 */ true) { - match_char('7'); - } else if (/* option 9 */ true) { - match_char('8'); - } else if (/* option 10 */ true) { - match_char('9'); - } - } - - void parse_alpha_num_char() { - if (/* option 1 */ true) { - parse_letter(); - } else if (/* option 2 */ true) { - parse_digit(); - } - } - - void parse_hex_digit() { - if (/* option 1 */ true) { - parse_digit(); - } else if (/* option 2 */ true) { - match_char('A'); - } else if (/* option 3 */ true) { - match_char('B'); - } else if (/* option 4 */ true) { - match_char('C'); - } else if (/* option 5 */ true) { - match_char('D'); - } else if (/* option 6 */ true) { - match_char('E'); - } else if (/* option 7 */ true) { - match_char('F'); - } else if (/* option 8 */ true) { - match_char('a'); - } else if (/* option 9 */ true) { - match_char('b'); - } else if (/* option 10 */ true) { - match_char('c'); - } else if (/* option 11 */ true) { - match_char('d'); - } else if (/* option 12 */ true) { - match_char('e'); - } else if (/* option 13 */ true) { - match_char('f'); - } - } - - void parse_octal_digit() { - if (/* option 1 */ true) { - match_char('0'); - } else if (/* option 2 */ true) { - match_char('1'); - } else if (/* option 3 */ true) { - match_char('2'); - } else if (/* option 4 */ true) { - match_char('3'); - } else if (/* option 5 */ true) { - match_char('4'); - } else if (/* option 6 */ true) { - match_char('5'); - } else if (/* option 7 */ true) { - match_char('6'); - } else if (/* option 8 */ true) { - match_char('7'); - } - } - - void parse_binary_digit() { - if (/* option 1 */ true) { - match_char('0'); - } else if (/* option 2 */ true) { - match_char('1'); - } - } - - void parse_ws_char() { - if (/* option 1 */ true) { - if (isWhithespaceCharNotCrLf()) { pos++; } else { throw std::runtime_error("Failed validation for isWhithespaceCharNotCrLf"); } - } - } - - void parse_ws_optional() { - if (/* option 1 */ true) { - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_ws_char(); - } - } - } - } - - void parse_whitespace() { - if (/* option 1 */ true) { - parse_ws_char(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_ws_char(); - } - } - } - } - - void parse_newline() { - if (/* option 1 */ true) { - match_char('\r'); - } else if (/* option 2 */ true) { - match_char('\n'); - } else if (/* option 3 */ true) { - match_string("\r\n"); - } - } - - void parse_utf8_char() { - if (/* option 1 */ true) { - if (isUTF8CharNotCrLf()) { pos++; } else { throw std::runtime_error("Failed validation for isUTF8CharNotCrLf"); } - } - } - - void parse_char_escape() { - if (/* option 1 */ true) { - match_char('\\'); - parse_utf8_char(); - } - } - - void parse_char_content() { - if (/* option 1 */ true) { - parse_char_escape(); - } else if (/* option 2 */ true) { - if (isUTF8CharLitCont()) { pos++; } else { throw std::runtime_error("Failed validation for isUTF8CharLitCont"); } - } - } - - void parse_char_lit() { - if (/* option 1 */ true) { - match_char('\''); - parse_char_content(); - match_char('\''); - } - } - - void parse_string_char() { - if (/* option 1 */ true) { - parse_char_escape(); - } else if (/* option 2 */ true) { - if (isUTF8StringLitCont()) { pos++; } else { throw std::runtime_error("Failed validation for isUTF8StringLitCont"); } - } - } - - void parse_string_lit() { - if (/* option 1 */ true) { - match_char('"'); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_string_char(); - } - } - match_char('"'); - } - } - - void parse_identifier() { - if (/* option 1 */ true) { - if (/* option 1 */ true) { - parse_letter(); - } else if (/* option 2 */ true) { - match_char('_'); - } - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_alpha_num_char(); - } else if (/* option 2 */ true) { - match_char('_'); - } - } - } - } - - void parse_comment() { - if (/* option 1 */ true) { - match_char(';'); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_utf8_char(); - } - } - } - } - - void parse_sign() { - if (/* option 1 */ true) { - match_char('+'); - } else if (/* option 2 */ true) { - match_char('-'); - } - } - - void parse_exponent_marker() { - if (/* option 1 */ true) { - match_char('e'); - } else if (/* option 2 */ true) { - match_char('E'); - } - } - - void parse_exponent() { - if (/* option 1 */ true) { - parse_exponent_marker(); - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_sign(); - } - } - parse_digit(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_digit(); - } - } - } - } - - void parse_decimal_lit() { - if (/* option 1 */ true) { - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_sign(); - } - } - parse_digit(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_digit(); - } - } - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - match_char('B'); - } else if (/* option 2 */ true) { - match_char('S'); - } else if (/* option 3 */ true) { - match_char('I'); - } else if (/* option 4 */ true) { - match_char('L'); - } - } - } - } - - void parse_float_lit() { - if (/* option 1 */ true) { - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_sign(); - } - } - if (/* option 1 */ true) { - if (/* option 1 */ true) { - parse_digit(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_digit(); - } - } - match_char('.'); - parse_digit(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_digit(); - } - } - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_exponent(); - } - } - } - } else if (/* option 2 */ true) { - if (/* option 1 */ true) { - match_char('.'); - parse_digit(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_digit(); - } - } - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_exponent(); - } - } - } - } else if (/* option 3 */ true) { - if (/* option 1 */ true) { - parse_digit(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_digit(); - } - } - parse_exponent(); - } - } - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - match_char('F'); - } else if (/* option 2 */ true) { - match_char('D'); - } - } - } - } - - void parse_hex_lit() { - if (/* option 1 */ true) { - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_sign(); - } - } - match_string("0x"); - parse_hex_digit(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_hex_digit(); - } - } - } - } - - void parse_octal_lit() { - if (/* option 1 */ true) { - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_sign(); - } - } - match_string("0c"); - parse_octal_digit(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_octal_digit(); - } - } - } - } - - void parse_binary_lit() { - if (/* option 1 */ true) { - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_sign(); - } - } - match_string("0b"); - parse_binary_digit(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_binary_digit(); - } - } - } - } - - void parse_literal() { - if (/* option 1 */ true) { - parse_decimal_lit(); - } else if (/* option 2 */ true) { - parse_float_lit(); - } else if (/* option 3 */ true) { - parse_hex_lit(); - } else if (/* option 4 */ true) { - parse_octal_lit(); - } else if (/* option 5 */ true) { - parse_binary_lit(); - } else if (/* option 6 */ true) { - parse_string_lit(); - } else if (/* option 7 */ true) { - parse_char_lit(); - } - } - - void parse_literal_cast() { - if (/* option 1 */ true) { - if (/* option 1 */ true) { - match_char('B'); - } else if (/* option 2 */ true) { - match_char('S'); - } else if (/* option 3 */ true) { - match_char('I'); - } else if (/* option 4 */ true) { - match_char('L'); - } else if (/* option 5 */ true) { - match_char('F'); - } else if (/* option 6 */ true) { - match_char('D'); - } - parse_ws_optional(); - match_char('('); - parse_ws_optional(); - parse_literal(); - parse_ws_optional(); - match_char(')'); - } - } - - void parse_literal_decl() { - if (/* option 1 */ true) { - parse_literal(); - } else if (/* option 2 */ true) { - parse_literal_cast(); - } - } - - void parse_register() { - if (/* option 1 */ true) { - match_char('R'); - parse_alpha_num_char(); - parse_alpha_num_char(); - } - } - - void parse_addrm_ind() { - if (/* option 1 */ true) { - match_char('['); - parse_ws_optional(); - parse_literal_decl(); - parse_ws_optional(); - match_char(']'); - } - } - - void parse_addrm_ptr() { - if (/* option 1 */ true) { - match_char('['); - parse_ws_optional(); - parse_register(); - parse_ws_optional(); - match_char(']'); - } - } - - void parse_addrm_idx() { - if (/* option 1 */ true) { - match_char('['); - parse_ws_optional(); - parse_register(); - parse_ws_optional(); - match_char('+'); - parse_ws_optional(); - parse_literal_decl(); - parse_ws_optional(); - match_char(']'); - } - } - - void parse_addrm_sca() { - if (/* option 1 */ true) { - match_char('['); - parse_ws_optional(); - parse_register(); - parse_ws_optional(); - match_char('+'); - parse_register(); - parse_ws_optional(); - match_char('*'); - parse_ws_optional(); - parse_literal_decl(); - parse_ws_optional(); - match_char(']'); - } - } - - void parse_addrm_dis() { - if (/* option 1 */ true) { - match_char('['); - parse_ws_optional(); - parse_register(); - parse_ws_optional(); - match_char('+'); - parse_register(); - parse_ws_optional(); - match_char('*'); - parse_ws_optional(); - parse_literal_decl(); - parse_ws_optional(); - match_char('+'); - parse_ws_optional(); - parse_literal_decl(); - parse_ws_optional(); - match_char(']'); - } - } - - void parse_addr_modes() { - if (/* option 1 */ true) { - parse_addrm_ind(); - } else if (/* option 2 */ true) { - parse_addrm_ptr(); - } else if (/* option 3 */ true) { - parse_addrm_idx(); - } else if (/* option 4 */ true) { - parse_addrm_sca(); - } else if (/* option 5 */ true) { - parse_addrm_dis(); - } - } - - void parse_operand() { - if (/* option 1 */ true) { - parse_register(); - } else if (/* option 2 */ true) { - parse_identifier(); - } else if (/* option 3 */ true) { - parse_literal_decl(); - } else if (/* option 4 */ true) { - parse_addr_modes(); - } - } - - void parse_opcode() { - if (/* option 1 */ true) { - parse_letter(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_alpha_num_char(); - } - } - } - } - - void parse_operand_list() { - if (/* option 1 */ true) { - parse_operand(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - match_char(','); - parse_ws_optional(); - parse_operand(); - } - } - } - } - - void parse_instruction() { - if (/* option 1 */ true) { - parse_opcode(); - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_whitespace(); - parse_operand_list(); - } - } - } - } - - void parse_include_decl() { - if (/* option 1 */ true) { - match_string("include"); - parse_whitespace(); - parse_string_lit(); - } - } - - void parse_annotation_oper() { - if (/* option 1 */ true) { - parse_identifier(); - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_ws_optional(); - match_char('='); - parse_ws_optional(); - parse_literal_decl(); - } - } - } - } - - void parse_annotation_ops() { - if (/* option 1 */ true) { - parse_annotation_oper(); - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_ws_optional(); - match_char(','); - parse_ws_optional(); - parse_annotation_oper(); - } - } - } - } - - void parse_annotation_args() { - if (/* option 1 */ true) { - match_char('('); - parse_ws_optional(); - parse_annotation_ops(); - parse_ws_optional(); - match_char(')'); - } - } - - void parse_annotation() { - if (/* option 1 */ true) { - match_char('@'); - parse_identifier(); - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_annotation_args(); - } - } - } - } - - void parse_section_decl() { - if (/* option 1 */ true) { - match_string("section"); - parse_whitespace(); - match_char('.'); - parse_identifier(); - } - } - - void parse_label() { - if (/* option 1 */ true) { - parse_identifier(); - match_char(':'); - } - } - - void parse_line_content() { - if (/* option 1 */ true) { - parse_include_decl(); - } else if (/* option 2 */ true) { - parse_section_decl(); - } else if (/* option 3 */ true) { - if (/* option 1 */ true) { - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_annotation(); - parse_whitespace(); - } - } - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_label(); - parse_ws_optional(); - } - } - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_instruction(); - } - } - } - } - } - - void parse_line() { - if (/* option 1 */ true) { - parse_ws_optional(); - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_line_content(); - } - } - parse_ws_optional(); - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_comment(); - } - } - parse_newline(); - } - } - - void parse_line_last() { - if (/* option 1 */ true) { - parse_ws_optional(); - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_line_content(); - } - } - parse_ws_optional(); - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_comment(); - } - } - } - } - - void parse_program() { - if (/* option 1 */ true) { - // Repeat block - while (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_line(); - } - } - // Optional block - if (/* lookahead check */ true) { - if (/* option 1 */ true) { - parse_line_last(); - } - } - } - } -}; diff --git a/src/spider/compiler/Compiler.cpp b/src/spider/compiler/Compiler.cpp index e69de29..3fc1059 100644 --- a/src/spider/compiler/Compiler.cpp +++ b/src/spider/compiler/Compiler.cpp @@ -0,0 +1,11 @@ +#include "spider/compiler/assembler/AsmParser.hpp" + +namespace spider { + + + +} + +int main() { + return 0; +} diff --git a/src/spider/compiler/assembler/AsmParser.hpp b/src/spider/compiler/assembler/AsmParser.hpp new file mode 100644 index 0000000..0f3db2b --- /dev/null +++ b/src/spider/compiler/assembler/AsmParser.hpp @@ -0,0 +1,27 @@ +#pragma once + +#include + +namespace spider { + + /** + * EBNF Parser for the Spider Assembly. + */ + class AsmParser { + private: + + public: + + AsmParser(); + + ~AsmParser(); + + public: + + public: + + void ebnf_(); + + }; + +} diff --git a/src/spider/compiler/assembler/Assembler.hpp b/src/spider/compiler/assembler/Assembler.hpp index 25ef9c0..9a15797 100644 --- a/src/spider/compiler/assembler/Assembler.hpp +++ b/src/spider/compiler/assembler/Assembler.hpp @@ -18,16 +18,11 @@ namespace spider { SUCCESS, FILE_NOT_FOUND, FILE_RECURSIVE_LOAD, }; - struct Level { - uptr reader; - RootToken root; - std::string source; - }; public: set fstack; - deque levels; + RootToken root; public: @@ -37,6 +32,12 @@ namespace spider { public: + /** + * Attempts to load a file, fails if it + * doesn't exist. + */ + Error loadText(const std::string& path); + /** * Attempts to load a file, fails if it * doesn't exist. diff --git a/src/spider/compiler/common.hpp b/src/spider/compiler/common.hpp index 2674162..f9bba5e 100644 --- a/src/spider/compiler/common.hpp +++ b/src/spider/compiler/common.hpp @@ -46,10 +46,11 @@ namespace spider { namespace fs = std::filesystem; struct pos { + isize byteoff; isize line; isize col; - pos(isize line = 1, isize col = 1) - : line(line), col(col) {} + pos(isize _byteoff = 0, isize _line = 1, isize _col = 1) + : byteoff(_byteoff), line(_line), col(_col) {} }; } diff --git a/src/spider/compiler/text/TextReader.cpp b/src/spider/compiler/text/TextReader.cpp index dc3a1c9..9286af2 100644 --- a/src/spider/compiler/text/TextReader.cpp +++ b/src/spider/compiler/text/TextReader.cpp @@ -8,66 +8,178 @@ namespace spider { // Text Reader // - int TextReader::nextByte() { - int ch = getStream().get(); - if (ch == std::istream::traits_type::eof()) { - return -1; - } - return ch; + TextReader::TextReader() : err(false), eof(false), bufferIndex(0) { + // Prime the buffer with the first character + // so current() is immediately valid + fillBufferTo(0); } - bool TextReader::nextChar(u32& ch) { - int n = nextByte(); - if(n == -1) return false; + TextReader::~TextReader() {} - isize len = utf8::seqlen(u8(n)); - if(len == 0) return false; + char TextReader::readByte() { + if (err) return 0; + auto& s = getStream(); - isize i = 1; - char arr[4]; - arr[0] = char(n); - - while(i < len) { - n = nextByte(); - if(n == -1) return false; - arr[i++] = char(n); + if(s.bad()) { + err = true; + errmsg = "Stream raised bad bit."; + return 0; } - ch = utf8::decodeArr(arr, len); - advance(ch); + int ch = s.get(); + if (ch == std::istream::traits_type::eof()) { + eof = true; + return 0; + } + + at.byteoff++; + return char(ch); + } + + /** + * Returns the current character. + */ + u32 TextReader::current() { + if (bufferIndex < buffer.size()) { + return buffer[bufferIndex]; + } + return 0; + } + + /** + * Reads the next character and advances the position tracker. + */ + u32 TextReader::nextChar() { + if (err) return 0; + + // Ensure the character we are moving TO exists + if (fillBufferTo(1)) { + // Track the cursor position using the character we are leaving behind + advance(current()); + bufferIndex++; + return current(); + } + + // If we couldn't fill the buffer, we hit EOF + eof = true; + return 0; + } + + /** + * Keeps the next n-th character (n = 0 is current). + */ + u32 TextReader::peekChar(isize n) { + if (err) return 0; + if (fillBufferTo(n)) return buffer[bufferIndex + n]; + return 0; + } + + /** + * Clears the buffer from previous characters, keeping current and future ones. + */ + void TextReader::commit() { + if (bufferIndex > 0) { + // Erase everything before the current buffer index + buffer.erase(buffer.begin(), buffer.begin() + bufferIndex); + bufferIndex = 0; + } + } + + /** + * Rolls back any previous characters within the limits of the uncommitted buffer. + */ + void TextReader::rollback(isize n) { + // Prevent rolling back past the start of our committed buffer + if (n > bufferIndex) n = bufferIndex; + + // We must track positions backward or recalculate if exact column match is needed. + // Assuming simple rollback of the pointer here per definition. + bufferIndex -= n; + eof = false; + } + + TextReader::operator bool() const { + return !err; + } + + /** + * Updates track position metrics based on the processed character. + */ + void TextReader::advance(u32 ch) { + if (ch == '\n') { + at.line++; + at.col = 1; + } else { + at.col++; + } + } + + bool TextReader::readChar() { + // Read one byte + char bytes[4]; + isize bindex = 0; + + bytes[bindex] = readByte(); + if (err) return false; + if (eof) return false; + + isize chsize = utf8::seqlen(u8(bytes[bindex])); + if(chsize == 0) { + err = true; + errmsg = "Invalid start of UTF-8 sequence."; + return false; + } + bindex++; + + while (bindex < chsize) { + bytes[bindex] = readByte(); + if (err) return false; + if (eof) return false; + if (!utf8::isCont(u8(bytes[bindex]))) { + err = true; + errmsg = "Invalid continuation of UTF-8 sequence."; + return false; + } + } + + u32 decodedChar = utf8::decodeArr(bytes, chsize); + buffer.push_back(decodedChar); return true; } - void TextReader::advance(u32 ch) { - if (ch == u32('\n')) { - if (lastWasCR) { - lastWasCR = false; // Mixed CRLF handling - } else { - at.line++; - at.col = 1; - } - } else if (ch == u32('\r')) { - at.line++; - at.col = 1; - lastWasCR = true; - } else { - at.col++; - lastWasCR = false; + /** + * Fills the buffer sequentially until it contains at least up + * to (bufferIndex + targetOffset). + */ + bool TextReader::fillBufferTo(isize targetOffset) { + isize targetSize = bufferIndex + targetOffset + 1; + while (buffer.size() < targetSize) { + if(readChar()) continue; + return false; } + return true; } + /** + * Returns true if the stream is consumed and no elements remain in the read buffer. + */ bool TextReader::isEOF() { - return getStream().peek() == std::istream::traits_type::eof(); + if (err) return false; + return eof && bufferIndex >= buffer.size(); } pos TextReader::getPosition() const { return at; } + std::string TextReader::getError() const { + return errmsg; + } + // File Reader // FileTextReader::FileTextReader(const std::string& filename) - : fileStream(filename, std::ios::binary) { + : fileStream(filename, std::ios::binary) { if (!fileStream.is_open()) { throw std::runtime_error("Failed to open file: " + filename); } @@ -81,7 +193,7 @@ namespace spider { StringTextReader::StringTextReader(std::string initialText) : buffer(std::move(initialText)), - stringStream(std::make_unique(buffer)) { + stringStream(std::make_unique(buffer)) { } std::istream& StringTextReader::getStream() { @@ -91,7 +203,6 @@ namespace spider { void StringTextReader::set(const std::string& newText) { buffer = newText; stringStream = std::make_unique(buffer); - lastWasCR = false; } void StringTextReader::append(const std::string& extraText) { diff --git a/src/spider/compiler/text/TextReader.hpp b/src/spider/compiler/text/TextReader.hpp index b1d0130..336a795 100644 --- a/src/spider/compiler/text/TextReader.hpp +++ b/src/spider/compiler/text/TextReader.hpp @@ -16,33 +16,108 @@ namespace spider { class TextReader { protected: + /** + * Error flag, in case of an error + * all operations is no-op. + */ + bool err; + + /** + * EOF reached + */ + bool eof; + + /** + * Current position. + */ pos at; - bool lastWasCR = false; + + std::string errmsg; + + struct stored_char { + u8 byte_count; + u32 value; + }; + + /** + * Buffer of extracted characters. + */ + vector buffer; + + /** + * Buffer index. + * buffer[bufferIndex] == current character. + * Then so, n < bufferIndex == past chars + * and n > bufferIndex == next chars + */ + isize bufferIndex; public: - TextReader() = default; + TextReader(); - virtual ~TextReader() = default; - - protected: - - int nextByte(); + virtual ~TextReader(); public: - bool nextChar(u32& ch); + /** + * Returns the current character. + */ + u32 current(); + /** + * Reads the next character. + */ + u32 nextChar(); + + /** + * Keeps the next n-th character + * n = 0 is the current one. + */ + u32 peekChar(isize n = 1); + + /** + * Clears the buffer from previous characters, + * removing the ability for rolling back + * any previous characters from this point on. + */ + void commit(); + + /** + * Rolls back any previous characters, + * so long as the state hasn't commited. + * n = 0 is a no op, since it's the current char. + */ + void rollback(isize n = isize(-1)); + + /** + * Returns true if the end of the stream has been reached. + * Returns false if the EOS hasn't been reached but + * an error has occurred + */ bool isEOF(); + /** + * Returns the position of the cursor. + */ pos getPosition() const; + operator bool() const; + + std::string getError() const; + protected: + char readByte(); + + bool readChar(); + void advance(u32 ch); virtual std::istream& getStream() = 0; + bool fillBufferTo(isize index); + }; /** diff --git a/src/spider/compiler/text/utf8.hpp b/src/spider/compiler/text/utf8.hpp index 5db2d90..dbabed8 100644 --- a/src/spider/compiler/text/utf8.hpp +++ b/src/spider/compiler/text/utf8.hpp @@ -86,6 +86,8 @@ namespace spider { return out; } + inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {} + } }