still not ready
This commit is contained in:
@@ -0,0 +1,11 @@
|
||||
#include "spider/compiler/assembler/AsmParser.hpp"
|
||||
|
||||
namespace spider {
|
||||
|
||||
|
||||
|
||||
}
|
||||
|
||||
int main() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
#pragma once
|
||||
|
||||
#include <spider/compiler/assembler/Assembler.hpp>
|
||||
|
||||
namespace spider {
|
||||
|
||||
/**
|
||||
* EBNF Parser for the Spider Assembly.
|
||||
*/
|
||||
class AsmParser {
|
||||
private:
|
||||
|
||||
public:
|
||||
|
||||
AsmParser();
|
||||
|
||||
~AsmParser();
|
||||
|
||||
public:
|
||||
|
||||
public:
|
||||
|
||||
void ebnf_();
|
||||
|
||||
};
|
||||
|
||||
}
|
||||
@@ -18,16 +18,11 @@ namespace spider {
|
||||
SUCCESS,
|
||||
FILE_NOT_FOUND, FILE_RECURSIVE_LOAD,
|
||||
};
|
||||
struct Level {
|
||||
uptr<TextReader> reader;
|
||||
RootToken root;
|
||||
std::string source;
|
||||
};
|
||||
|
||||
public:
|
||||
|
||||
set<fs::path> fstack;
|
||||
deque<Level> levels;
|
||||
RootToken root;
|
||||
|
||||
public:
|
||||
|
||||
@@ -37,6 +32,12 @@ namespace spider {
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
* Attempts to load a file, fails if it
|
||||
* doesn't exist.
|
||||
*/
|
||||
Error loadText(const std::string& path);
|
||||
|
||||
/**
|
||||
* Attempts to load a file, fails if it
|
||||
* doesn't exist.
|
||||
|
||||
@@ -46,10 +46,11 @@ namespace spider {
|
||||
namespace fs = std::filesystem;
|
||||
|
||||
struct pos {
|
||||
isize byteoff;
|
||||
isize line;
|
||||
isize col;
|
||||
pos(isize line = 1, isize col = 1)
|
||||
: line(line), col(col) {}
|
||||
pos(isize _byteoff = 0, isize _line = 1, isize _col = 1)
|
||||
: byteoff(_byteoff), line(_line), col(_col) {}
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
@@ -8,66 +8,178 @@ namespace spider {
|
||||
|
||||
// Text Reader //
|
||||
|
||||
int TextReader::nextByte() {
|
||||
int ch = getStream().get();
|
||||
if (ch == std::istream::traits_type::eof()) {
|
||||
return -1;
|
||||
}
|
||||
return ch;
|
||||
TextReader::TextReader() : err(false), eof(false), bufferIndex(0) {
|
||||
// Prime the buffer with the first character
|
||||
// so current() is immediately valid
|
||||
fillBufferTo(0);
|
||||
}
|
||||
|
||||
bool TextReader::nextChar(u32& ch) {
|
||||
int n = nextByte();
|
||||
if(n == -1) return false;
|
||||
TextReader::~TextReader() {}
|
||||
|
||||
isize len = utf8::seqlen(u8(n));
|
||||
if(len == 0) return false;
|
||||
char TextReader::readByte() {
|
||||
if (err) return 0;
|
||||
auto& s = getStream();
|
||||
|
||||
isize i = 1;
|
||||
char arr[4];
|
||||
arr[0] = char(n);
|
||||
|
||||
while(i < len) {
|
||||
n = nextByte();
|
||||
if(n == -1) return false;
|
||||
arr[i++] = char(n);
|
||||
if(s.bad()) {
|
||||
err = true;
|
||||
errmsg = "Stream raised bad bit.";
|
||||
return 0;
|
||||
}
|
||||
|
||||
ch = utf8::decodeArr(arr, len);
|
||||
advance(ch);
|
||||
int ch = s.get();
|
||||
if (ch == std::istream::traits_type::eof()) {
|
||||
eof = true;
|
||||
return 0;
|
||||
}
|
||||
|
||||
at.byteoff++;
|
||||
return char(ch);
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns the current character.
|
||||
*/
|
||||
u32 TextReader::current() {
|
||||
if (bufferIndex < buffer.size()) {
|
||||
return buffer[bufferIndex];
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Reads the next character and advances the position tracker.
|
||||
*/
|
||||
u32 TextReader::nextChar() {
|
||||
if (err) return 0;
|
||||
|
||||
// Ensure the character we are moving TO exists
|
||||
if (fillBufferTo(1)) {
|
||||
// Track the cursor position using the character we are leaving behind
|
||||
advance(current());
|
||||
bufferIndex++;
|
||||
return current();
|
||||
}
|
||||
|
||||
// If we couldn't fill the buffer, we hit EOF
|
||||
eof = true;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Keeps the next n-th character (n = 0 is current).
|
||||
*/
|
||||
u32 TextReader::peekChar(isize n) {
|
||||
if (err) return 0;
|
||||
if (fillBufferTo(n)) return buffer[bufferIndex + n];
|
||||
return 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Clears the buffer from previous characters, keeping current and future ones.
|
||||
*/
|
||||
void TextReader::commit() {
|
||||
if (bufferIndex > 0) {
|
||||
// Erase everything before the current buffer index
|
||||
buffer.erase(buffer.begin(), buffer.begin() + bufferIndex);
|
||||
bufferIndex = 0;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Rolls back any previous characters within the limits of the uncommitted buffer.
|
||||
*/
|
||||
void TextReader::rollback(isize n) {
|
||||
// Prevent rolling back past the start of our committed buffer
|
||||
if (n > bufferIndex) n = bufferIndex;
|
||||
|
||||
// We must track positions backward or recalculate if exact column match is needed.
|
||||
// Assuming simple rollback of the pointer here per definition.
|
||||
bufferIndex -= n;
|
||||
eof = false;
|
||||
}
|
||||
|
||||
TextReader::operator bool() const {
|
||||
return !err;
|
||||
}
|
||||
|
||||
/**
|
||||
* Updates track position metrics based on the processed character.
|
||||
*/
|
||||
void TextReader::advance(u32 ch) {
|
||||
if (ch == '\n') {
|
||||
at.line++;
|
||||
at.col = 1;
|
||||
} else {
|
||||
at.col++;
|
||||
}
|
||||
}
|
||||
|
||||
bool TextReader::readChar() {
|
||||
// Read one byte
|
||||
char bytes[4];
|
||||
isize bindex = 0;
|
||||
|
||||
bytes[bindex] = readByte();
|
||||
if (err) return false;
|
||||
if (eof) return false;
|
||||
|
||||
isize chsize = utf8::seqlen(u8(bytes[bindex]));
|
||||
if(chsize == 0) {
|
||||
err = true;
|
||||
errmsg = "Invalid start of UTF-8 sequence.";
|
||||
return false;
|
||||
}
|
||||
bindex++;
|
||||
|
||||
while (bindex < chsize) {
|
||||
bytes[bindex] = readByte();
|
||||
if (err) return false;
|
||||
if (eof) return false;
|
||||
if (!utf8::isCont(u8(bytes[bindex]))) {
|
||||
err = true;
|
||||
errmsg = "Invalid continuation of UTF-8 sequence.";
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
u32 decodedChar = utf8::decodeArr(bytes, chsize);
|
||||
buffer.push_back(decodedChar);
|
||||
return true;
|
||||
}
|
||||
|
||||
void TextReader::advance(u32 ch) {
|
||||
if (ch == u32('\n')) {
|
||||
if (lastWasCR) {
|
||||
lastWasCR = false; // Mixed CRLF handling
|
||||
} else {
|
||||
at.line++;
|
||||
at.col = 1;
|
||||
}
|
||||
} else if (ch == u32('\r')) {
|
||||
at.line++;
|
||||
at.col = 1;
|
||||
lastWasCR = true;
|
||||
} else {
|
||||
at.col++;
|
||||
lastWasCR = false;
|
||||
/**
|
||||
* Fills the buffer sequentially until it contains at least up
|
||||
* to (bufferIndex + targetOffset).
|
||||
*/
|
||||
bool TextReader::fillBufferTo(isize targetOffset) {
|
||||
isize targetSize = bufferIndex + targetOffset + 1;
|
||||
while (buffer.size() < targetSize) {
|
||||
if(readChar()) continue;
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns true if the stream is consumed and no elements remain in the read buffer.
|
||||
*/
|
||||
bool TextReader::isEOF() {
|
||||
return getStream().peek() == std::istream::traits_type::eof();
|
||||
if (err) return false;
|
||||
return eof && bufferIndex >= buffer.size();
|
||||
}
|
||||
|
||||
pos TextReader::getPosition() const {
|
||||
return at;
|
||||
}
|
||||
|
||||
std::string TextReader::getError() const {
|
||||
return errmsg;
|
||||
}
|
||||
|
||||
// File Reader //
|
||||
|
||||
FileTextReader::FileTextReader(const std::string& filename)
|
||||
: fileStream(filename, std::ios::binary) {
|
||||
: fileStream(filename, std::ios::binary) {
|
||||
if (!fileStream.is_open()) {
|
||||
throw std::runtime_error("Failed to open file: " + filename);
|
||||
}
|
||||
@@ -81,7 +193,7 @@ namespace spider {
|
||||
|
||||
StringTextReader::StringTextReader(std::string initialText)
|
||||
: buffer(std::move(initialText)),
|
||||
stringStream(std::make_unique<std::istringstream>(buffer)) {
|
||||
stringStream(std::make_unique<std::istringstream>(buffer)) {
|
||||
}
|
||||
|
||||
std::istream& StringTextReader::getStream() {
|
||||
@@ -91,7 +203,6 @@ namespace spider {
|
||||
void StringTextReader::set(const std::string& newText) {
|
||||
buffer = newText;
|
||||
stringStream = std::make_unique<std::istringstream>(buffer);
|
||||
lastWasCR = false;
|
||||
}
|
||||
|
||||
void StringTextReader::append(const std::string& extraText) {
|
||||
|
||||
@@ -16,33 +16,108 @@ namespace spider {
|
||||
class TextReader {
|
||||
protected:
|
||||
|
||||
/**
|
||||
* Error flag, in case of an error
|
||||
* all operations is no-op.
|
||||
*/
|
||||
bool err;
|
||||
|
||||
/**
|
||||
* EOF reached
|
||||
*/
|
||||
bool eof;
|
||||
|
||||
/**
|
||||
* Current position.
|
||||
*/
|
||||
pos at;
|
||||
bool lastWasCR = false;
|
||||
|
||||
std::string errmsg;
|
||||
|
||||
struct stored_char {
|
||||
u8 byte_count;
|
||||
u32 value;
|
||||
};
|
||||
|
||||
/**
|
||||
* Buffer of extracted characters.
|
||||
*/
|
||||
vector<u32> buffer;
|
||||
|
||||
/**
|
||||
* Buffer index.
|
||||
* buffer[bufferIndex] == current character.
|
||||
* Then so, n < bufferIndex == past chars
|
||||
* and n > bufferIndex == next chars
|
||||
*/
|
||||
isize bufferIndex;
|
||||
|
||||
public:
|
||||
|
||||
TextReader() = default;
|
||||
TextReader();
|
||||
|
||||
virtual ~TextReader() = default;
|
||||
|
||||
protected:
|
||||
|
||||
int nextByte();
|
||||
virtual ~TextReader();
|
||||
|
||||
public:
|
||||
|
||||
bool nextChar(u32& ch);
|
||||
/**
|
||||
* Returns the current character.
|
||||
*/
|
||||
u32 current();
|
||||
|
||||
/**
|
||||
* Reads the next character.
|
||||
*/
|
||||
u32 nextChar();
|
||||
|
||||
/**
|
||||
* Keeps the next n-th character
|
||||
* n = 0 is the current one.
|
||||
*/
|
||||
u32 peekChar(isize n = 1);
|
||||
|
||||
/**
|
||||
* Clears the buffer from previous characters,
|
||||
* removing the ability for rolling back
|
||||
* any previous characters from this point on.
|
||||
*/
|
||||
void commit();
|
||||
|
||||
/**
|
||||
* Rolls back any previous characters,
|
||||
* so long as the state hasn't commited.
|
||||
* n = 0 is a no op, since it's the current char.
|
||||
*/
|
||||
void rollback(isize n = isize(-1));
|
||||
|
||||
/**
|
||||
* Returns true if the end of the stream has been reached.
|
||||
* Returns false if the EOS hasn't been reached but
|
||||
* an error has occurred
|
||||
*/
|
||||
bool isEOF();
|
||||
|
||||
/**
|
||||
* Returns the position of the cursor.
|
||||
*/
|
||||
pos getPosition() const;
|
||||
|
||||
operator bool() const;
|
||||
|
||||
std::string getError() const;
|
||||
|
||||
protected:
|
||||
|
||||
char readByte();
|
||||
|
||||
bool readChar();
|
||||
|
||||
void advance(u32 ch);
|
||||
|
||||
virtual std::istream& getStream() = 0;
|
||||
|
||||
bool fillBufferTo(isize index);
|
||||
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
@@ -86,6 +86,8 @@ namespace spider {
|
||||
return out;
|
||||
}
|
||||
|
||||
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user