compiles! Still working on tokens
This commit is contained in:
Vendored
+3
@@ -0,0 +1,3 @@
|
|||||||
|
{
|
||||||
|
"C_Cpp.default.compilerPath": "C:/msys64/ucrt64/bin/g++.exe"
|
||||||
|
}
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
|
||||||
|
BOLD="\033[1m"
|
||||||
|
RESET="\033[0m"
|
||||||
|
|
||||||
|
SUCCESS="\033[38;2;80;220;100m"
|
||||||
|
FAIL="\033[38;2;255;90;90m"
|
||||||
|
INFO="\033[38;2;100;200;255m"
|
||||||
|
DIM="\033[2m"
|
||||||
|
|
||||||
|
LOG="./out/out.log"
|
||||||
|
|
||||||
|
# Start
|
||||||
|
printf "${DIM}────────────────────────────────────────${RESET}\n"
|
||||||
|
printf "${INFO}${BOLD}> Running...${RESET}\n"
|
||||||
|
printf "${DIM}────────────────────────────────────────${RESET}\n\n"
|
||||||
|
|
||||||
|
# Send both stdout and stderr to the console and the log file.
|
||||||
|
start=$(date +%s.%N)
|
||||||
|
./out/out "$@" 2>&1 | tee "$LOG"
|
||||||
|
status=${PIPESTATUS[0]}
|
||||||
|
end=$(date +%s.%N)
|
||||||
|
elapsed=$(awk "BEGIN { printf \"%.3f\", $end - $start }")
|
||||||
|
|
||||||
|
# Status!
|
||||||
|
printf "\n${DIM}────────────────────────────────────────${RESET}\n"
|
||||||
|
if (( status == 0 )); then
|
||||||
|
printf "${SUCCESS}${BOLD}✓ Success${RESET} (exit %d)\n" "$status"
|
||||||
|
else
|
||||||
|
printf "${FAIL}${BOLD}✗ Failed${RESET} (exit %d)\n" "$status"
|
||||||
|
fi
|
||||||
|
printf "${INFO}${BOLD}> Time${RESET} ${elapsed}s\n"
|
||||||
|
printf "${INFO}${BOLD}> Log${RESET} %s\n" "$LOG"
|
||||||
|
printf "${DIM}────────────────────────────────────────${RESET}"
|
||||||
|
exit "$status"
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
#include "spider/compiler/assembler/AsmParser.hpp"
|
#include <iostream>
|
||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
|
|
||||||
@@ -7,5 +7,6 @@ namespace spider {
|
|||||||
}
|
}
|
||||||
|
|
||||||
int main() {
|
int main() {
|
||||||
|
std::cout << "Happy Day!" << std::endl;
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,383 +0,0 @@
|
|||||||
#include "AsmParser.hpp"
|
|
||||||
|
|
||||||
namespace spider {
|
|
||||||
|
|
||||||
AsmParser::AsmParser(uptr<TextReader> srcReader)
|
|
||||||
: reader(std::move(srcReader)) {
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::isDigit(u32 ch) const {
|
|
||||||
return ch >= '0' && ch <= '9';
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::isOctalDigit(u32 ch) const {
|
|
||||||
return ch >= '0' && ch <= '7';
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::isBinaryDigit(u32 ch) const {
|
|
||||||
return ch == '0' || ch == '1';
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::isHexDigit(u32 ch) const {
|
|
||||||
return (ch >= '0' && ch <= '9') ||
|
|
||||||
(ch >= 'A' && ch <= 'F') ||
|
|
||||||
(ch >= 'a' && ch <= 'f');
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::isLetter(u32 ch) const {
|
|
||||||
return (ch >= 'a' && ch <= 'z') || (ch >= 'A' && ch <= 'Z');
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::isAlphaNum(u32 ch) const {
|
|
||||||
return isLetter(ch) || isDigit(ch);
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::isWhitespaceChar(u32 ch) const {
|
|
||||||
return ch == ' ' || ch == '\t';
|
|
||||||
}
|
|
||||||
|
|
||||||
void AsmParser::parse_ws_optional() {
|
|
||||||
while (isWhitespaceChar(reader->current())) {
|
|
||||||
reader->nextChar();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_whitespace() {
|
|
||||||
if (!isWhitespaceChar(reader->current())) return false;
|
|
||||||
while (isWhitespaceChar(reader->current())) {
|
|
||||||
reader->nextChar();
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_newline() {
|
|
||||||
u32 cur = reader->current();
|
|
||||||
if (cur == '\n') {
|
|
||||||
reader->nextChar();
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
if (cur == '\r') {
|
|
||||||
reader->nextChar();
|
|
||||||
if (reader->current() == '\n') {
|
|
||||||
reader->nextChar();
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_comment() {
|
|
||||||
if (!reader->eat(';')) return false;
|
|
||||||
while (!reader->isEOF() && reader->current() != '\n' && reader->current() != '\r') {
|
|
||||||
reader->nextChar();
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_identifier(std::string& out_id) {
|
|
||||||
u32 cur = reader->current();
|
|
||||||
if (!isLetter(cur) && cur != '_') return false;
|
|
||||||
|
|
||||||
out_id.clear();
|
|
||||||
out_id += static_cast<char>(cur);
|
|
||||||
reader->nextChar();
|
|
||||||
|
|
||||||
while (isAlphaNum(reader->current()) || reader->current() == '_') {
|
|
||||||
out_id += static_cast<char>(reader->current());
|
|
||||||
reader->nextChar();
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_string_lit(std::string& out_str) {
|
|
||||||
if (!reader->eat('"')) return false;
|
|
||||||
out_str.clear();
|
|
||||||
while (!reader->isEOF() && reader->current() != '"') {
|
|
||||||
if (reader->current() == '\\') { // Escape rules
|
|
||||||
reader->nextChar();
|
|
||||||
}
|
|
||||||
out_str += static_cast<char>(reader->current());
|
|
||||||
reader->nextChar();
|
|
||||||
}
|
|
||||||
return reader->eat('"');
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_char_lit(u32& out_char) {
|
|
||||||
if (!reader->eat('\'')) return false;
|
|
||||||
if (reader->current() == '\\') {
|
|
||||||
reader->nextChar(); // Handle escapes
|
|
||||||
}
|
|
||||||
out_char = reader->current();
|
|
||||||
reader->nextChar();
|
|
||||||
return reader->eat('\'');
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_literal() {
|
|
||||||
reader->commit();
|
|
||||||
|
|
||||||
// Try complex prefixes first: 0x (Hex), 0c (Octal), 0b (Binary)
|
|
||||||
bool has_sign = reader->eat('+') || reader->eat('-');
|
|
||||||
if (reader->current() == '0') {
|
|
||||||
u32 prefix = reader->peekChar(1);
|
|
||||||
if (prefix == 'x' || prefix == 'X') {
|
|
||||||
reader->nextChar(); reader->nextChar(); // Consume 0x
|
|
||||||
if (!isHexDigit(reader->current())) { reader->rollback(); return false; }
|
|
||||||
while (isHexDigit(reader->current())) reader->nextChar();
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
if (prefix == 'c' || prefix == 'C') {
|
|
||||||
reader->nextChar(); reader->nextChar();
|
|
||||||
if (!isOctalDigit(reader->current())) { reader->rollback(); return false; }
|
|
||||||
while (isOctalDigit(reader->current())) reader->nextChar();
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
if (prefix == 'b' || prefix == 'B') {
|
|
||||||
reader->nextChar(); reader->nextChar();
|
|
||||||
if (!isBinaryDigit(reader->current())) { reader->rollback(); return false; }
|
|
||||||
while (isBinaryDigit(reader->current())) reader->nextChar();
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Reset if pure prefix didn't match to try floats/decimals cleanly
|
|
||||||
reader->rollback();
|
|
||||||
reader->commit();
|
|
||||||
|
|
||||||
reader->eat('+');
|
|
||||||
reader->eat('-'); // optional sign
|
|
||||||
|
|
||||||
// Float or Decimal
|
|
||||||
if (isDigit(reader->current()) || reader->current() == '.') {
|
|
||||||
bool saw_dot = false;
|
|
||||||
if (reader->eat('.')) saw_dot = true;
|
|
||||||
|
|
||||||
if (!isDigit(reader->current()) && saw_dot) { reader->rollback(); return false; }
|
|
||||||
while (isDigit(reader->current())) reader->nextChar();
|
|
||||||
|
|
||||||
if (!saw_dot && reader->eat('.')) {
|
|
||||||
while (isDigit(reader->current())) reader->nextChar();
|
|
||||||
saw_dot = true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Exponent marker
|
|
||||||
if (reader->current() == 'e' || reader->current() == 'E') {
|
|
||||||
reader->nextChar();
|
|
||||||
if (reader->current() == '+' || reader->current() == '-') reader->nextChar();
|
|
||||||
if (!isDigit(reader->current())) { reader->rollback(); return false; }
|
|
||||||
while (isDigit(reader->current())) reader->nextChar();
|
|
||||||
saw_dot = true; // Forcing it to evaluate as float behavior if needed
|
|
||||||
}
|
|
||||||
|
|
||||||
// Width Suffixes
|
|
||||||
u32 suffix = reader->current();
|
|
||||||
if (saw_dot) {
|
|
||||||
if (suffix == 'F' || suffix == 'D') reader->nextChar();
|
|
||||||
} else {
|
|
||||||
if (suffix == 'B' || suffix == 'S' || suffix == 'I' || suffix == 'L') reader->nextChar();
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Check strings or chars
|
|
||||||
std::string dummy_str; u32 dummy_ch;
|
|
||||||
if (parse_string_lit(dummy_str) || parse_char_lit(dummy_ch)) return true;
|
|
||||||
|
|
||||||
reader->rollback();
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_literal_decl() {
|
|
||||||
reader->commit();
|
|
||||||
u32 cast = reader->current();
|
|
||||||
if (cast == 'B' || cast == 'S' || cast == 'I' || cast == 'L' || cast == 'F' || cast == 'D') {
|
|
||||||
if (reader->peekChar(1) == '(' || (isWhitespaceChar(reader->peekChar(1)) && reader->peekChar(2) == '(')) {
|
|
||||||
reader->nextChar(); // cast width char
|
|
||||||
parse_ws_optional();
|
|
||||||
reader->eat('(');
|
|
||||||
parse_ws_optional();
|
|
||||||
if (!parse_literal()) { reader->rollback(); return false; }
|
|
||||||
parse_ws_optional();
|
|
||||||
if (reader->eat(')')) return true;
|
|
||||||
reader->rollback();
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return parse_literal();
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_register() {
|
|
||||||
if (reader->eat(u32('R'))) {
|
|
||||||
if (isAlphaNum(reader->current())) {
|
|
||||||
reader->nextChar();
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_addr_modes() {
|
|
||||||
if (!reader->eat(u32('['))) return false;
|
|
||||||
parse_ws_optional();
|
|
||||||
|
|
||||||
reader->commit();
|
|
||||||
|
|
||||||
// This parses all the permutations of nested elements inside `[...]` safely via back-tracking
|
|
||||||
// permutation 1: addrm_ind -> [ literal_decl ]
|
|
||||||
if (parse_literal_decl()) {
|
|
||||||
parse_ws_optional();
|
|
||||||
if (reader->eat(u32(']'))) return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
reader->rollback();
|
|
||||||
reader->commit();
|
|
||||||
|
|
||||||
// Base register is required for remaining components
|
|
||||||
if (parse_register()) {
|
|
||||||
parse_ws_optional();
|
|
||||||
if (reader->eat(u32(']'))) return true; // permutation 2: addrm_ptr -> [ register ]
|
|
||||||
|
|
||||||
if (reader->eat('+')) {
|
|
||||||
parse_ws_optional();
|
|
||||||
|
|
||||||
// Could be another register or a literal offset
|
|
||||||
reader->commit();
|
|
||||||
if (parse_register()) { // Scale/Displacement modes
|
|
||||||
parse_ws_optional();
|
|
||||||
if (reader->eat('*')) {
|
|
||||||
parse_ws_optional();
|
|
||||||
if (parse_literal_decl()) {
|
|
||||||
parse_ws_optional();
|
|
||||||
if (reader->eat(']')) return true; // permutation 4: addrm_sca
|
|
||||||
if (reader->eat('+')) {
|
|
||||||
parse_ws_optional();
|
|
||||||
if (parse_literal_decl()) {
|
|
||||||
parse_ws_optional();
|
|
||||||
if (reader->eat(']')) return true; // permutation 5: addrm_dis
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
reader->rollback();
|
|
||||||
|
|
||||||
// Fall back into standard index offset: permutation 3: addrm_idx -> [ reg + literal ]
|
|
||||||
if (parse_literal_decl()) {
|
|
||||||
parse_ws_optional();
|
|
||||||
if (reader->eat(']')) return true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
reader->rollback();
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_operand() {
|
|
||||||
if (parse_register()) return true;
|
|
||||||
if (parse_addr_modes()) return true;
|
|
||||||
if (parse_literal_decl()) return true;
|
|
||||||
std::string dummy_id;
|
|
||||||
if (parse_identifier(dummy_id)) return true;
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- Higher Level Statements ---
|
|
||||||
|
|
||||||
bool AsmParser::parse_instruction() {
|
|
||||||
u32 first = reader->current();
|
|
||||||
if (!isLetter(first)) return false;
|
|
||||||
|
|
||||||
// opcode name extraction
|
|
||||||
while (isAlphaNum(reader->current())) reader->nextChar();
|
|
||||||
|
|
||||||
reader->commit();
|
|
||||||
if (parse_whitespace()) {
|
|
||||||
if (parse_operand()) {
|
|
||||||
while (reader->eat(',')) {
|
|
||||||
parse_ws_optional();
|
|
||||||
if (!parse_operand()) { reader->rollback(); return false; }
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
reader->rollback(); // No valid operand list followed whitespace
|
|
||||||
}
|
|
||||||
return true; // Simple parameterless opcode
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_annotation() {
|
|
||||||
if (!reader->eat('@')) return false;
|
|
||||||
std::string tag;
|
|
||||||
if (!parse_identifier(tag)) return false;
|
|
||||||
|
|
||||||
if (reader->eat('(')) {
|
|
||||||
parse_ws_optional();
|
|
||||||
do {
|
|
||||||
std::string arg;
|
|
||||||
if (!parse_identifier(arg)) return false;
|
|
||||||
parse_ws_optional();
|
|
||||||
if (reader->eat('=')) {
|
|
||||||
parse_ws_optional();
|
|
||||||
if (!parse_literal_decl()) return false;
|
|
||||||
parse_ws_optional();
|
|
||||||
}
|
|
||||||
} while (reader->eat(','));
|
|
||||||
parse_ws_optional();
|
|
||||||
if (!reader->eat(')')) return false;
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_line_content() {
|
|
||||||
if (reader->eat("include")) {
|
|
||||||
if (!parse_whitespace()) return false;
|
|
||||||
std::string path;
|
|
||||||
return parse_string_lit(path);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (reader->eat("section")) {
|
|
||||||
if (!parse_whitespace()) return false;
|
|
||||||
if (!reader->eat('.')) return false;
|
|
||||||
std::string sec_name;
|
|
||||||
return parse_identifier(sec_name);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Main structural execution flow: [annotation] [label] [instruction]
|
|
||||||
reader->commit();
|
|
||||||
if (parse_annotation()) {
|
|
||||||
if (!parse_whitespace()) { reader->rollback(); return false; }
|
|
||||||
}
|
|
||||||
|
|
||||||
std::string lbl;
|
|
||||||
reader->commit();
|
|
||||||
if (parse_identifier(lbl)) {
|
|
||||||
if (reader->eat(':')) {
|
|
||||||
parse_ws_optional();
|
|
||||||
} else {
|
|
||||||
reader->rollback(); // Wasn't a label statement layout
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Optional structural tailing statement instruction
|
|
||||||
parse_instruction();
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool AsmParser::parse_program() {
|
|
||||||
while (!reader->isEOF()) {
|
|
||||||
parse_ws_optional();
|
|
||||||
|
|
||||||
parse_line_content();
|
|
||||||
|
|
||||||
parse_ws_optional();
|
|
||||||
if (reader->eat(';')) {
|
|
||||||
parse_comment();
|
|
||||||
}
|
|
||||||
|
|
||||||
if (!parse_newline() && !reader->isEOF()) {
|
|
||||||
// Handle compilation/lex error layout safely
|
|
||||||
reader->nextChar();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
}
|
|
||||||
@@ -1,75 +0,0 @@
|
|||||||
#pragma once
|
|
||||||
|
|
||||||
#include <spider/compiler/assembler/Assembler.hpp>
|
|
||||||
|
|
||||||
namespace spider {
|
|
||||||
|
|
||||||
/**
|
|
||||||
* EBNF Parser for the Spider Assembly.
|
|
||||||
*/
|
|
||||||
class AsmParser {
|
|
||||||
private:
|
|
||||||
|
|
||||||
uptr<TextReader> reader;
|
|
||||||
|
|
||||||
public:
|
|
||||||
|
|
||||||
AsmParser(uptr<TextReader> srcReader);
|
|
||||||
|
|
||||||
private:
|
|
||||||
|
|
||||||
bool isDigit(u32 ch) const;
|
|
||||||
|
|
||||||
bool isOctalDigit(u32 ch) const;
|
|
||||||
|
|
||||||
bool isBinaryDigit(u32 ch) const;
|
|
||||||
|
|
||||||
bool isHexDigit(u32 ch) const;
|
|
||||||
|
|
||||||
bool isLetter(u32 ch) const;
|
|
||||||
|
|
||||||
bool isAlphaNum(u32 ch) const;
|
|
||||||
|
|
||||||
bool isWhitespaceChar(u32 ch) const;
|
|
||||||
|
|
||||||
public:
|
|
||||||
|
|
||||||
void parse_ws_optional();
|
|
||||||
|
|
||||||
bool parse_whitespace();
|
|
||||||
|
|
||||||
bool parse_newline();
|
|
||||||
|
|
||||||
bool parse_comment();
|
|
||||||
|
|
||||||
bool parse_identifier(std::u32string& out_id);
|
|
||||||
|
|
||||||
bool parse_string_lit(std::u32string& out_str);
|
|
||||||
|
|
||||||
bool parse_char_lit(u32& out_char);
|
|
||||||
|
|
||||||
bool parse_literal();
|
|
||||||
|
|
||||||
bool parse_literal_decl();
|
|
||||||
|
|
||||||
// --- Operands & Registers ---
|
|
||||||
|
|
||||||
bool parse_register();
|
|
||||||
|
|
||||||
bool parse_addr_modes();
|
|
||||||
|
|
||||||
bool parse_operand();
|
|
||||||
|
|
||||||
// --- Higher Level Statements ---
|
|
||||||
|
|
||||||
bool parse_instruction();
|
|
||||||
|
|
||||||
bool parse_annotation();
|
|
||||||
|
|
||||||
bool parse_line_content();
|
|
||||||
|
|
||||||
bool parse_program();
|
|
||||||
|
|
||||||
};
|
|
||||||
|
|
||||||
}
|
|
||||||
@@ -16,10 +16,10 @@ namespace spider {
|
|||||||
auto ir = fstack.insert(abs_path);
|
auto ir = fstack.insert(abs_path);
|
||||||
|
|
||||||
// Actually load!
|
// Actually load!
|
||||||
levels.emplace_back(Level {
|
//levels.emplace_back(Level {
|
||||||
.reader = std::make_unique<TextReader>(new FileTextReader(abs_path.string())),
|
// .reader = std::make_unique<TextReader>(new FileTextReader(abs_path.string())),
|
||||||
.source = abs_path.string(),
|
// .source = abs_path.string(),
|
||||||
});
|
//});
|
||||||
parseCurrentLevel();
|
parseCurrentLevel();
|
||||||
|
|
||||||
// alright!
|
// alright!
|
||||||
@@ -28,7 +28,7 @@ namespace spider {
|
|||||||
}
|
}
|
||||||
|
|
||||||
void Assembler::parseCurrentLevel() {
|
void Assembler::parseCurrentLevel() {
|
||||||
auto& lvl = levels.back();
|
//auto& lvl = levels.back();
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -2,7 +2,6 @@
|
|||||||
|
|
||||||
#include <spider/compiler/common.hpp>
|
#include <spider/compiler/common.hpp>
|
||||||
|
|
||||||
#include <spider/compiler/text/TextReader.hpp>
|
|
||||||
#include <spider/compiler/text/Token.hpp>
|
#include <spider/compiler/text/Token.hpp>
|
||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
@@ -21,7 +20,6 @@ namespace spider {
|
|||||||
public:
|
public:
|
||||||
|
|
||||||
set<fs::path> fstack;
|
set<fs::path> fstack;
|
||||||
CompToken root;
|
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
|
||||||
|
|||||||
@@ -109,7 +109,7 @@ namespace spider {
|
|||||||
void TextReader::commit() {
|
void TextReader::commit() {
|
||||||
if (bufferIndex > 0) {
|
if (bufferIndex > 0) {
|
||||||
// Erase everything before the current buffer index
|
// Erase everything before the current buffer index
|
||||||
buffer.erase(buffer.begin(), buffer.begin() + bufferIndex);
|
buffer.erase(buffer.begin(), buffer.begin() + std::ptrdiff_t(bufferIndex));
|
||||||
bufferIndex = 0;
|
bufferIndex = 0;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,33 +1,157 @@
|
|||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
#include <spider/compiler/common.hpp>
|
#include "spider/compiler/common.hpp"
|
||||||
|
|
||||||
#include <variant>
|
#include "spider/compiler/text/utf8.hpp"
|
||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
|
|
||||||
struct LitToken {
|
struct TokenContext {
|
||||||
public:
|
|
||||||
const range at;
|
std::u32string_view input;
|
||||||
const std::u32string type;
|
size_t cursor = 0;
|
||||||
const std::u32string value;
|
|
||||||
|
bool has_more() const { return cursor < input.size(); }
|
||||||
|
|
||||||
|
char32_t peek() const { return input[cursor]; }
|
||||||
|
|
||||||
|
void advance(size_t n = 1) { cursor += n; }
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
struct SimpleToken {
|
struct TokenResult {
|
||||||
public:
|
bool success;
|
||||||
const range at;
|
std::u32string match;
|
||||||
const std::u32string type;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
struct CompToken;
|
class Token {
|
||||||
|
|
||||||
using Token = std::variant<LitToken, SimpleToken, LitToken>;
|
|
||||||
|
|
||||||
struct CompToken {
|
|
||||||
public:
|
public:
|
||||||
const range at;
|
|
||||||
const std::u32string type;
|
virtual ~Token() = default;
|
||||||
vector<Token> inners;
|
|
||||||
|
virtual TokenResult parse(TokenContext& ctx) const = 0;
|
||||||
|
|
||||||
|
};
|
||||||
|
|
||||||
|
class LitToken : public Token {
|
||||||
|
private:
|
||||||
|
|
||||||
|
std::u32string literal;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
explicit LitToken(std::string_view lit) {
|
||||||
|
if(!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
|
||||||
|
}
|
||||||
|
|
||||||
|
explicit LitToken(std::u32string lit) : literal(std::move(lit)) {}
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
TokenResult parse(TokenContext& ctx) const override {
|
||||||
|
if (ctx.cursor + literal.size() > ctx.input.size()) return { false, {} };
|
||||||
|
std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size());
|
||||||
|
if (sub == literal) {
|
||||||
|
ctx.advance(literal.size());
|
||||||
|
return { true, std::u32string(sub) };
|
||||||
|
}
|
||||||
|
return { false, {} };
|
||||||
|
}
|
||||||
|
|
||||||
|
};
|
||||||
|
|
||||||
|
// AndToken references existing static tokens rather than managing their lifecycles
|
||||||
|
class AndToken : public Token {
|
||||||
|
private:
|
||||||
|
|
||||||
|
const Token& lhs;
|
||||||
|
const Token& rhs;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
AndToken(const Token& l, const Token& r) : lhs(l), rhs(r) {}
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
TokenResult parse(TokenContext& ctx) const override {
|
||||||
|
size_t start_pos = ctx.cursor;
|
||||||
|
if (!lhs.parse(ctx).success) return { false, {} };
|
||||||
|
if (!rhs.parse(ctx).success) {
|
||||||
|
ctx.cursor = start_pos; // Backtrack
|
||||||
|
return { false, {} };
|
||||||
|
}
|
||||||
|
return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) };
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
class OrToken : public Token {
|
||||||
|
private:
|
||||||
|
|
||||||
|
const Token& lhs;
|
||||||
|
const Token& rhs;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
OrToken(const Token& l, const Token& r) : lhs(l), rhs(r) {}
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
TokenResult parse(TokenContext& ctx) const override {
|
||||||
|
size_t start_pos = ctx.cursor;
|
||||||
|
auto res = lhs.parse(ctx);
|
||||||
|
if (res.success) return res;
|
||||||
|
ctx.cursor = start_pos; // Backtrack
|
||||||
|
return rhs.parse(ctx);
|
||||||
|
}
|
||||||
|
|
||||||
|
};
|
||||||
|
|
||||||
|
class OptToken : public Token {
|
||||||
|
private:
|
||||||
|
|
||||||
|
const Token& target;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
explicit OptToken(const Token& t) : target(t) {}
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
TokenResult parse(TokenContext& ctx) const override {
|
||||||
|
size_t start_pos = ctx.cursor;
|
||||||
|
if (target.parse(ctx).success) return {
|
||||||
|
true,
|
||||||
|
std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos))
|
||||||
|
};
|
||||||
|
ctx.cursor = start_pos;
|
||||||
|
return { true, std::u32string(ctx.input.substr(start_pos, 0)) };
|
||||||
|
}
|
||||||
|
|
||||||
|
};
|
||||||
|
|
||||||
|
class RepToken : public Token {
|
||||||
|
private:
|
||||||
|
|
||||||
|
const Token& target;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
explicit RepToken(const Token& t) : target(t) {}
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
TokenResult parse(TokenContext& ctx) const override {
|
||||||
|
size_t start_pos = ctx.cursor;
|
||||||
|
while (ctx.has_more()) {
|
||||||
|
size_t loop_start = ctx.cursor;
|
||||||
|
if (!target.parse(ctx).success || ctx.cursor == loop_start) {
|
||||||
|
ctx.cursor = loop_start;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) };
|
||||||
|
}
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -87,13 +87,27 @@ namespace spider {
|
|||||||
}
|
}
|
||||||
|
|
||||||
inline bool charAt(const std::string& str, isize& index, u32& out) {
|
inline bool charAt(const std::string& str, isize& index, u32& out) {
|
||||||
u8 ch0 = u8(str[index]);
|
|
||||||
isize chlen = isValidSeq(str.c_str(), str.size());
|
isize chlen = isValidSeq(str.c_str(), str.size());
|
||||||
if(chlen == 0) return false;
|
if(chlen == 0) return false;
|
||||||
out = decodeArr(str.c_str(), chlen);
|
out = decodeArr(str.c_str(), chlen);
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline bool toUTF32(std::string_view str, std::u32string& out) {
|
||||||
|
isize _i = 0, _size;
|
||||||
|
|
||||||
|
auto cptr = str.cbegin();
|
||||||
|
auto csize = str.size();
|
||||||
|
|
||||||
|
while(_i < csize) {
|
||||||
|
_size = isValidSeq(cptr + _i, csize - _i);
|
||||||
|
if(_size == 0) return false;
|
||||||
|
out += decodeArr(cptr + _i, _size);
|
||||||
|
}
|
||||||
|
|
||||||
|
return _i == csize;
|
||||||
|
}
|
||||||
|
|
||||||
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {}
|
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user