Compare commits

..

13 Commits

Author SHA1 Message Date
Kittycannon 9254c9e0b0 next phase 2026-09-27 01:33:11 -06:00
Kittycannon a3fc9d3d7b removed .vscode 2026-09-27 01:18:15 -06:00
Kittycannon 9388c504c2 cleaned crlf from repo 2026-09-27 01:15:49 -06:00
Kittycannon 5dd0c85d78 ParseTree 2026-09-27 01:05:03 -06:00
Kittycannon df2a7c8289 common fix 2026-09-27 00:17:06 -06:00
Kittycannon 2d9a962cf7 fixed tokens 2026-09-27 00:14:31 -06:00
Kittycannon 6e13571f60 unicode 2026-09-26 22:21:12 -06:00
Kittycannon 8764556e71 synch 2026-08-17 21:17:20 -06:00
Kittycannon 897e155b0e text reader passes tests 2026-08-04 14:22:57 -06:00
Kittycannon fa511520c6 okay, changed structure a lot 2026-08-04 08:05:52 -06:00
Kittycannon 3779848355 time to test 2026-08-02 19:04:52 -06:00
Kittycannon a228c0c59c Merge branch 'main' of git.sintekanalytics.com:SpiderLang/spider-compiler 2026-08-02 05:45:09 -06:00
Kittycannon a3211dbeac removed pygen 2026-08-02 05:44:59 -06:00
22 changed files with 1649 additions and 590 deletions
+4
View File
@@ -0,0 +1,4 @@
# Line endings are LF everywhere, in the repository and in the working tree.
# Without this, core.autocrlf on Windows rewrites every file to CRLF on
# checkout, which mixes endings inside the workspace and breaks shell scripts.
* text=auto eol=lf
+4
View File
@@ -4,3 +4,7 @@
# So hold on
/bin
/out
# VS Code secret folder is used to hold
# user-specific paths and settings
/.vscode
-3
View File
@@ -1,3 +0,0 @@
{
"C_Cpp.default.compilerPath": "C:/msys64/ucrt64/bin/g++.exe"
}
+3
View File
@@ -5,6 +5,9 @@
#Compiler and Linker
CC := g++
# Ensure POSIX utilities in MSYS2 take precedence over Windows System32
export PATH := /usr/bin:/ucrt64/bin:$(PATH)
#The Target Binary Program
TARGET := out.exe
+25
View File
@@ -0,0 +1,25 @@
[CmdletBinding()]
param(
[Parameter(ValueFromRemainingArguments = $true)]
[string[]]$Command
)
$msysRoot = if ($env:MSYS2_ROOT) { $env:MSYS2_ROOT } else { "C:\msys64" }
$bash = Join-Path $msysRoot "usr\bin\bash.exe"
if (-not (Test-Path -LiteralPath $bash)) {
Write-Error "MSYS2 bash was not found at: $bash"
exit 1
}
$env:MSYSTEM = "UCRT64"
$env:CHERE_INVOKING = "1"
if ($Command.Count -eq 0) {
& $bash --login
} else {
$cmdString = $Command -join " "
& $bash -lc $cmdString
}
exit $LASTEXITCODE
-302
View File
@@ -1,302 +0,0 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": 95,
"id": "00e26c5b",
"metadata": {},
"outputs": [],
"source": [
"from lark import Lark, Transformer\n",
"import os"
]
},
{
"cell_type": "code",
"execution_count": 96,
"id": "cc16be1a",
"metadata": {},
"outputs": [],
"source": [
"ebnf_targets = {\n",
" \"assembly\": {\n",
" \"src\": \"./samples/assembly.ebnf\",\n",
" \"dst\": \"./spider/compiler/assembly/AssemblyParser.hpp\",\n",
" \"cnt\": None,\n",
" },\n",
"}\n"
]
},
{
"cell_type": "code",
"execution_count": 97,
"id": "e88d212f",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"\n",
"--- Loading EBNF Targets ---\n",
"✅ Success [assembly]: Loaded './samples/assembly.ebnf' -> Target destination: './spider/compiler/assembly/AssemblyParser.hpp'\n"
]
}
],
"source": [
"print(\"\\n--- Loading EBNF Targets ---\")\n",
"for target_name, paths in ebnf_targets.items():\n",
" src_path = paths[\"src\"]\n",
" dst_path = paths[\"dst\"]\n",
" \n",
" try:\n",
" with open(src_path, \"r\", encoding=\"utf-8\") as file:\n",
" paths[\"cnt\"] = file.read()\n",
" print(f\"✅ Success [{target_name}]: Loaded '{src_path}' -> Target destination: '{dst_path}'\")\n",
" \n",
" except FileNotFoundError:\n",
" print(f\"❌ Error [{target_name}]: Source file not found at '{src_path}'\")\n"
]
},
{
"cell_type": "code",
"execution_count": 98,
"id": "e8095002",
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"from lark import Lark, Transformer\n",
"\n",
"iso_ebnf_meta_grammar = r\"\"\"\n",
" start: rule+\n",
" rule: RULE_NAME \"=\" expression \";\"\n",
" \n",
" ?expression: alternation\n",
" alternation: sequence (\"|\" sequence)*\n",
" \n",
" sequence: item ( [\",\"] item )*\n",
" \n",
" ?item: atom\n",
" | atom \"?\" -> optional\n",
" | atom \"*\" -> repeat\n",
" | \"[\" expression \"]\" -> optional\n",
" | \"{\" expression \"}\" -> repeat\n",
" \n",
" ?atom: RULE_NAME -> call_rule\n",
" | TERMINAL -> match_terminal\n",
" | SPECIAL_SEQ -> handle_special\n",
" | \"(\" expression \")\" -> group\n",
"\n",
" RULE_NAME: /[a-zA-Z_][a-zA-Z0-9_]*/\n",
" TERMINAL: /\"[^\"\\\\]*(?:\\\\.[^\"\\\\]*)*\"/ | /'[^'\\\\]*(?:\\\\.[^'\\\\]*)*'/\n",
" SPECIAL_SEQ: /\\?[\\s\\S]*?\\?/\n",
" COMMENT: /\\(\\*([\\s\\S]*?)\\*\\)/\n",
"\n",
" %import common.WS\n",
" %ignore WS\n",
" %ignore COMMENT\n",
"\"\"\"\n",
"\n",
"class AssemblyCppGenerator(Transformer):\n",
" def start(self, rules):\n",
" cpp_functions = \"\\n\\n\".join(rules)\n",
" return f\"\"\"#pragma once\n",
"\n",
"#include <iostream>\n",
"#include <string>\n",
"#include <vector>\n",
"#include <stdexcept>\n",
"\n",
"class AssemblyParser {{\n",
"private:\n",
" std::string src;\n",
" size_t pos = 0;\n",
"\n",
" std::string peek_str(size_t len) {{\n",
" if (pos + len <= src.length()) return src.substr(pos, len);\n",
" return src.substr(pos);\n",
" }}\n",
"\n",
" char peek() {{ return pos < src.length() ? src[pos] : '\\\\0'; }}\n",
" \n",
" void match_char(char expected) {{\n",
" if (peek() == expected) pos++;\n",
" else throw std::runtime_error(\"Unexpected token matching character\");\n",
" }}\n",
"\n",
" void match_string(std::string expected) {{\n",
" if (peek_str(expected.length()) == expected) pos += expected.length();\n",
" else throw std::runtime_error(\"Unexpected token matching string: \" + expected);\n",
" }}\n",
"\n",
" bool isUTF8Alpha() {{ return isalpha(peek()); }}\n",
" bool isWhithespaceCharNotCrLf() {{ return peek() == ' ' || peek() == '\\\\t'; }}\n",
" bool isUTF8CharNotCrLf() {{ return peek() != '\\\\r' && peek() != '\\\\n' && peek() != '\\\\0'; }}\n",
" bool isUTF8CharLitCont() {{ return peek() != '\\'' && peek() != '\\\\\\\\'; }}\n",
" bool isUTF8StringLitCont() {{ return peek() != '\"' && peek() != '\\\\\\\\'; }}\n",
"\n",
"public:\n",
" AssemblyParser(std::string input) : src(input) {{}}\n",
"\n",
" void parse() {{\n",
" parse_program(); \n",
" if (pos < src.length()) throw std::runtime_error(\"Trailing characters left unparsed.\");\n",
" std::cout << \"Assembly source compiled cleanly!\" << std::endl;\n",
" }}\n",
"\n",
"{cpp_functions}\n",
"}};\n",
"\"\"\"\n",
"\n",
" def rule(self, args):\n",
" name, expr = args\n",
" return f\" void parse_{name}() {{\\n{expr}\\n }}\"\n",
"\n",
" # FIX 1: Explicitly handle choice logic using C++ style paths\n",
" def alternation(self, items):\n",
" code_lines = []\n",
" for i, item in enumerate(items):\n",
" # Clean up padding whitespace if any\n",
" clean_item = str(item).strip()\n",
" if not clean_item: continue\n",
" \n",
" # Since lookahead processing requires FIRST sets, we scaffold a sequential fallback\n",
" if i == 0:\n",
" code_lines.append(f\" if (/* option {i+1} */ true) {{\\n {clean_item}\\n }}\")\n",
" else:\n",
" code_lines.append(f\" else if (/* option {i+1} */ true) {{\\n {clean_item}\\n }}\")\n",
" return \"\\n\".join(code_lines)\n",
"\n",
" def sequence(self, items):\n",
" flattened_items = []\n",
" for item in items:\n",
" if isinstance(item, list):\n",
" for sub_item in item:\n",
" if sub_item: flattened_items.append(str(sub_item).strip())\n",
" elif item:\n",
" flattened_items.append(str(item).strip())\n",
" return \"\\n\".join(f\" {item}\" for item in flattened_items if item)\n",
"\n",
" def call_rule(self, token):\n",
" rule_name = token[0].value if isinstance(token, list) else token.value\n",
" return f\"parse_{rule_name}();\"\n",
"\n",
" # FIX 2: Generate match_string instead of match_char for multi-char string keywords like \"include\"\n",
" def match_terminal(self, token):\n",
" raw_token_str = token[0].value if isinstance(token, list) else token.value\n",
" raw_val = raw_token_str[1:-1]\n",
" \n",
" if raw_val == r\"\\r\": return \"match_char('\\\\r');\"\n",
" if raw_val == r\"\\n\": return \"match_char('\\\\n');\"\n",
" if raw_val == r\"\\t\": return \"match_char('\\\\t');\"\n",
" if raw_val == r\"\\\\\": return \"match_char('\\\\\\\\');\"\n",
" if not raw_val: return \"// Empty string match\"\n",
" \n",
" if len(raw_val) > 1:\n",
" return f\"match_string(\\\"{raw_val}\\\");\"\n",
" return f\"match_char('{raw_val}');\"\n",
"\n",
" def handle_special(self, token):\n",
" raw_string = token[0].value if isinstance(token, list) else token.value\n",
" func_name = raw_string.strip('?').strip()\n",
" return f\"if ({func_name}()) {{ pos++; }} else {{ throw std::runtime_error(\\\"Failed validation for {func_name}\\\"); }}\"\n",
"\n",
" def optional(self, args):\n",
" content = args[0] if not isinstance(args[0], list) else \"\\n \".join(args[0])\n",
" return f\"// Optional block\\n if (/* lookahead check */ true) {{\\n {content}\\n }}\"\n",
"\n",
" def repeat(self, args):\n",
" content = args[0] if not isinstance(args[0], list) else \"\\n \".join(args[0])\n",
" return f\"// Repeat block\\n while (/* lookahead check */ true) {{\\n {content}\\n }}\"\n",
"\n",
" def group(self, args):\n",
" # Flatten grouped elements cleanly to strings\n",
" if isinstance(args, list):\n",
" return \"\\n\".join(str(x) for x in args)\n",
" return str(args)"
]
},
{
"cell_type": "code",
"execution_count": 99,
"id": "558915ff",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"--- Starting C++ Compilation Loop ---\n",
"Parsing and converting target rule sets for: assembly\n",
"🎉 Code generation complete! Output stored in './spider/compiler/assembly/AssemblyParser.hpp'\n"
]
}
],
"source": [
"print(\"--- Starting C++ Compilation Loop ---\")\n",
"\n",
"try:\n",
" meta_parser = Lark(iso_ebnf_meta_grammar, parser='lalr')\n",
" \n",
" for name, target in ebnf_targets.items():\n",
" print(f\"Parsing and converting target rule sets for: {name}\")\n",
" \n",
" # Build the compiler AST tree from your exact text\n",
" syntax_tree = meta_parser.parse(target[\"cnt\"])\n",
" \n",
" # Transform the AST structural nodes into pure C++ Source strings\n",
" compiler_transformer = AssemblyCppGenerator()\n",
" compiled_cpp_header = compiler_transformer.transform(syntax_tree)\n",
" \n",
" # Output directly to your destination path\n",
" os.makedirs(os.path.dirname(target[\"dst\"]), exist_ok=True)\n",
" with open(target[\"dst\"], \"w\", encoding=\"utf-8\") as f:\n",
" f.write(compiled_cpp_header)\n",
" \n",
" print(f\"🎉 Code generation complete! Output stored in '{target['dst']}'\")\n",
"\n",
"except Exception as e:\n",
" print(f\"❌ Failed to process custom architecture. Error details: \\n{e}\")\n",
"\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "366688c3",
"metadata": {},
"outputs": [],
"source": []
},
{
"cell_type": "code",
"execution_count": null,
"id": "cd1aca3f",
"metadata": {},
"outputs": [],
"source": []
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3",
"version": "3.13.7"
}
},
"nbformat": 4,
"nbformat_minor": 5
}
+1
View File
@@ -5,6 +5,7 @@
}
],
"settings": {
"files.eol": "\n",
"gitlens.remotes": [
{
"domain": "git.sintekanalytics.com",
View File
+7
View File
@@ -0,0 +1,7 @@
#pragma once
namespace spider {
}
+272 -56
View File
@@ -1,80 +1,296 @@
#include <iostream>
#include <spider/compiler/Compiler.hpp>
#include <spider/compiler/common.hpp>
#include <spider/compiler/text/utf8.hpp>
#include <spider/compiler/text/unicode.hpp>
#include <spider/compiler/text/TextReader.hpp>
#include <spider/compiler/assembler/AsmEBNF.hpp>
#include <spider/compiler/text/ParseTree.hpp>
using namespace spider;
// ============================================================================
// Compiler Entry Points
// ============================================================================
namespace spider {
TokenFactory& assemblyGrammar() {
static TokenFactory grammar;
static bool loaded = false;
if (!loaded) {
asm_ebnf::initTokens(grammar);
loaded = true;
}
return grammar;
}
bool compileProgram(const std::string& source, ParseTree& out) {
return out.parse(asm_ebnf::program, source);
}
}
// Test runner helper
void run_test(const std::string& name, const std::string& input) {
std::cout << "========================================\n";
std::cout << " TEST: " << name << "\n";
std::cout << "========================================\n";
// ============================================================================
// Test Bookkeeping
// ============================================================================
spider::pos tracking_pos;
spider::utf8::hexdump(input.data(), input.size(), tracking_pos, std::cout);
std::cout << "\n";
// Suite wide bookkeeping, so a failing case can never hide behind a clean exit code.
static isize total_tests = 0;
static isize failed_tests = 0;
static void report(bool ok, const std::string& name) {
++total_tests;
if (ok) {
std::cout << "[PASS] " << name << "\n";
} else {
++failed_tests;
std::cerr << "[FAIL] " << name << "\n";
}
}
void utf8sequences() {
// Permutation 1: Standard, valid mixed UTF-8 sequences (1, 2, 3, and 4 bytes)
// - 'A' -> 1 byte (U+0041)
// - '¢' (cents) -> 2 bytes (U+00A2)
// - '€' (euro) -> 3 bytes (U+20AC)
// - '𐍈' (gothic) -> 4 bytes (U+10348)
run_test("Valid Mixed Length Sequences", "A\xC2\xA2\xE2\x82\xAC\xF0\x90\x8D\x88");
// Inline evaluator that executes the reader and prints standard output
static bool run_test_case(const Token* token, const std::string& input, bool expectedSuccess, const std::u32string& expectedMatch = U"") {
StringTextReader reader(input);
TokenResult res = token->test(reader);
// Permutation 2: Embedded Control Characters
// Should display mnemonics like (HT), (LF), (CR) without breaking formatting
run_test("ASCII Control Characters", "Text\tWith\r\nNewlines");
bool status_ok = (res.success == expectedSuccess);
bool match_ok = (!expectedSuccess) || expectedMatch.empty() || (res.flatMatch() == expectedMatch);
// Permutation 3: Invalid Lead Byte
// The byte 0xFF is structurally illegal under any UTF-8 definition.
// Expected behavior: Display single byte as INVALID LEAD, shift 1 byte over.
run_test("Invalid Lead Byte (0xFF)", "ABC\xFFXYZ");
++total_tests;
if (status_ok && match_ok) {
std::cout << "[PASS] Input: \"" << input << "\" -> "
<< (res.success ? "SUCCESS" : "FAILURE")
<< (res.success ? (" (Matched: \"" + unicode::toUTF8(res.flatMatch()) + "\")") : "")
<< "\n";
return true;
}
// Permutation 4: Invalid Continuation Sequence
// A 3-byte header (\xE2) where the second byte (\x00) is a bad continuation.
// Expected behavior: Show the entire sequence up to 'm' bytes, flag as INVALID SEQUENCE.
run_test("Invalid Continuation Structure", std::string("Before \xE2\x00\xAC After", 16));
// Permutation 5: Truncated Sequence at End-of-Buffer
// A 4-byte emoji header (\xF0\x9F) but the string completely cuts off.
// Expected behavior: Display remaining space placeholders as '??' -> TRUNCATED SEQUENCE.
run_test("Truncated Sequence (Missing trailing bytes)", "Hello \xF0\x9F");
// Permutation 6: Overlong Encoding Security Vulnerability
// Attempting to write ASCII 'I' (normally 0x49) using 2 bytes: \xC1\x89
// Expected behavior: Caught by constraints checks, flagged as INVALID SEQUENCE.
run_test("Security Hack: Overlong Encoding", "Safe\xC1\x89Hack");
// Permutation 7: Out-of-bounds / Restricted Ranges
// - \xED\xA0\x80 is a UTF-16 Surrogate (U+D800)
// - \xF4\x90\x80\x80 is outside valid Unicode space (> U+10FFFF)
// Expected behavior: Flagged securely as INVALID SEQUENCE.
run_test("Security Hack: Restricted Ranges (Surrogates & Out-of-bounds)", "Surrogate: \xED\xA0\x80 MaxBounds: \xF4\x90\x80\x80");
std::cerr << "[FAIL] Input: \"" << input << "\"\n"
<< " Expected Success: " << (expectedSuccess ? "true" : "false") << ", Got: " << (res.success ? "true" : "false") << "\n";
if (expectedSuccess && !expectedMatch.empty()) {
std::cerr << " Expected Match: \"" << unicode::toUTF8(expectedMatch) << "\", Got: \"" << unicode::toUTF8(res.flatMatch()) << "\"\n";
}
++failed_tests;
return false;
}
// ============================================================================
// Core Combinator & Grammar Tests
// ============================================================================
void test_primitives_and_literals(TokenFactory& tf) {
std::cout << "\n--- Testing Primitives & Literals ---\n";
Token* hello = tf.lit("hello");
run_test_case(hello, "hello world", true, U"hello");
run_test_case(hello, "hell", false);
run_test_case(asm_ebnf::letter, "a", true);
run_test_case(asm_ebnf::letter, "f", true);
run_test_case(asm_ebnf::letter, "A", true);
run_test_case(asm_ebnf::letter, "Y", true);
run_test_case(asm_ebnf::letter, "9", false);
run_test_case(asm_ebnf::digit, "0", true);
run_test_case(asm_ebnf::digit, "7", true);
run_test_case(asm_ebnf::digit, "x", false);
run_test_case(asm_ebnf::hex_digit, "F", true);
run_test_case(asm_ebnf::hex_digit, "g", false);
}
void test_choice_and_seq(TokenFactory& tf) {
std::cout << "\n--- Testing Choice & Sequence ---\n";
Token* seq_test = tf.seq({ tf["foo"], tf["bar"] });
run_test_case(seq_test, "foobar", true, U"foobar");
run_test_case(seq_test, "foobaz", false);
Token* choice_test = tf.choice({ tf["apple"], tf["banana"] });
run_test_case(choice_test, "banana", true, U"banana");
run_test_case(choice_test, "cherry", false);
}
void test_opt_and_rep(TokenFactory& tf) {
std::cout << "\n--- Testing Optional & Repeat ---\n";
Token* opt_test = tf.seq({ tf.opt(tf["+"]), asm_ebnf::digit });
run_test_case(opt_test, "+5", true, U"+5");
run_test_case(opt_test, "5", true, U"5");
Token* rep_digits = tf.rep(asm_ebnf::digit);
run_test_case(rep_digits, "12345abc", true, U"12345");
run_test_case(rep_digits, "abc", true, U"");
}
void test_literals(TokenFactory& tf) {
std::cout << "\n--- Testing Grammatical Literals ---\n";
run_test_case(asm_ebnf::identifier, "valid_var1", true, U"valid_var1");
run_test_case(asm_ebnf::identifier, "_private", true, U"_private");
run_test_case(asm_ebnf::identifier, "123invalid", false);
run_test_case(asm_ebnf::decimal_lit, "1234", true);
run_test_case(asm_ebnf::decimal_lit, "-567L", true);
run_test_case(asm_ebnf::hex_lit, "0x1A2B", true);
run_test_case(asm_ebnf::octal_lit, "0c755", true);
run_test_case(asm_ebnf::binary_lit, "0b10101", true);
run_test_case(asm_ebnf::float_lit, "3.14159F", true);
run_test_case(asm_ebnf::float_lit, "1e-10D", true);
run_test_case(asm_ebnf::char_lit, "'a'", true);
run_test_case(asm_ebnf::char_lit, "'\\n'", true);
run_test_case(asm_ebnf::string_lit, "\"Hello World\"", true);
run_test_case(asm_ebnf::string_lit, "\"Escape \\\" Test\"", true);
}
void test_addressing_modes(TokenFactory& tf) {
std::cout << "\n--- Testing Addressing Modes ---\n";
run_test_case(asm_ebnf::register_tok, "R0", true);
run_test_case(asm_ebnf::register_tok, "R15", false); // R15 does not exist!
run_test_case(asm_ebnf::addrm_ind, "[ 0x1000 ]", true);
run_test_case(asm_ebnf::addrm_ptr, "[R1]", true);
run_test_case(asm_ebnf::addrm_idx, "[R1 + 4]", true);
run_test_case(asm_ebnf::addrm_sca, "[R1 + R2 * 4]", true);
run_test_case(asm_ebnf::addrm_dis, "[ R1 + R2 * 4 + 16 ]", true);
}
void test_instructions_and_lines(TokenFactory& tf) {
std::cout << "\n--- Testing Instructions & Lines ---\n";
run_test_case(asm_ebnf::instruction, "NOP", true);
run_test_case(asm_ebnf::instruction, "MOV R1, [R2 + 4]", true);
run_test_case(asm_ebnf::instruction, "ADD R0, 100", true);
run_test_case(asm_ebnf::preprocessor, "#define MAX_BUF", true);
run_test_case(asm_ebnf::annotation, "@inline", true);
run_test_case(asm_ebnf::annotation, "@align(4)", true);
run_test_case(asm_ebnf::annotation, "@section(name=\"text\", flags=1)", true);
run_test_case(asm_ebnf::line, "main: MOV R0, R1 ; copy reg\n", true);
run_test_case(asm_ebnf::line, " @deprecated NOP\n", true);
run_test_case(asm_ebnf::line, "; only a comment line\n", true);
}
void test_full_program(TokenFactory& tf) {
std::cout << "\n--- Testing Full Program Parser ---\n";
std::string asm_code =
"#include stdio\n"
"\n"
"start:\n"
" MOV R1, 0x20 ; Load constant\n"
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
" JMP start\n";
ParseTree tree;
report(compileProgram(asm_code, tree), "full assembly program parses completely");
}
void test_parse_tree(TokenFactory& tf) {
std::cout << "\n--- Testing Parse Tree Inspection ---\n";
const std::string asm_code =
"#include stdio\n"
"\n"
"start:\n"
" MOV R1, 0x20 ; Load constant\n"
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
" JMP start\n";
ParseTree tree;
report(compileProgram(asm_code, tree), "program becomes a tree of nodes");
ParseNode* root = tree.root();
report(root != nullptr, "tree exposes a root node");
if (root == nullptr) return;
report(root->hasTag("program"), "root node is tagged as program");
report(root->textUtf8() == asm_code, "root text covers the whole source");
report(root->childCount() > 0, "root holds child nodes");
vector<ParseNode*> lines = root->findAll("line");
report(lines.size() == 6, "program holds one node per line");
if (lines.size() == 6) {
report(lines.front()->parentNode() == root, "a line knows its parent");
report(lines.front()->depth() == 1, "lines sit one level under the root");
report(lines.front()->nextSibling() == lines[1], "sibling walk reaches the next line");
report(lines[1]->previousSibling() == lines.front(), "sibling walk goes back");
report(lines.back()->nextSibling() == nullptr, "the last line has no next sibling");
ParseNode* preprocessor = lines.front()->firstChild("preprocessor");
report(preprocessor != nullptr, "first line holds a preprocessor child");
report(preprocessor != nullptr && preprocessor->textUtf8() == "#include stdio", "preprocessor text is readable");
report(lines[2]->firstChild("line_label") != nullptr, "label line is recognized");
ParseNode* label = lines[2]->find("label");
report(label != nullptr && label->textUtf8() == "start:", "label text is readable");
ParseNode* mov = lines[3]->find("instruction");
report(mov != nullptr, "instruction is found inside its line");
if (mov != nullptr) {
report(mov->firstChild("opcode") != nullptr, "instruction exposes its opcode child");
ParseNode* operands = mov->firstChild("operand_list");
report(operands != nullptr && operands->childCount() == 2, "operand list exposes both operands");
ParseNode* hexlit = mov->find("hex_lit");
report(hexlit != nullptr && hexlit->textUtf8() == "0x20", "hex literal text is readable");
}
ParseNode* comment = lines[3]->firstChild("comment");
report(comment != nullptr && comment->textUtf8() == "; Load constant", "trailing comment is captured");
report(lines[4]->firstChild("line_annotation") != nullptr, "annotation line is recognized");
ParseNode* displacement = lines[4]->find("addrm_dis");
report(displacement != nullptr && displacement->textUtf8() == "[R2 + R3 * 8 + 4]", "displacement operand is readable");
report(tree.count("register") == 4, "every register of the program is reachable");
}
ParseTree single;
report(single.parse(asm_ebnf::instruction, "MOV R1, [R2 + 4]"), "a single instruction parses on its own");
ParseNode* instruction = single.root();
report(instruction != nullptr && instruction->hasTag("instruction"), "instruction root is tagged");
if (instruction != nullptr) {
ParseNode* operands = instruction->firstChild("operand_list");
report(operands != nullptr && operands->childCount() == 2, "operand list holds two operands");
ParseNode* second = operands == nullptr ? nullptr : operands->childAt(1);
report(second != nullptr && second->textUtf8() == "[R2 + 4]", "operand text is readable");
report(second != nullptr && second->ancestor("instruction") == instruction, "an operand can climb back to its instruction");
}
std::cout << "\n--- Parsed Tree ---\n" << tree.toString() << "\n";
}
// ============================================================================
// Main Execution
// ============================================================================
int main() {
std::string test = "UTF-STR WITH EMOJIS 😀🚀";
// Extracted UTF-8 chars:
std::u32string out;
bool r = spider::utf8::toUTF32(test, out);
TokenFactory tf;
assemblyGrammar();
std::cout << "INPUT: " << test << std::endl;
std::cout << "RESULT: " << int(r) << std::endl;
for(spider::u32 ch : out) {
std::cout << ch << " ";
std::cout << "Running Token Framework Tests...\n";
test_primitives_and_literals(tf);
test_choice_and_seq(tf);
test_opt_and_rep(tf);
test_literals(tf);
test_addressing_modes(tf);
test_instructions_and_lines(tf);
test_full_program(tf);
test_parse_tree(tf);
std::cout << "\n========================================\n";
std::cout << "Test Results: " << (total_tests - failed_tests) << "/" << total_tests << " passed.\n";
std::cout << "========================================\n";
if (failed_tests > 0) {
std::cerr << failed_tests << " test(s) FAILED.\n";
return 1;
}
std::cout << std::endl;
std::cout << "Happy Day!" << std::endl;
spider::utf8::hexdump(test.data(), test.size(), spider::pos(), std::cout);
std::cout << std::endl;
utf8sequences();
std::cout << "All token framework tests passed successfully!\n";
return 0;
}
+20 -3
View File
@@ -1,10 +1,27 @@
#pragma
#pragma once
#include <spider/compiler/common.hpp>
#include <spider/compiler/text/Token.hpp>
#include <spider/compiler/text/ParseTree.hpp>
namespace spider {
class Token;
class RootToken;
/**
* @brief The token factory that owns the assembly grammar, loaded on first use.
* @details Every rule of the assembly language lives in this factory, so a caller
* can test a single rule or a whole program against the same grammar.
*/
TokenFactory& assemblyGrammar();
/**
* @brief Parses a whole assembly program into an inspectable tree of nodes.
* @details The grammar has to consume the entire source, so a malformed program
* is rejected instead of quietly producing a partial tree. On success the
* resulting tree can be walked like a document: program, line, label,
* instruction, operands, literals and comments.
* @returns True on success, with the parsed nodes left inside out.
*/
bool compileProgram(const std::string& source, ParseTree& out);
}
+186 -31
View File
@@ -2,12 +2,14 @@
namespace spider::asm_ebnf {
// Char Functions
bool isUTF8Alpha(u32 ch) {
return false;
return (u32('a') <= ch && ch <= u32('z')) || (u32('A') <= ch && ch <= u32('Z'));
}
bool isWhithespaceCharNotCrLf(u32 ch) {
return false;
return ch == u32(' ') || ch == u32('\t');
}
bool isUTF8CharNotCrLf(u32 ch) {
@@ -15,44 +17,197 @@ namespace spider::asm_ebnf {
}
bool isUTF8CharLitCont(u32 ch) {
return ch != u32('\'');
return ch != u32('\'') && ch != u32('\\') && ch != u32('\r') && ch != u32('\n');
}
bool isUTF8StringLitCont(u32 ch) {
return ch != u32('"');
return ch != u32('"') && ch != u32('\\') && ch != u32('\r') && ch != u32('\n');
}
LitToken numbers[] = {
"0",
"1","2","3",
"4","5","6",
"7","8","9",
};
const Token* letter;
const Token* digit;
const Token* alpha_num_char;
LitToken hex_digits[][2] = {
{"A", "a"},
{"B", "b"},
{"C", "c"},
{"D", "d"},
{"E", "e"},
{"F", "f"},
};
const Token* hex_digit;
const Token* octal_digit;
const Token* binary_digit;
LitToken new_line[] = { "\r\n", "\r", "\n" };
const Token* ws_char;
const Token* ws_optional;
const Token* whitespace;
const Token* newline;
const Token* utf8_char;
LitToken symbols[] = {
"\\", "\'", "\"",
"_" , ";" , "(" ,
"#",
"$" , "." , "+" , "-", ",", ")", "@", ":",
};
const Token* char_escape;
const Token* char_content;
const Token* char_lit;
LitToken lit_letter[] = {
"x", "c", "b"
};
const Token* string_char;
const Token* string_lit;
LitToken type_letter[] = {
"B", "S", "I", "L", "F", "D"
};
const Token* identifier;
const Token* comment;
const Token* sign;
const Token* exponent_marker;
const Token* exponent;
const Token* decimal_lit;
const Token* float_lit;
const Token* hex_lit;
const Token* octal_lit;
const Token* binary_lit;
const Token* literal;
const Token* literal_cast;
const Token* literal_decl;
const Token* register_tok;
const Token* addrm_ind;
const Token* addrm_ptr;
const Token* addrm_idx;
const Token* addrm_sca;
const Token* addrm_dis;
const Token* addr_modes;
const Token* operand;
const Token* opcode;
const Token* operand_list;
const Token* instruction;
const Token* annotation_named;
const Token* annotation_arg;
const Token* annotation_args;
const Token* annotation_pars;
const Token* annotation;
const Token* preprocessor_val;
const Token* preprocessor;
const Token* label;
const Token* line_label;
const Token* line_annotation;
const Token* line_content;
const Token* line;
const Token* line_last;
const Token* program;
void initTokens(TokenFactory& tf) {
// (* Characters & Basic Predicates *)
letter = tf.fn(isUTF8Alpha);
digit = tf.choice("0123456789");
alpha_num_char = tf.choice({ letter, digit });
hex_digit = tf.choice("0123456789ABCDEFabcdef");
octal_digit = tf.choice("01234567");
binary_digit = tf.choice("01");
ws_char = tf.fn(isWhithespaceCharNotCrLf);
ws_optional = tf.rep(ws_char);
whitespace = tf.seq({ ws_char, tf.rep(ws_char) });
newline = tf.tag(tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }), "newline", true);
utf8_char = tf.fn(isUTF8CharNotCrLf);
char_escape = tf.seq({ tf["\\"], utf8_char });
char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
char_lit = tf.tag(tf.seq({ tf["'"], char_content, tf["'"] }), "char_lit", true);
string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
string_lit = tf.tag(tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }), "string_lit", true);
// (* Literals *)
identifier = tf.tag(tf.seq({
tf.choice({ letter, tf["_"] }),
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
}), "identifier", true);
comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true);
sign = tf.choice("+-");
exponent_marker = tf.choice("eE");
exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
decimal_lit = tf.tag(tf.seq({
tf.opt(sign),
digit,
tf.rep(digit),
tf.opt(tf.choice("BSIL"))
}), "decimal_lit", true);
float_lit = tf.tag(tf.seq({
tf.opt(sign),
tf.choice({
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ digit, tf.rep(digit), exponent })
}),
tf.opt(tf.choice("FD"))
}), "float_lit", true);
hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true);
octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true);
binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true);
literal = tf.tag(tf.choice({ hex_lit, octal_lit, binary_lit, float_lit, decimal_lit, string_lit, char_lit }), "literal");
literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast");
literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl");
// (* Operands *)
register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char, tf.not_(tf.choice({ alpha_num_char, tf["_"] })) }), "register", true);
addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true);
addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true);
addrm_idx = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_idx", true);
addrm_sca = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_sca", true);
addrm_dis = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_dis", true);
addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm");
operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand");
// (* Generalized Instructions *)
opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
operand_list = tf.tag(tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }), "operand_list", true);
instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction", true);
// (* Added Preprocessor, Annotation *)
annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named", true);
annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg", true);
annotation_args = tf.tag(tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }), "annotation_args", true);
annotation_pars = tf.tag(tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }), "annotation_pars", true);
annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation", true);
preprocessor_val = tf.choice({ identifier, literal_decl });
preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor", true);
// (* Line Structure & Program *)
label = tf.tag(tf.seq({ identifier, tf[":"] }), "label", true);
line_label = tf.tag(tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }), "line_label", true);
line_annotation = tf.tag(tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }), "line_annotation", true);
line_content = tf.tag(tf.choice({ preprocessor, line_annotation, line_label, instruction }), "line_content", true);
line = tf.tag(tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }), "line", true);
line_last = tf.tag(tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }), "line_last", true);
program = tf.tag(tf.seq({ tf.rep(line), tf.opt(line_last) }), "program", true);
}
}
+72 -1
View File
@@ -4,6 +4,77 @@
namespace spider::asm_ebnf {
extern LitToken letter;
extern const Token* letter;
extern const Token* digit;
extern const Token* alpha_num_char;
extern const Token* hex_digit;
extern const Token* octal_digit;
extern const Token* binary_digit;
extern const Token* ws_char;
extern const Token* ws_optional;
extern const Token* whitespace;
extern const Token* newline;
extern const Token* utf8_char;
extern const Token* char_escape;
extern const Token* char_content;
extern const Token* char_lit;
extern const Token* string_char;
extern const Token* string_lit;
extern const Token* identifier;
extern const Token* comment;
extern const Token* sign;
extern const Token* exponent_marker;
extern const Token* exponent;
extern const Token* decimal_lit;
extern const Token* float_lit;
extern const Token* hex_lit;
extern const Token* octal_lit;
extern const Token* binary_lit;
extern const Token* literal;
extern const Token* literal_cast;
extern const Token* literal_decl;
extern const Token* register_tok;
extern const Token* addrm_ind;
extern const Token* addrm_ptr;
extern const Token* addrm_idx;
extern const Token* addrm_sca;
extern const Token* addrm_dis;
extern const Token* addr_modes;
extern const Token* operand;
extern const Token* opcode;
extern const Token* operand_list;
extern const Token* instruction;
extern const Token* annotation_named;
extern const Token* annotation_arg;
extern const Token* annotation_args;
extern const Token* annotation_pars;
extern const Token* annotation;
extern const Token* preprocessor_val;
extern const Token* preprocessor;
extern const Token* label;
extern const Token* line_label;
extern const Token* line_annotation;
extern const Token* line_content;
extern const Token* line;
extern const Token* line_last;
extern const Token* program;
void initTokens(TokenFactory& tf);
}
+7
View File
@@ -1,5 +1,10 @@
#pragma once
// Prevents conflicts if the runtime
// is included, which includes its own
// identical common.hpp
#ifndef SPIDER_RUNTIME_COMMON
#include <cstdint>
#include <vector>
#include <deque>
@@ -62,3 +67,5 @@ namespace spider {
};
}
#endif
+334
View File
@@ -0,0 +1,334 @@
#include "ParseTree.hpp"
namespace spider {
// ============================================================================
// Internal Helpers
// ============================================================================
/**
* @brief Renders control characters as escapes, so a node can be printed on one line.
*/
static std::string escapeText(std::string_view raw) {
std::string out;
out.reserve(raw.size());
for (char c : raw) {
switch (c) {
case '\n': out += "\\n"; break;
case '\r': out += "\\r"; break;
case '\t': out += "\\t"; break;
default: out += c; break;
}
}
return out;
}
// ============================================================================
// ParseNode Navigation
// ============================================================================
isize ParseNode::depth() const {
isize levels = 0;
for (const ParseNode* n = parentNode(); n != nullptr; n = n->parentNode()) ++levels;
return levels;
}
const ParseNode* ParseNode::parentNode() const {
if (owner == nullptr || parent == ParseNode::npos) return nullptr;
return owner->resolve(parent);
}
const ParseNode* ParseNode::firstChild() const {
if (owner == nullptr || first_child == ParseNode::npos) return nullptr;
return owner->resolve(first_child);
}
const ParseNode* ParseNode::lastChild() const {
if (owner == nullptr || last_child == ParseNode::npos) return nullptr;
return owner->resolve(last_child);
}
const ParseNode* ParseNode::nextSibling() const {
if (owner == nullptr || next_sibling == ParseNode::npos) return nullptr;
return owner->resolve(next_sibling);
}
const ParseNode* ParseNode::previousSibling() const {
if (owner == nullptr || prev_sibling == ParseNode::npos) return nullptr;
return owner->resolve(prev_sibling);
}
const ParseNode* ParseNode::childAt(isize index) const {
isize seen = 0;
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
if (seen == index) return c;
++seen;
}
return nullptr;
}
ParseNode* ParseNode::parentNode() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->parentNode()); }
ParseNode* ParseNode::firstChild() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->firstChild()); }
ParseNode* ParseNode::lastChild() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->lastChild()); }
ParseNode* ParseNode::nextSibling() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->nextSibling()); }
ParseNode* ParseNode::previousSibling() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->previousSibling()); }
ParseNode* ParseNode::childAt(isize index) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->childAt(index)); }
// ============================================================================
// ParseNode Navigation Filtered By Tag
// ============================================================================
const ParseNode* ParseNode::firstChild(std::string_view name) const {
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
if (c->hasTag(name)) return c;
}
return nullptr;
}
const ParseNode* ParseNode::lastChild(std::string_view name) const {
const ParseNode* found = nullptr;
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
if (c->hasTag(name)) found = c;
}
return found;
}
const ParseNode* ParseNode::nextSibling(std::string_view name) const {
for (const ParseNode* s = nextSibling(); s != nullptr; s = s->nextSibling()) {
if (s->hasTag(name)) return s;
}
return nullptr;
}
const ParseNode* ParseNode::ancestor(std::string_view name) const {
for (const ParseNode* p = parentNode(); p != nullptr; p = p->parentNode()) {
if (p->hasTag(name)) return p;
}
return nullptr;
}
ParseNode* ParseNode::firstChild(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->firstChild(name)); }
ParseNode* ParseNode::lastChild(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->lastChild(name)); }
ParseNode* ParseNode::nextSibling(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->nextSibling(name)); }
ParseNode* ParseNode::ancestor(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->ancestor(name)); }
// ============================================================================
// ParseNode Queries
// ============================================================================
const ParseNode* ParseNode::find(std::string_view name) const {
if (hasTag(name)) return this;
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
if (const ParseNode* hit = c->find(name)) return hit;
}
return nullptr;
}
vector<const ParseNode*> ParseNode::findAll(std::string_view name) const {
vector<const ParseNode*> hits;
if (hasTag(name)) hits.push_back(this);
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
for (const ParseNode* hit : c->findAll(name)) hits.push_back(hit);
}
return hits;
}
isize ParseNode::count(std::string_view name) const { return findAll(name).size(); }
ParseNode* ParseNode::find(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->find(name)); }
vector<ParseNode*> ParseNode::findAll(std::string_view name) {
vector<const ParseNode*> hits = static_cast<const ParseNode*>(this)->findAll(name);
vector<ParseNode*> out;
out.reserve(hits.size());
for (const ParseNode* hit : hits) out.push_back(const_cast<ParseNode*>(hit));
return out;
}
// ============================================================================
// ParseNode Rendering
// ============================================================================
std::string ParseNode::describe(isize depth) const {
const std::string pad(depth * 2, ' ');
if (!tag.has_value()) {
if (isLeaf()) return pad + escapeText(ownTextUtf8());
std::string out;
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
out += c->describe(depth) + "\n";
}
if (!out.empty()) out.pop_back();
return out;
}
const std::string name(*tag);
if (isLeaf()) return pad + "<" + name + ">" + escapeText(ownTextUtf8()) + "</" + name + ">";
std::string out = pad + "<" + name + ">";
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
out += "\n" + c->describe(depth + 1);
}
out += "\n" + pad + "</" + name + ">";
return out;
}
std::string ParseNode::toString() const { return describe(0); }
std::string ParseNode::toString(isize depth) const { return describe(depth); }
// ============================================================================
// ParseTree Building
// ============================================================================
ParseNode* ParseTree::resolve(isize index) {
if (index == ParseNode::npos) return nullptr;
return &nodes[index];
}
const ParseNode* ParseTree::resolve(isize index) const {
if (index == ParseNode::npos) return nullptr;
return &nodes[index];
}
isize ParseTree::addNode(const TokenResult& result) {
ParseNode node;
node.owner = this;
node.tag = result.tag;
node.own = result.match;
node.text = result.flatMatch();
isize self = nodes.size();
nodes.push_back(std::move(node));
return self;
}
void ParseTree::linkChildren(isize parent_index, const vector<isize>& children) {
ParseNode& parent = nodes[parent_index];
parent.first_child = ParseNode::npos;
parent.last_child = ParseNode::npos;
parent.children_count = 0;
for (isize child : children) {
ParseNode& kid = nodes[child];
kid.parent = parent_index;
kid.prev_sibling = parent.last_child;
kid.next_sibling = ParseNode::npos;
if (parent.last_child != ParseNode::npos) {
nodes[parent.last_child].next_sibling = child;
} else {
parent.first_child = child;
}
parent.last_child = child;
parent.children_count++;
}
}
vector<isize> ParseTree::collapse(const TokenResult& result) {
vector<isize> kids;
for (const TokenResult& sub : result.child) {
for (isize kid : collapse(sub)) kids.push_back(kid);
}
// Untagged rules are grammar scaffolding, never part of the exposed syntax.
if (!result.tag.has_value()) return kids;
const std::u32string folded = result.flatMatch();
isize self = addNode(result);
// An empty shell has nothing to show.
if (kids.empty() && folded.empty()) return {};
linkChildren(self, kids);
return { self };
}
void ParseTree::build(const TokenResult& result) {
nodes.clear();
root_index = ParseNode::npos;
if (!result.success) return;
vector<isize> tops = collapse(result);
if (tops.size() == 1) {
root_index = tops.front();
return;
}
// A grammar that exposes several top level rules still gets one container.
ParseNode root;
root.owner = this;
root_index = nodes.size();
nodes.push_back(std::move(root));
linkChildren(root_index, tops);
}
bool ParseTree::parsePrefix(const Token* token, TextReader& reader) {
nodes.clear();
root_index = ParseNode::npos;
if (token == nullptr) return false;
TokenResult result = token->test(reader);
if (!result.success) return false;
build(result);
return true;
}
bool ParseTree::parse(const Token* token, TextReader& reader) {
nodes.clear();
root_index = ParseNode::npos;
if (token == nullptr) return false;
TokenResult result = token->test(reader);
if (!result.success) return false;
// A complete parse leaves nothing behind in the reader.
if (reader.hasError()) return false;
if (reader.current().has_value()) return false;
build(result);
return true;
}
bool ParseTree::parse(const Token* token, const std::string& source) {
StringTextReader reader(source);
return parse(token, reader);
}
// ============================================================================
// ParseTree Inspection
// ============================================================================
const ParseNode* ParseTree::find(std::string_view name) const {
const ParseNode* r = root();
return r == nullptr ? nullptr : r->find(name);
}
vector<const ParseNode*> ParseTree::findAll(std::string_view name) const {
const ParseNode* r = root();
if (r == nullptr) return {};
return r->findAll(name);
}
isize ParseTree::count(std::string_view name) const {
const ParseNode* r = root();
return r == nullptr ? 0 : r->count(name);
}
std::string ParseTree::toString() const {
const ParseNode* r = root();
return r == nullptr ? std::string() : r->toString();
}
ParseNode* ParseTree::find(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseTree*>(this)->find(name)); }
vector<ParseNode*> ParseTree::findAll(std::string_view name) {
vector<const ParseNode*> hits = static_cast<const ParseTree*>(this)->findAll(name);
vector<ParseNode*> out;
out.reserve(hits.size());
for (const ParseNode* hit : hits) out.push_back(const_cast<ParseNode*>(hit));
return out;
}
}
+295
View File
@@ -0,0 +1,295 @@
#pragma once
#include <spider/compiler/common.hpp>
#include <spider/compiler/text/unicode.hpp>
#include <spider/compiler/text/Token.hpp>
namespace spider {
class ParseTree;
/**
* @brief DOM style handle over a single node of a parsed token tree.
* @details Every node knows its parent, its children and its siblings, so a
* parsed program can be walked and inspected much like a document.
* Nodes are owned by the ParseTree that produced them and stay
* valid while that tree is alive. The tree itself is built by
* ParseTree, never by hand.
*/
class ParseNode {
friend class ParseTree;
public:
/** @brief Index value used for "no node" links, since isize is unsigned. */
static constexpr isize npos = static_cast<isize>(-1);
private:
ParseTree* owner = nullptr;
isize parent = npos;
isize first_child = npos;
isize last_child = npos;
isize next_sibling = npos;
isize prev_sibling = npos;
isize children_count = 0;
optional<std::string_view> tag = {};
std::u32string own = U"";
std::u32string text = U"";
public:
ParseNode() = default;
public:
/**
* @brief The grammar tag of this node, or nothing if untagged.
*/
const optional<std::string_view>& tagName() const { return tag; }
/**
* @brief Checks if this node carries the given tag.
*/
bool hasTag(std::string_view name) const { return tag.has_value() && *tag == name; }
/**
* @brief The full source text covered by this node and all of its children.
*/
const std::u32string& fullText() const { return text; }
/**
* @brief The source text held by this node alone, without its children.
* @note For tagged nodes that fold their children, this is the full text.
*/
const std::u32string& ownText() const { return own; }
/**
* @brief UTF-8 rendering of text().
*/
std::string textUtf8() const { return unicode::toUTF8(text); }
/**
* @brief UTF-8 rendering of ownText().
*/
std::string ownTextUtf8() const { return unicode::toUTF8(own); }
/**
* @brief Amount of direct children of this node.
*/
isize childCount() const { return children_count; }
/**
* @brief True when this node holds no text and no children.
*/
bool isLeaf() const { return children_count == 0; }
/**
* @brief Distance from the root of the tree, zero for the root itself.
*/
isize depth() const;
/**
* @brief Storage index of this node inside its parent, npos for the root.
*/
isize indexInParent() const { return parent; }
public:
// ---------------------------------------------------------------- //
// Navigation //
// ---------------------------------------------------------------- //
ParseNode* parentNode();
const ParseNode* parentNode() const;
ParseNode* firstChild();
const ParseNode* firstChild() const;
ParseNode* lastChild();
const ParseNode* lastChild() const;
ParseNode* nextSibling();
const ParseNode* nextSibling() const;
ParseNode* previousSibling();
const ParseNode* previousSibling() const;
/**
* @brief The index-th direct child of this node, nullptr when out of range.
*/
ParseNode* childAt(isize index);
const ParseNode* childAt(isize index) const;
public:
// ---------------------------------------------------------------- //
// Navigation filtered by tag //
// ---------------------------------------------------------------- //
/**
* @brief First direct child carrying the given tag.
*/
ParseNode* firstChild(std::string_view name);
const ParseNode* firstChild(std::string_view name) const;
/**
* @brief Last direct child carrying the given tag.
*/
ParseNode* lastChild(std::string_view name);
const ParseNode* lastChild(std::string_view name) const;
/**
* @brief Next sibling of this node carrying the given tag.
*/
ParseNode* nextSibling(std::string_view name);
const ParseNode* nextSibling(std::string_view name) const;
/**
* @brief Nearest ancestor carrying the given tag, nullptr when there is none.
*/
ParseNode* ancestor(std::string_view name);
const ParseNode* ancestor(std::string_view name) const;
public:
// ---------------------------------------------------------------- //
// Queries //
// ---------------------------------------------------------------- //
/**
* @brief First node in document order carrying the given tag, this node included.
*/
ParseNode* find(std::string_view name);
const ParseNode* find(std::string_view name) const;
/**
* @brief Every node in document order carrying the given tag, this node included.
*/
vector<ParseNode*> findAll(std::string_view name);
vector<const ParseNode*> findAll(std::string_view name) const;
/**
* @brief Amount of nodes in document order carrying the given tag.
*/
isize count(std::string_view name) const;
/**
* @brief True when at least one node carries the given tag.
*/
bool contains(std::string_view name) const { return find(name) != nullptr; }
public:
/**
* @brief Human readable XML-like rendering of this node and its children.
*/
std::string toString() const;
/**
* @brief XML-like rendering of this node alone, indented by the given depth.
*/
std::string toString(isize depth) const;
private:
std::string describe(isize depth) const;
};
/**
* @brief Owner of a parsed token tree, exposing a document like interface.
* @details Use parse() to turn source text into an inspectable tree. The tree
* is materialized once and then only read, so any number of consumers
* can walk the same nodes safely.
*/
class ParseTree {
friend class ParseNode;
private:
deque<ParseNode> nodes;
isize root_index = ParseNode::npos;
private:
ParseNode* resolve(isize index);
const ParseNode* resolve(isize index) const;
isize addNode(const TokenResult& result);
void linkChildren(isize parent, const vector<isize>& children);
/**
* @brief Materializes a parse result, dropping the grammar scaffolding.
* @details Only tagged rules become nodes, so a consumer walks syntax and not
* combinators: untagged rules are pure plumbing and simply hoist their
* children upwards, and empty shells are discarded. Returns the top
* level nodes produced by this subtree.
*/
vector<isize> collapse(const TokenResult& result);
public:
ParseTree() = default;
~ParseTree() = default;
// Nodes point back at their owning tree, so trees are never copied or moved.
ParseTree(const ParseTree&) = delete;
ParseTree& operator=(const ParseTree&) = delete;
ParseTree(ParseTree&&) = delete;
ParseTree& operator=(ParseTree&&) = delete;
public:
/**
* @brief Parses source text, requiring the grammar to consume all of it.
* @returns True on success, in which case the tree holds the parsed nodes.
*/
bool parse(const Token* token, const std::string& source);
/**
* @brief Parses from a reader, requiring the grammar to consume all of it.
* @returns True on success, in which case the tree holds the parsed nodes.
*/
bool parse(const Token* token, TextReader& reader);
/**
* @brief Parses the longest matching prefix of a reader.
* @returns True when at least something was matched.
*/
bool parsePrefix(const Token* token, TextReader& reader);
/**
* @brief Adopts an already produced parse result.
*/
void build(const TokenResult& result);
public:
ParseNode* root() { return resolve(root_index); }
const ParseNode* root() const { return resolve(root_index); }
bool empty() const { return root_index == ParseNode::npos; }
public:
ParseNode* find(std::string_view name);
const ParseNode* find(std::string_view name) const;
vector<ParseNode*> findAll(std::string_view name);
vector<const ParseNode*> findAll(std::string_view name) const;
isize count(std::string_view name) const;
/**
* @brief Human readable XML-like rendering of the whole tree.
*/
std::string toString() const;
};
}
+65 -53
View File
@@ -1,6 +1,6 @@
#include "TextReader.hpp"
#include <spider/compiler/text/utf8.hpp>
#include <spider/compiler/text/unicode.hpp>
#include <stdexcept>
@@ -8,11 +8,7 @@ namespace spider {
// Text Reader //
TextReader::TextReader() : err(false), eof(false), bufferIndex(0) {
// Prime the buffer with the first character
// so current() is immediately valid
fillBufferTo(0);
}
TextReader::TextReader() : err(false), eof(false), bufferIndex(0) {}
TextReader::~TextReader() {}
@@ -32,19 +28,30 @@ namespace spider {
// instead of whatever that was, convert to UTF-32
// and then do an easy compare!
std::u32string str;
if(!utf8::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!");
if(!unicode::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!");
return eat(str);
}
bool TextReader::eat(const std::u32string& str) {
// case 0: no str
if(str.empty()) return true;
// prepare n chars
isize index_space = str.size() - 1;
fillBufferTo(index_space);
// fast reject
if(!hasBufferTo(index_space)) return false;
// compare now
isize index;
for(index = 0; index < str.size(); index++) {
if(str[index] != peekChar(index)) return false;
for(isize i = 0; i <= index_space; i++) {
if(str[i] != buffer[bufferIndex + i]) {
return false;
}
}
// success!
nextChar(index);
consumeChars(str.size());
return true;
}
@@ -68,37 +75,30 @@ namespace spider {
return char(ch);
}
/**
* Returns the current character.
*/
optional<u32> TextReader::current() {
fillBufferTo(0);
if (bufferIndex < buffer.size()) {
return buffer[bufferIndex];
}
return {};
}
/**
* Reads the next character and advances the position tracker.
*/
optional<u32> TextReader::nextChar(isize n) {
// Ensure the character we are moving TO exists
if (fillBufferTo(n)) {
// advance n characters
while(n--) {
advance(buffer[bufferIndex]);
bufferIndex++;
}
return current();
}
return {};
fillBufferTo(n); // index = n will be accessible
// from [0, n] inclusive, equal to (n + 1) chars
// remember partial success
consumeChars(n); // n chars will be removed
// return current char
// current char, index = 0
return current();
}
/**
* Keeps the next n-th character (n = 0 is current).
*/
optional<u32> TextReader::peekChar(isize n) {
if (fillBufferTo(n)) return buffer[bufferIndex + n];
fillBufferTo(n);
if (hasBufferTo(n)) return buffer[bufferIndex + n];
return {};
}
@@ -110,12 +110,16 @@ namespace spider {
}
}
isize TextReader::push() {
return bufferIndex;
TextReader::State TextReader::push() {
return { .err = err, .eof = eof, .at = at, .errmsg = errmsg, .index = bufferIndex };
}
void TextReader::pop(isize index) {
bufferIndex = std::min(index, bufferIndex);
void TextReader::pop(TextReader::State s) {
err = s.err;
eof = s.eof;
at = s.at;
errmsg = s.errmsg;
bufferIndex = s.index;
}
/**
@@ -139,7 +143,7 @@ namespace spider {
if (err) return false;
if (eof) return false;
isize chsize = utf8::seqlen(u8(bytes[bindex]));
isize chsize = unicode::seqlen(u8(bytes[bindex]));
if(chsize == 0) {
err = true;
errmsg = "Invalid start of UTF-8 sequence.";
@@ -154,7 +158,7 @@ namespace spider {
}
u32 decodedChar;
if(!utf8::decodeArr(bytes, chsize, decodedChar)) {
if(!unicode::decodeArr(bytes, chsize, decodedChar)) {
err = true;
errmsg = "Invalid UTF-8 sequence.";
return false;
@@ -163,19 +167,27 @@ namespace spider {
return true;
}
/**
* Fills the buffer sequentially until it contains at least up
* to (bufferIndex + targetOffset).
*/
bool TextReader::fillBufferTo(isize targetOffset) {
isize targetSize = bufferIndex + targetOffset + 1;
while (buffer.size() < targetSize) {
isize targetSize = bufferIndex + targetOffset;
while (targetSize >= buffer.size()) {
if(readChar()) continue;
return false;
}
return true;
}
bool TextReader::hasBufferTo(isize index) {
return bufferIndex + index < buffer.size();
}
void TextReader::consumeChars(isize n) {
// advance up to specified char.
for(isize i = 0; i < n && hasBufferTo(i); i++) {
advance(buffer[bufferIndex]);
}
bufferIndex += n;
}
pos TextReader::getPosition() const {
return at;
}
@@ -212,24 +224,24 @@ namespace spider {
// String Reader //
StringTextReader::StringTextReader(std::string initialText)
: buffer(std::move(initialText)),
stringStream(std::make_unique<std::istringstream>(buffer)) {
}
: txt_buffer(std::move(initialText)),
stringStream(std::make_unique<std::istringstream>(txt_buffer)) { }
std::istream& StringTextReader::getStream() {
return *stringStream;
}
void StringTextReader::set(const std::string& newText) {
buffer = newText;
stringStream = std::make_unique<std::istringstream>(buffer);
}
txt_buffer = newText;
stringStream = std::make_unique<std::istringstream>(txt_buffer);
void StringTextReader::append(const std::string& extraText) {
std::streampos pos = stringStream->tellg();
buffer += extraText;
stringStream = std::make_unique<std::istringstream>(buffer);
stringStream->seekg(pos);
buffer.clear();
txt_buffer.clear();
bufferIndex = 0;
err = false;
eof = false;
at = pos();
}
}
+39 -12
View File
@@ -34,11 +34,6 @@ namespace spider {
std::string errmsg;
struct stored_char {
u8 byte_count;
u32 value;
};
/**
* Buffer of extracted characters.
*/
@@ -52,6 +47,16 @@ namespace spider {
*/
isize bufferIndex;
public:
struct State {
bool err;
bool eof;
pos at;
std::string errmsg;
isize index;
};
public:
TextReader();
@@ -94,13 +99,16 @@ namespace spider {
optional<u32> current();
/**
* Reads the next n-th character.
* Skips n number of characters and returns
* the current one in that position.
* n = 0 is a noop, since it's the current one.
*
* Will advance until the EOF is reached.
*/
optional<u32> nextChar(isize n = 1);
/**
* Keeps the next n-th character
* Returns the n-th character following the current one.
* n = 0 is the current one.
*/
optional<u32> peekChar(isize n = 1);
@@ -121,14 +129,14 @@ namespace spider {
* Inside a parser, this allows to roll
* back the index to a specific position.
*/
isize push();
State push();
/**
* Sets the current buffer index.
* Inside a parser, rolls back to
* a previous position.
*/
void pop(isize index);
void pop(State s);
/**
* Returns true if the end of the stream has been reached.
@@ -160,8 +168,29 @@ namespace spider {
virtual std::istream& getStream() = 0;
/**
* Fills the buffer sequentially until it the passed
* index can be safely accessed, relative to the current
* buffer position.
*
* Returns false if that index could not be reached.
* Partial success is possible, check buffer.size()!
*/
bool fillBufferTo(isize index);
/**
* Verifies that the index can be safely accessed,
* relative to the current buffer position.
*/
bool hasBufferTo(isize index);
/**
* Triggers the buffer to consume this number
* of characters from the buffer. This is a reverseable
* operation.
*/
void consumeChars(isize n);
};
/**
@@ -188,7 +217,7 @@ namespace spider {
class StringTextReader : public TextReader {
private:
std::string buffer;
std::string txt_buffer;
std::unique_ptr<std::istringstream> stringStream;
public:
@@ -199,8 +228,6 @@ namespace spider {
void set(const std::string& newText);
void append(const std::string& extraText);
protected:
std::istream& getStream() override;
+157 -71
View File
@@ -1,25 +1,94 @@
#include "Token.hpp"
#include <span>
namespace spider {
// ============================================================================
// Token Implementation
// Token Factory
// ============================================================================
SeqToken Token::operator&(const Token& tok) {
return SeqToken({ tok, *this });
Token* TokenFactory::lit(std::string_view text) {
auto it = lit_cache.find(std::string(text));
if (it != lit_cache.end()) return it->second;
auto p = std::make_unique<LitToken>(text);
auto t = p.get();
arena.emplace_back(std::move(p));
lit_cache.emplace(text, t);
return t;
}
OrToken Token::operator|(const Token& tok) {
return OrToken({ tok, *this });
Token* TokenFactory::operator[](std::string_view text) {
return lit(text);
}
OptToken Token::operator~() {
return OptToken(*this);
Token* TokenFactory::fn(FnTokenFn predicate) {
uptr<Token> p = std::make_unique<FnToken>(predicate);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
RepToken Token::operator*() {
return RepToken(*this);
Token* TokenFactory::seq(const vector<const Token*>& tokens) {
uptr<Token> p = std::make_unique<SeqToken>(tokens);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
Token* TokenFactory::choice(std::string_view opts) {
vector<const Token*> toks;
for (char c : opts) {
std::string s = std::string(1, c);
toks.push_back(lit(s));
}
return choice(toks);
}
Token* TokenFactory::choice(const vector<const Token*>& tokens) {
uptr<Token> p = std::make_unique<OrToken>(tokens);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
Token* TokenFactory::opt(const Token* target) {
uptr<Token> p = std::make_unique<OptToken>(target);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
Token* TokenFactory::rep(const Token* target) {
uptr<Token> p = std::make_unique<RepToken>(target);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
Token* TokenFactory::not_(const Token* target) {
uptr<Token> p = std::make_unique<NotToken>(target);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten) {
uptr<Token> p = std::make_unique<TagToken>(target, tagname, flatten);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
std::u32string TokenResult::flatMatch() const {
if (folded) return match;
std::u32string s = match;
for (const auto& c : child) s += c.flatMatch();
return s;
}
// ============================================================================
@@ -27,60 +96,56 @@ namespace spider {
// ============================================================================
LitToken::LitToken(std::string_view lit) {
if (!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
if (!unicode::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
}
LitToken::LitToken(const char* lit) : LitToken(std::string_view(lit)) {}
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
TokenResult LitToken::test(TextReader& ctx) const {
if(ctx.eat(literal)) return { true, literal };
return { false, {} };
if (ctx.eat(literal)) return { .success = true, .match = literal };
return { .success = false };
}
FnToken::FnToken(FnTokenFn chfn) : fn(chfn) {}
FnToken::FnToken(FnTokenFn chfn) : fn(std::move(chfn)) {}
TokenResult FnToken::test(TextReader& ctx) const {
std::u32string acc;
fn()
return { false, {} };
auto c = ctx.current();
if (c && fn(*c)) {
ctx.nextChar();
return { .success = true, .match = std::u32string(1, char32_t(*c)) };
}
return { .success = false };
}
// ============================================================================
// SeqToken Implementation
// ============================================================================
// ============================================================================30520370
SeqToken::SeqToken(const ilist<ref<const Token>>& list) : tokens(list) {}
SeqToken::SeqToken(const vector<const Token*>& _tokens) : tokens(_tokens) {}
TokenResult SeqToken::test(TextReader& ctx) const {
// this is a common branch point
std::u32string acc;
auto tri = ctx.push();
TokenResult r;
auto i = ctx.push();
// All matching steps within a sequence must pass consecutively.
for (const auto& token_ref : tokens) {
TokenResult res = token_ref.get().test(ctx);
TokenResult res = token_ref->test(ctx);
if (!res.success) {
// Strict ACID Transaction: Roll back context pointer entirely
// if any nested condition in the sequence fails.
ctx.pop(tri);
return { false, {} };
ctx.pop(i);
return { .success = false };
}
// Piecewise accumulation of individual matching sub-tokens
acc += res.match;
//r.match += res.match;
r.child.push_back(res);
}
return { true, acc };
}
SeqToken SeqToken::operator&(const Token& tok) {
// Intrusive chaining optimization: Appends the next token directly into the existing
// registry vector instead of nesting structures, keeping the layout flattened.
tokens.push_back(std::cref(tok));
return *this;
r.success = true;
return r;
}
@@ -88,42 +153,35 @@ namespace spider {
// OrToken Implementation
// ============================================================================
OrToken::OrToken(const ilist<ref<const Token>>& list) : tokens(list) {}
OrToken::OrToken(const vector<const Token*>& _tokens) : tokens(_tokens) {}
TokenResult OrToken::test(TextReader& ctx) const {
// this is a common branch point
auto tri = ctx.push();
// All matching steps within a sequence must pass consecutively.
auto i = ctx.push();
for (const auto& token_ref : tokens) {
// Short-circuit branch: return immediately on first valid choice match
TokenResult res = token_ref.get().test(ctx);
if (res.success) return res;
TokenResult res = token_ref->test(ctx);
if (res.success) {
return res;
}
// Backtrack isolation: Reset the cursor position before testing the next alternative path
ctx.pop(tri);
ctx.pop(i);
}
return { false, {} };
return { .success = false };
}
OrToken OrToken::operator|(const Token& tok) {
// Intrusive grouping layout optimization:
// flattens alternative tokens at code evaluation time.
tokens.push_back(tok);
return *this;
}
// ============================================================================
// OptToken Implementation
// ============================================================================
OptToken::OptToken(const Token& t) : target(t) {}
OptToken::OptToken(const Token* t) : target(t) {}
TokenResult OptToken::test(TextReader& ctx) const {
auto tri = ctx.push();
TokenResult res = target.test(ctx);
TokenResult res = target->test(ctx);
if (res.success) {
return res; // Option matched exactly 1 instance successfully
@@ -132,46 +190,74 @@ namespace spider {
// Recovery path: If sub-rule fails, clean up the dirty state mutation
// and successfully return an empty match payload (0 instances).
ctx.pop(tri);
return { true, U"" };
return { .success = true };
}
OptToken OptToken::operator~() {
// Redundant layer trap protection: returning self
// prevents wrapping an Optional in an Optional
return *this;
}
// ============================================================================
// NotToken Implementation
// ============================================================================
NotToken::NotToken(const Token* t) : target(t) {}
TokenResult NotToken::test(TextReader& ctx) const {
auto i = ctx.push();
TokenResult res = target->test(ctx);
ctx.pop(i);
return { .success = !res.success };
}
// ============================================================================
// RepToken Implementation
// ============================================================================
RepToken::RepToken(const Token& t) : target(t) {}
RepToken::RepToken(const Token* t) : target(t) {}
TokenResult RepToken::test(TextReader& ctx) const {
std::u32string acc;
auto tri = ctx.push();
TokenResult r;
for (;;) {
auto i = ctx.push();
TokenResult res = target->test(ctx);
for(;;) {
TokenResult res = target.test(ctx);
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
if(!res.success || tri == ctx.push()) {
ctx.pop(tri);
if (!res.success || i.index == ctx.push().index) {
ctx.pop(i);
break;
}
acc += res.match;
//r.match += res.match;
r.child.push_back(res);
}
// Repetition rules (* token) always evaluate to successful
// completion state, even with 0 matches.
return { true, acc };
r.success = true;
return r;
}
RepToken RepToken::operator*() {
// Redundant layer trap protection: returning self prevents wrapping
// a Repetition rule inside a Repetition rule
return *this;
// Tagged Token
TagToken::TagToken(const Token* t, std::string_view tag, bool doflatten)
: target(t), tag_name(tag), flatten(doflatten) {
}
TokenResult TagToken::test(TextReader& ctx) const {
auto r = target->test(ctx);
if (!r.success) return r;
// The innermost tag wins, so wrapping a rule that is already tagged, as in
// a choice of tagged rules, never hides what was actually matched.
if (!r.tag.has_value()) r.tag = tag_name;
if (flatten) {
// Fold the subtree text into this node, but keep the children so the
// parsed result stays a walkable tree instead of a flat string.
r.match = r.flatMatch();
r.folded = true;
}
return r;
}
}
+101 -50
View File
@@ -2,7 +2,7 @@
#include <spider/compiler/common.hpp>
#include <spider/compiler/text/utf8.hpp>
#include <spider/compiler/text/unicode.hpp>
#include <spider/compiler/text/TextReader.hpp>
namespace spider {
@@ -13,21 +13,38 @@ namespace spider {
struct TokenResult {
/** @brief Indicates if the token composition successfully matched the input boundary. */
bool success;
bool success = false;
optional<std::string_view> tag = {};
/**
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
* @note Returns empty when success is false.
*/
std::u32string match;
std::u32string match = U"";
vector<TokenResult> child = {};
/**
* @brief Set when match already holds the folded text of the whole subtree.
* @details Tagged rules that flatten their children keep those children around
* so the parsed tree stays walkable, and flag the folded text here.
*/
bool folded = false;
public:
/**
* @brief The full matched text of this node and of all of its children.
*/
std::u32string flatMatch() const;
};
// Forward declarations required by the abstract interface for operator returns.
class SeqToken;
class OrToken;
class OptToken;
class RepToken;
class Token;
class TokenFactory;
using FnTokenFn = std::function<bool(u32)>;
/**
* @brief Pure virtual base class defining the EBNF combinator node contract.
@@ -50,31 +67,50 @@ namespace spider {
*/
virtual TokenResult test(TextReader& ctx) const = 0;
};
class TokenFactory {
private:
// Arena owning all created tokens
std::vector<std::unique_ptr<Token>> arena;
// Deduplication caches
std::unordered_map<std::string, Token*> lit_cache;
public:
/**
* @brief Chains this token and another sequentially.
* @return A temporary structural bridge matching both tokens sequentially.
*/
virtual SeqToken operator&(const Token& tok);
TokenFactory() = default;
/**
* @brief Combines this token and another under alternation.
* @return A structural bridge matching either this token or the fallback selection.
*/
virtual OrToken operator|(const Token& tok);
// Prevent copying to maintain valid internal pointers
TokenFactory(const TokenFactory&) = delete;
/**
* @brief Wraps this node in an optional layout rule.
* @return A structure matching zero or one instances of this current node.
*/
virtual OptToken operator~();
TokenFactory& operator=(const TokenFactory&) = delete;
/**
* @brief Wraps this node in a repetitive loop match framework.
* @return A structure matching zero or more occurrences of this current node.
*/
virtual RepToken operator*();
public:
// --- Primitive Constructors ---
Token* lit(std::string_view text);
Token* operator[](std::string_view text);
Token* fn(FnTokenFn predicate);
// --- Combinator Constructors ---
Token* seq(const vector<const Token*>& tokens);
Token* choice(std::string_view opts);
Token* choice(const vector<const Token*>& tokens);
Token* opt(const Token* target);
Token* rep(const Token* target);
Token* not_(const Token* target);
Token* tag(const Token* target, std::string_view tagname, bool flatten = false);
};
@@ -93,8 +129,6 @@ namespace spider {
*/
LitToken(std::string_view lit);
LitToken(const char* lit);
/** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */
explicit LitToken(std::u32string lit);
@@ -106,8 +140,6 @@ namespace spider {
TokenResult test(TextReader& ctx) const override;
};
using FnTokenFn = std::function<bool(u32) >;
/**
* @brief Function based token
*/
@@ -135,11 +167,10 @@ namespace spider {
* @brief Internal contiguous layout registry storing lightweight, zero-overhead references.
* @details Avoids heap allocation penalties by referencing static instances immutably.
*/
vector<ref<const Token>> tokens;
vector<const Token*> tokens;
public:
/** @brief Constructs a sequence container directly out of an inline brace-enclosed listing. */
SeqToken(const ilist<ref<const Token>>& list);
SeqToken(const vector<const Token*>& tokens);
public:
/**
@@ -149,8 +180,6 @@ namespace spider {
*/
TokenResult test(TextReader& ctx) const override;
/** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */
SeqToken operator&(const Token& tok) override;
};
/**
@@ -159,11 +188,10 @@ namespace spider {
class OrToken : public Token {
private:
/** @brief Ordered registry of possible alternate structural paths. */
vector<ref<const Token>> tokens;
vector<const Token*> tokens;
public:
/** @brief Constructs an alternation choice layout from brace-enclosed tokens. */
OrToken(const ilist<ref<const Token>>& list);
OrToken(const vector<const Token*>& tokens);
public:
/**
@@ -172,8 +200,6 @@ namespace spider {
*/
TokenResult test(TextReader& ctx) const override;
/** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */
OrToken operator|(const Token& tok) override;
};
/**
@@ -182,11 +208,11 @@ namespace spider {
class OptToken : public Token {
private:
/** @brief Read-only target node reference to test optional status against. */
const Token& target;
const Token* target;
public:
/** @brief Binds the target node rule structural dependency layout wrapper. */
explicit OptToken(const Token& t);
explicit OptToken(const Token* t);
public:
/**
@@ -195,8 +221,20 @@ namespace spider {
*/
TokenResult test(TextReader& ctx) const override;
/** @brief Stub override providing standard compliance with the base Token interface signature. */
OptToken operator~() override;
};
/**
* @brief Negative lookahead combinator (NotToken).
*/
class NotToken : public Token {
private:
const Token* target;
public:
explicit NotToken(const Token* t);
TokenResult test(TextReader& ctx) const override;
};
/**
@@ -205,11 +243,11 @@ namespace spider {
class RepToken : public Token {
private:
/** @brief The base token node sequence layer evaluated in loops. */
const Token& target;
const Token* target;
public:
/** @brief Binds the repeated structural blueprint node wrapper configuration. */
explicit RepToken(const Token& t);
explicit RepToken(const Token* t);
public:
/**
@@ -219,8 +257,21 @@ namespace spider {
*/
TokenResult test(TextReader& ctx) const override;
/** @brief Stub override providing standard compliance with the base Token interface signature. */
RepToken operator*() override;
};
class TagToken : public Token {
private:
const Token* target;
std::string tag_name;
bool flatten;
public:
TagToken(const Token* t, std::string_view tag, bool doflatten);
TokenResult test(TextReader& ctx) const override;
};
}
@@ -8,7 +8,7 @@
namespace spider {
namespace utf8 {
namespace unicode {
// --------------------- //
// UTF-8 Sequence Length //
@@ -95,6 +95,55 @@ namespace spider {
return _i == csize;
}
// ----------------- //
// UTF-32 into UTF-8 //
// ----------------- //
inline void append_utf32_to_utf8(u32 code_point, std::string& out) {
if (code_point <= 0x7F) {
// 1-byte sequence (ASCII)
out.push_back(static_cast<char>(code_point));
} else if (code_point <= 0x7FF) {
// 2-byte sequence
out.push_back(static_cast<char>(0xC0 | ((code_point >> 6) & 0x1F)));
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
} else if (code_point <= 0xFFFF) {
// 3-byte sequence
// Filter out surrogate pairs (U+D800 to U+DFFF) as they are invalid Unicode scalar values
if (code_point >= 0xD800 && code_point <= 0xDFFF) {
code_point = 0xFFFD; // Replacement character
}
out.push_back(static_cast<char>(0xE0 | ((code_point >> 12) & 0x0F)));
out.push_back(static_cast<char>(0x80 | ((code_point >> 6) & 0x3F)));
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
} else if (code_point <= 0x10FFFF) {
// 4-byte sequence
out.push_back(static_cast<char>(0xF0 | ((code_point >> 18) & 0x07)));
out.push_back(static_cast<char>(0x80 | ((code_point >> 12) & 0x3F)));
out.push_back(static_cast<char>(0x80 | ((code_point >> 6) & 0x3F)));
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
} else {
// Code point out of Unicode range -> insert UTF-8 replacement character U+FFFD
append_utf32_to_utf8(0xFFFD, out);
}
}
inline std::string toUTF8(u32 cp) {
std::string s;
append_utf32_to_utf8(cp, s);
return s;
}
inline std::string toUTF8(const std::u32string& str) {
std::string s;
for(u32 ch : str) append_utf32_to_utf8(ch, s);
return s;
}
// ----------------- //
// STRINGS //
// ----------------- //
inline const char* getControlCharName(u8 c) {
static const char* names[32] = {
"NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL",
@@ -107,7 +156,7 @@ namespace spider {
return nullptr;
}
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {
inline void hexdump_utf8(const char* data, isize length, pos at, std::ostream& ostr) {
auto old_flags = ostr.flags();
auto old_fill = ostr.fill();