Compare commits
13 Commits
4e0461cb2e
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
| 9254c9e0b0 | |||
| a3fc9d3d7b | |||
| 9388c504c2 | |||
| 5dd0c85d78 | |||
| df2a7c8289 | |||
| 2d9a962cf7 | |||
| 6e13571f60 | |||
| 8764556e71 | |||
| 897e155b0e | |||
| fa511520c6 | |||
| 3779848355 | |||
| a228c0c59c | |||
| a3211dbeac |
@@ -0,0 +1,4 @@
|
|||||||
|
# Line endings are LF everywhere, in the repository and in the working tree.
|
||||||
|
# Without this, core.autocrlf on Windows rewrites every file to CRLF on
|
||||||
|
# checkout, which mixes endings inside the workspace and breaks shell scripts.
|
||||||
|
* text=auto eol=lf
|
||||||
@@ -4,3 +4,7 @@
|
|||||||
# So hold on
|
# So hold on
|
||||||
/bin
|
/bin
|
||||||
/out
|
/out
|
||||||
|
|
||||||
|
# VS Code secret folder is used to hold
|
||||||
|
# user-specific paths and settings
|
||||||
|
/.vscode
|
||||||
|
|||||||
Vendored
-3
@@ -1,3 +0,0 @@
|
|||||||
{
|
|
||||||
"C_Cpp.default.compilerPath": "C:/msys64/ucrt64/bin/g++.exe"
|
|
||||||
}
|
|
||||||
@@ -5,6 +5,9 @@
|
|||||||
#Compiler and Linker
|
#Compiler and Linker
|
||||||
CC := g++
|
CC := g++
|
||||||
|
|
||||||
|
# Ensure POSIX utilities in MSYS2 take precedence over Windows System32
|
||||||
|
export PATH := /usr/bin:/ucrt64/bin:$(PATH)
|
||||||
|
|
||||||
#The Target Binary Program
|
#The Target Binary Program
|
||||||
TARGET := out.exe
|
TARGET := out.exe
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,25 @@
|
|||||||
|
[CmdletBinding()]
|
||||||
|
param(
|
||||||
|
[Parameter(ValueFromRemainingArguments = $true)]
|
||||||
|
[string[]]$Command
|
||||||
|
)
|
||||||
|
|
||||||
|
$msysRoot = if ($env:MSYS2_ROOT) { $env:MSYS2_ROOT } else { "C:\msys64" }
|
||||||
|
$bash = Join-Path $msysRoot "usr\bin\bash.exe"
|
||||||
|
|
||||||
|
if (-not (Test-Path -LiteralPath $bash)) {
|
||||||
|
Write-Error "MSYS2 bash was not found at: $bash"
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
|
||||||
|
$env:MSYSTEM = "UCRT64"
|
||||||
|
$env:CHERE_INVOKING = "1"
|
||||||
|
|
||||||
|
if ($Command.Count -eq 0) {
|
||||||
|
& $bash --login
|
||||||
|
} else {
|
||||||
|
$cmdString = $Command -join " "
|
||||||
|
& $bash -lc $cmdString
|
||||||
|
}
|
||||||
|
|
||||||
|
exit $LASTEXITCODE
|
||||||
-302
@@ -1,302 +0,0 @@
|
|||||||
{
|
|
||||||
"cells": [
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 95,
|
|
||||||
"id": "00e26c5b",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [],
|
|
||||||
"source": [
|
|
||||||
"from lark import Lark, Transformer\n",
|
|
||||||
"import os"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 96,
|
|
||||||
"id": "cc16be1a",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [],
|
|
||||||
"source": [
|
|
||||||
"ebnf_targets = {\n",
|
|
||||||
" \"assembly\": {\n",
|
|
||||||
" \"src\": \"./samples/assembly.ebnf\",\n",
|
|
||||||
" \"dst\": \"./spider/compiler/assembly/AssemblyParser.hpp\",\n",
|
|
||||||
" \"cnt\": None,\n",
|
|
||||||
" },\n",
|
|
||||||
"}\n"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 97,
|
|
||||||
"id": "e88d212f",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [
|
|
||||||
{
|
|
||||||
"name": "stdout",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"\n",
|
|
||||||
"--- Loading EBNF Targets ---\n",
|
|
||||||
"✅ Success [assembly]: Loaded './samples/assembly.ebnf' -> Target destination: './spider/compiler/assembly/AssemblyParser.hpp'\n"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"source": [
|
|
||||||
"print(\"\\n--- Loading EBNF Targets ---\")\n",
|
|
||||||
"for target_name, paths in ebnf_targets.items():\n",
|
|
||||||
" src_path = paths[\"src\"]\n",
|
|
||||||
" dst_path = paths[\"dst\"]\n",
|
|
||||||
" \n",
|
|
||||||
" try:\n",
|
|
||||||
" with open(src_path, \"r\", encoding=\"utf-8\") as file:\n",
|
|
||||||
" paths[\"cnt\"] = file.read()\n",
|
|
||||||
" print(f\"✅ Success [{target_name}]: Loaded '{src_path}' -> Target destination: '{dst_path}'\")\n",
|
|
||||||
" \n",
|
|
||||||
" except FileNotFoundError:\n",
|
|
||||||
" print(f\"❌ Error [{target_name}]: Source file not found at '{src_path}'\")\n"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 98,
|
|
||||||
"id": "e8095002",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [],
|
|
||||||
"source": [
|
|
||||||
"import os\n",
|
|
||||||
"from lark import Lark, Transformer\n",
|
|
||||||
"\n",
|
|
||||||
"iso_ebnf_meta_grammar = r\"\"\"\n",
|
|
||||||
" start: rule+\n",
|
|
||||||
" rule: RULE_NAME \"=\" expression \";\"\n",
|
|
||||||
" \n",
|
|
||||||
" ?expression: alternation\n",
|
|
||||||
" alternation: sequence (\"|\" sequence)*\n",
|
|
||||||
" \n",
|
|
||||||
" sequence: item ( [\",\"] item )*\n",
|
|
||||||
" \n",
|
|
||||||
" ?item: atom\n",
|
|
||||||
" | atom \"?\" -> optional\n",
|
|
||||||
" | atom \"*\" -> repeat\n",
|
|
||||||
" | \"[\" expression \"]\" -> optional\n",
|
|
||||||
" | \"{\" expression \"}\" -> repeat\n",
|
|
||||||
" \n",
|
|
||||||
" ?atom: RULE_NAME -> call_rule\n",
|
|
||||||
" | TERMINAL -> match_terminal\n",
|
|
||||||
" | SPECIAL_SEQ -> handle_special\n",
|
|
||||||
" | \"(\" expression \")\" -> group\n",
|
|
||||||
"\n",
|
|
||||||
" RULE_NAME: /[a-zA-Z_][a-zA-Z0-9_]*/\n",
|
|
||||||
" TERMINAL: /\"[^\"\\\\]*(?:\\\\.[^\"\\\\]*)*\"/ | /'[^'\\\\]*(?:\\\\.[^'\\\\]*)*'/\n",
|
|
||||||
" SPECIAL_SEQ: /\\?[\\s\\S]*?\\?/\n",
|
|
||||||
" COMMENT: /\\(\\*([\\s\\S]*?)\\*\\)/\n",
|
|
||||||
"\n",
|
|
||||||
" %import common.WS\n",
|
|
||||||
" %ignore WS\n",
|
|
||||||
" %ignore COMMENT\n",
|
|
||||||
"\"\"\"\n",
|
|
||||||
"\n",
|
|
||||||
"class AssemblyCppGenerator(Transformer):\n",
|
|
||||||
" def start(self, rules):\n",
|
|
||||||
" cpp_functions = \"\\n\\n\".join(rules)\n",
|
|
||||||
" return f\"\"\"#pragma once\n",
|
|
||||||
"\n",
|
|
||||||
"#include <iostream>\n",
|
|
||||||
"#include <string>\n",
|
|
||||||
"#include <vector>\n",
|
|
||||||
"#include <stdexcept>\n",
|
|
||||||
"\n",
|
|
||||||
"class AssemblyParser {{\n",
|
|
||||||
"private:\n",
|
|
||||||
" std::string src;\n",
|
|
||||||
" size_t pos = 0;\n",
|
|
||||||
"\n",
|
|
||||||
" std::string peek_str(size_t len) {{\n",
|
|
||||||
" if (pos + len <= src.length()) return src.substr(pos, len);\n",
|
|
||||||
" return src.substr(pos);\n",
|
|
||||||
" }}\n",
|
|
||||||
"\n",
|
|
||||||
" char peek() {{ return pos < src.length() ? src[pos] : '\\\\0'; }}\n",
|
|
||||||
" \n",
|
|
||||||
" void match_char(char expected) {{\n",
|
|
||||||
" if (peek() == expected) pos++;\n",
|
|
||||||
" else throw std::runtime_error(\"Unexpected token matching character\");\n",
|
|
||||||
" }}\n",
|
|
||||||
"\n",
|
|
||||||
" void match_string(std::string expected) {{\n",
|
|
||||||
" if (peek_str(expected.length()) == expected) pos += expected.length();\n",
|
|
||||||
" else throw std::runtime_error(\"Unexpected token matching string: \" + expected);\n",
|
|
||||||
" }}\n",
|
|
||||||
"\n",
|
|
||||||
" bool isUTF8Alpha() {{ return isalpha(peek()); }}\n",
|
|
||||||
" bool isWhithespaceCharNotCrLf() {{ return peek() == ' ' || peek() == '\\\\t'; }}\n",
|
|
||||||
" bool isUTF8CharNotCrLf() {{ return peek() != '\\\\r' && peek() != '\\\\n' && peek() != '\\\\0'; }}\n",
|
|
||||||
" bool isUTF8CharLitCont() {{ return peek() != '\\'' && peek() != '\\\\\\\\'; }}\n",
|
|
||||||
" bool isUTF8StringLitCont() {{ return peek() != '\"' && peek() != '\\\\\\\\'; }}\n",
|
|
||||||
"\n",
|
|
||||||
"public:\n",
|
|
||||||
" AssemblyParser(std::string input) : src(input) {{}}\n",
|
|
||||||
"\n",
|
|
||||||
" void parse() {{\n",
|
|
||||||
" parse_program(); \n",
|
|
||||||
" if (pos < src.length()) throw std::runtime_error(\"Trailing characters left unparsed.\");\n",
|
|
||||||
" std::cout << \"Assembly source compiled cleanly!\" << std::endl;\n",
|
|
||||||
" }}\n",
|
|
||||||
"\n",
|
|
||||||
"{cpp_functions}\n",
|
|
||||||
"}};\n",
|
|
||||||
"\"\"\"\n",
|
|
||||||
"\n",
|
|
||||||
" def rule(self, args):\n",
|
|
||||||
" name, expr = args\n",
|
|
||||||
" return f\" void parse_{name}() {{\\n{expr}\\n }}\"\n",
|
|
||||||
"\n",
|
|
||||||
" # FIX 1: Explicitly handle choice logic using C++ style paths\n",
|
|
||||||
" def alternation(self, items):\n",
|
|
||||||
" code_lines = []\n",
|
|
||||||
" for i, item in enumerate(items):\n",
|
|
||||||
" # Clean up padding whitespace if any\n",
|
|
||||||
" clean_item = str(item).strip()\n",
|
|
||||||
" if not clean_item: continue\n",
|
|
||||||
" \n",
|
|
||||||
" # Since lookahead processing requires FIRST sets, we scaffold a sequential fallback\n",
|
|
||||||
" if i == 0:\n",
|
|
||||||
" code_lines.append(f\" if (/* option {i+1} */ true) {{\\n {clean_item}\\n }}\")\n",
|
|
||||||
" else:\n",
|
|
||||||
" code_lines.append(f\" else if (/* option {i+1} */ true) {{\\n {clean_item}\\n }}\")\n",
|
|
||||||
" return \"\\n\".join(code_lines)\n",
|
|
||||||
"\n",
|
|
||||||
" def sequence(self, items):\n",
|
|
||||||
" flattened_items = []\n",
|
|
||||||
" for item in items:\n",
|
|
||||||
" if isinstance(item, list):\n",
|
|
||||||
" for sub_item in item:\n",
|
|
||||||
" if sub_item: flattened_items.append(str(sub_item).strip())\n",
|
|
||||||
" elif item:\n",
|
|
||||||
" flattened_items.append(str(item).strip())\n",
|
|
||||||
" return \"\\n\".join(f\" {item}\" for item in flattened_items if item)\n",
|
|
||||||
"\n",
|
|
||||||
" def call_rule(self, token):\n",
|
|
||||||
" rule_name = token[0].value if isinstance(token, list) else token.value\n",
|
|
||||||
" return f\"parse_{rule_name}();\"\n",
|
|
||||||
"\n",
|
|
||||||
" # FIX 2: Generate match_string instead of match_char for multi-char string keywords like \"include\"\n",
|
|
||||||
" def match_terminal(self, token):\n",
|
|
||||||
" raw_token_str = token[0].value if isinstance(token, list) else token.value\n",
|
|
||||||
" raw_val = raw_token_str[1:-1]\n",
|
|
||||||
" \n",
|
|
||||||
" if raw_val == r\"\\r\": return \"match_char('\\\\r');\"\n",
|
|
||||||
" if raw_val == r\"\\n\": return \"match_char('\\\\n');\"\n",
|
|
||||||
" if raw_val == r\"\\t\": return \"match_char('\\\\t');\"\n",
|
|
||||||
" if raw_val == r\"\\\\\": return \"match_char('\\\\\\\\');\"\n",
|
|
||||||
" if not raw_val: return \"// Empty string match\"\n",
|
|
||||||
" \n",
|
|
||||||
" if len(raw_val) > 1:\n",
|
|
||||||
" return f\"match_string(\\\"{raw_val}\\\");\"\n",
|
|
||||||
" return f\"match_char('{raw_val}');\"\n",
|
|
||||||
"\n",
|
|
||||||
" def handle_special(self, token):\n",
|
|
||||||
" raw_string = token[0].value if isinstance(token, list) else token.value\n",
|
|
||||||
" func_name = raw_string.strip('?').strip()\n",
|
|
||||||
" return f\"if ({func_name}()) {{ pos++; }} else {{ throw std::runtime_error(\\\"Failed validation for {func_name}\\\"); }}\"\n",
|
|
||||||
"\n",
|
|
||||||
" def optional(self, args):\n",
|
|
||||||
" content = args[0] if not isinstance(args[0], list) else \"\\n \".join(args[0])\n",
|
|
||||||
" return f\"// Optional block\\n if (/* lookahead check */ true) {{\\n {content}\\n }}\"\n",
|
|
||||||
"\n",
|
|
||||||
" def repeat(self, args):\n",
|
|
||||||
" content = args[0] if not isinstance(args[0], list) else \"\\n \".join(args[0])\n",
|
|
||||||
" return f\"// Repeat block\\n while (/* lookahead check */ true) {{\\n {content}\\n }}\"\n",
|
|
||||||
"\n",
|
|
||||||
" def group(self, args):\n",
|
|
||||||
" # Flatten grouped elements cleanly to strings\n",
|
|
||||||
" if isinstance(args, list):\n",
|
|
||||||
" return \"\\n\".join(str(x) for x in args)\n",
|
|
||||||
" return str(args)"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": 99,
|
|
||||||
"id": "558915ff",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [
|
|
||||||
{
|
|
||||||
"name": "stdout",
|
|
||||||
"output_type": "stream",
|
|
||||||
"text": [
|
|
||||||
"--- Starting C++ Compilation Loop ---\n",
|
|
||||||
"Parsing and converting target rule sets for: assembly\n",
|
|
||||||
"🎉 Code generation complete! Output stored in './spider/compiler/assembly/AssemblyParser.hpp'\n"
|
|
||||||
]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"source": [
|
|
||||||
"print(\"--- Starting C++ Compilation Loop ---\")\n",
|
|
||||||
"\n",
|
|
||||||
"try:\n",
|
|
||||||
" meta_parser = Lark(iso_ebnf_meta_grammar, parser='lalr')\n",
|
|
||||||
" \n",
|
|
||||||
" for name, target in ebnf_targets.items():\n",
|
|
||||||
" print(f\"Parsing and converting target rule sets for: {name}\")\n",
|
|
||||||
" \n",
|
|
||||||
" # Build the compiler AST tree from your exact text\n",
|
|
||||||
" syntax_tree = meta_parser.parse(target[\"cnt\"])\n",
|
|
||||||
" \n",
|
|
||||||
" # Transform the AST structural nodes into pure C++ Source strings\n",
|
|
||||||
" compiler_transformer = AssemblyCppGenerator()\n",
|
|
||||||
" compiled_cpp_header = compiler_transformer.transform(syntax_tree)\n",
|
|
||||||
" \n",
|
|
||||||
" # Output directly to your destination path\n",
|
|
||||||
" os.makedirs(os.path.dirname(target[\"dst\"]), exist_ok=True)\n",
|
|
||||||
" with open(target[\"dst\"], \"w\", encoding=\"utf-8\") as f:\n",
|
|
||||||
" f.write(compiled_cpp_header)\n",
|
|
||||||
" \n",
|
|
||||||
" print(f\"🎉 Code generation complete! Output stored in '{target['dst']}'\")\n",
|
|
||||||
"\n",
|
|
||||||
"except Exception as e:\n",
|
|
||||||
" print(f\"❌ Failed to process custom architecture. Error details: \\n{e}\")\n",
|
|
||||||
"\n"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": null,
|
|
||||||
"id": "366688c3",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [],
|
|
||||||
"source": []
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"cell_type": "code",
|
|
||||||
"execution_count": null,
|
|
||||||
"id": "cd1aca3f",
|
|
||||||
"metadata": {},
|
|
||||||
"outputs": [],
|
|
||||||
"source": []
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"metadata": {
|
|
||||||
"kernelspec": {
|
|
||||||
"display_name": "Python 3",
|
|
||||||
"language": "python",
|
|
||||||
"name": "python3"
|
|
||||||
},
|
|
||||||
"language_info": {
|
|
||||||
"codemirror_mode": {
|
|
||||||
"name": "ipython",
|
|
||||||
"version": 3
|
|
||||||
},
|
|
||||||
"file_extension": ".py",
|
|
||||||
"mimetype": "text/x-python",
|
|
||||||
"name": "python",
|
|
||||||
"nbconvert_exporter": "python",
|
|
||||||
"pygments_lexer": "ipython3",
|
|
||||||
"version": "3.13.7"
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"nbformat": 4,
|
|
||||||
"nbformat_minor": 5
|
|
||||||
}
|
|
||||||
@@ -5,6 +5,7 @@
|
|||||||
}
|
}
|
||||||
],
|
],
|
||||||
"settings": {
|
"settings": {
|
||||||
|
"files.eol": "\n",
|
||||||
"gitlens.remotes": [
|
"gitlens.remotes": [
|
||||||
{
|
{
|
||||||
"domain": "git.sintekanalytics.com",
|
"domain": "git.sintekanalytics.com",
|
||||||
|
|||||||
@@ -0,0 +1,7 @@
|
|||||||
|
#pragma once
|
||||||
|
|
||||||
|
namespace spider {
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
}
|
||||||
@@ -1,80 +1,296 @@
|
|||||||
#include <iostream>
|
#include <iostream>
|
||||||
|
|
||||||
|
#include <spider/compiler/Compiler.hpp>
|
||||||
#include <spider/compiler/common.hpp>
|
#include <spider/compiler/common.hpp>
|
||||||
#include <spider/compiler/text/utf8.hpp>
|
|
||||||
|
#include <spider/compiler/text/unicode.hpp>
|
||||||
|
#include <spider/compiler/text/TextReader.hpp>
|
||||||
|
|
||||||
|
#include <spider/compiler/assembler/AsmEBNF.hpp>
|
||||||
|
#include <spider/compiler/text/ParseTree.hpp>
|
||||||
|
|
||||||
|
using namespace spider;
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// Compiler Entry Points
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
|
|
||||||
|
TokenFactory& assemblyGrammar() {
|
||||||
|
static TokenFactory grammar;
|
||||||
|
static bool loaded = false;
|
||||||
|
if (!loaded) {
|
||||||
|
asm_ebnf::initTokens(grammar);
|
||||||
|
loaded = true;
|
||||||
|
}
|
||||||
|
return grammar;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool compileProgram(const std::string& source, ParseTree& out) {
|
||||||
|
return out.parse(asm_ebnf::program, source);
|
||||||
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Test runner helper
|
// ============================================================================
|
||||||
void run_test(const std::string& name, const std::string& input) {
|
// Test Bookkeeping
|
||||||
std::cout << "========================================\n";
|
// ============================================================================
|
||||||
std::cout << " TEST: " << name << "\n";
|
|
||||||
std::cout << "========================================\n";
|
|
||||||
|
|
||||||
spider::pos tracking_pos;
|
// Suite wide bookkeeping, so a failing case can never hide behind a clean exit code.
|
||||||
spider::utf8::hexdump(input.data(), input.size(), tracking_pos, std::cout);
|
static isize total_tests = 0;
|
||||||
std::cout << "\n";
|
static isize failed_tests = 0;
|
||||||
|
|
||||||
|
static void report(bool ok, const std::string& name) {
|
||||||
|
++total_tests;
|
||||||
|
if (ok) {
|
||||||
|
std::cout << "[PASS] " << name << "\n";
|
||||||
|
} else {
|
||||||
|
++failed_tests;
|
||||||
|
std::cerr << "[FAIL] " << name << "\n";
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void utf8sequences() {
|
// Inline evaluator that executes the reader and prints standard output
|
||||||
// Permutation 1: Standard, valid mixed UTF-8 sequences (1, 2, 3, and 4 bytes)
|
static bool run_test_case(const Token* token, const std::string& input, bool expectedSuccess, const std::u32string& expectedMatch = U"") {
|
||||||
// - 'A' -> 1 byte (U+0041)
|
StringTextReader reader(input);
|
||||||
// - '¢' (cents) -> 2 bytes (U+00A2)
|
TokenResult res = token->test(reader);
|
||||||
// - '€' (euro) -> 3 bytes (U+20AC)
|
|
||||||
// - '𐍈' (gothic) -> 4 bytes (U+10348)
|
|
||||||
run_test("Valid Mixed Length Sequences", "A\xC2\xA2\xE2\x82\xAC\xF0\x90\x8D\x88");
|
|
||||||
|
|
||||||
// Permutation 2: Embedded Control Characters
|
bool status_ok = (res.success == expectedSuccess);
|
||||||
// Should display mnemonics like (HT), (LF), (CR) without breaking formatting
|
bool match_ok = (!expectedSuccess) || expectedMatch.empty() || (res.flatMatch() == expectedMatch);
|
||||||
run_test("ASCII Control Characters", "Text\tWith\r\nNewlines");
|
|
||||||
|
|
||||||
// Permutation 3: Invalid Lead Byte
|
++total_tests;
|
||||||
// The byte 0xFF is structurally illegal under any UTF-8 definition.
|
if (status_ok && match_ok) {
|
||||||
// Expected behavior: Display single byte as INVALID LEAD, shift 1 byte over.
|
std::cout << "[PASS] Input: \"" << input << "\" -> "
|
||||||
run_test("Invalid Lead Byte (0xFF)", "ABC\xFFXYZ");
|
<< (res.success ? "SUCCESS" : "FAILURE")
|
||||||
|
<< (res.success ? (" (Matched: \"" + unicode::toUTF8(res.flatMatch()) + "\")") : "")
|
||||||
|
<< "\n";
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
// Permutation 4: Invalid Continuation Sequence
|
std::cerr << "[FAIL] Input: \"" << input << "\"\n"
|
||||||
// A 3-byte header (\xE2) where the second byte (\x00) is a bad continuation.
|
<< " Expected Success: " << (expectedSuccess ? "true" : "false") << ", Got: " << (res.success ? "true" : "false") << "\n";
|
||||||
// Expected behavior: Show the entire sequence up to 'm' bytes, flag as INVALID SEQUENCE.
|
if (expectedSuccess && !expectedMatch.empty()) {
|
||||||
run_test("Invalid Continuation Structure", std::string("Before \xE2\x00\xAC After", 16));
|
std::cerr << " Expected Match: \"" << unicode::toUTF8(expectedMatch) << "\", Got: \"" << unicode::toUTF8(res.flatMatch()) << "\"\n";
|
||||||
|
}
|
||||||
// Permutation 5: Truncated Sequence at End-of-Buffer
|
++failed_tests;
|
||||||
// A 4-byte emoji header (\xF0\x9F) but the string completely cuts off.
|
return false;
|
||||||
// Expected behavior: Display remaining space placeholders as '??' -> TRUNCATED SEQUENCE.
|
|
||||||
run_test("Truncated Sequence (Missing trailing bytes)", "Hello \xF0\x9F");
|
|
||||||
|
|
||||||
// Permutation 6: Overlong Encoding Security Vulnerability
|
|
||||||
// Attempting to write ASCII 'I' (normally 0x49) using 2 bytes: \xC1\x89
|
|
||||||
// Expected behavior: Caught by constraints checks, flagged as INVALID SEQUENCE.
|
|
||||||
run_test("Security Hack: Overlong Encoding", "Safe\xC1\x89Hack");
|
|
||||||
|
|
||||||
// Permutation 7: Out-of-bounds / Restricted Ranges
|
|
||||||
// - \xED\xA0\x80 is a UTF-16 Surrogate (U+D800)
|
|
||||||
// - \xF4\x90\x80\x80 is outside valid Unicode space (> U+10FFFF)
|
|
||||||
// Expected behavior: Flagged securely as INVALID SEQUENCE.
|
|
||||||
run_test("Security Hack: Restricted Ranges (Surrogates & Out-of-bounds)", "Surrogate: \xED\xA0\x80 MaxBounds: \xF4\x90\x80\x80");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// Core Combinator & Grammar Tests
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
void test_primitives_and_literals(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Primitives & Literals ---\n";
|
||||||
|
Token* hello = tf.lit("hello");
|
||||||
|
|
||||||
|
run_test_case(hello, "hello world", true, U"hello");
|
||||||
|
run_test_case(hello, "hell", false);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::letter, "a", true);
|
||||||
|
run_test_case(asm_ebnf::letter, "f", true);
|
||||||
|
run_test_case(asm_ebnf::letter, "A", true);
|
||||||
|
run_test_case(asm_ebnf::letter, "Y", true);
|
||||||
|
run_test_case(asm_ebnf::letter, "9", false);
|
||||||
|
run_test_case(asm_ebnf::digit, "0", true);
|
||||||
|
run_test_case(asm_ebnf::digit, "7", true);
|
||||||
|
run_test_case(asm_ebnf::digit, "x", false);
|
||||||
|
run_test_case(asm_ebnf::hex_digit, "F", true);
|
||||||
|
run_test_case(asm_ebnf::hex_digit, "g", false);
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_choice_and_seq(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Choice & Sequence ---\n";
|
||||||
|
Token* seq_test = tf.seq({ tf["foo"], tf["bar"] });
|
||||||
|
run_test_case(seq_test, "foobar", true, U"foobar");
|
||||||
|
run_test_case(seq_test, "foobaz", false);
|
||||||
|
|
||||||
|
Token* choice_test = tf.choice({ tf["apple"], tf["banana"] });
|
||||||
|
run_test_case(choice_test, "banana", true, U"banana");
|
||||||
|
run_test_case(choice_test, "cherry", false);
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_opt_and_rep(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Optional & Repeat ---\n";
|
||||||
|
Token* opt_test = tf.seq({ tf.opt(tf["+"]), asm_ebnf::digit });
|
||||||
|
run_test_case(opt_test, "+5", true, U"+5");
|
||||||
|
run_test_case(opt_test, "5", true, U"5");
|
||||||
|
|
||||||
|
Token* rep_digits = tf.rep(asm_ebnf::digit);
|
||||||
|
run_test_case(rep_digits, "12345abc", true, U"12345");
|
||||||
|
run_test_case(rep_digits, "abc", true, U"");
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_literals(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Grammatical Literals ---\n";
|
||||||
|
run_test_case(asm_ebnf::identifier, "valid_var1", true, U"valid_var1");
|
||||||
|
run_test_case(asm_ebnf::identifier, "_private", true, U"_private");
|
||||||
|
run_test_case(asm_ebnf::identifier, "123invalid", false);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::decimal_lit, "1234", true);
|
||||||
|
run_test_case(asm_ebnf::decimal_lit, "-567L", true);
|
||||||
|
run_test_case(asm_ebnf::hex_lit, "0x1A2B", true);
|
||||||
|
run_test_case(asm_ebnf::octal_lit, "0c755", true);
|
||||||
|
run_test_case(asm_ebnf::binary_lit, "0b10101", true);
|
||||||
|
run_test_case(asm_ebnf::float_lit, "3.14159F", true);
|
||||||
|
run_test_case(asm_ebnf::float_lit, "1e-10D", true);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::char_lit, "'a'", true);
|
||||||
|
run_test_case(asm_ebnf::char_lit, "'\\n'", true);
|
||||||
|
run_test_case(asm_ebnf::string_lit, "\"Hello World\"", true);
|
||||||
|
run_test_case(asm_ebnf::string_lit, "\"Escape \\\" Test\"", true);
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_addressing_modes(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Addressing Modes ---\n";
|
||||||
|
run_test_case(asm_ebnf::register_tok, "R0", true);
|
||||||
|
run_test_case(asm_ebnf::register_tok, "R15", false); // R15 does not exist!
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::addrm_ind, "[ 0x1000 ]", true);
|
||||||
|
run_test_case(asm_ebnf::addrm_ptr, "[R1]", true);
|
||||||
|
run_test_case(asm_ebnf::addrm_idx, "[R1 + 4]", true);
|
||||||
|
run_test_case(asm_ebnf::addrm_sca, "[R1 + R2 * 4]", true);
|
||||||
|
run_test_case(asm_ebnf::addrm_dis, "[ R1 + R2 * 4 + 16 ]", true);
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_instructions_and_lines(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Instructions & Lines ---\n";
|
||||||
|
run_test_case(asm_ebnf::instruction, "NOP", true);
|
||||||
|
run_test_case(asm_ebnf::instruction, "MOV R1, [R2 + 4]", true);
|
||||||
|
run_test_case(asm_ebnf::instruction, "ADD R0, 100", true);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::preprocessor, "#define MAX_BUF", true);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::annotation, "@inline", true);
|
||||||
|
run_test_case(asm_ebnf::annotation, "@align(4)", true);
|
||||||
|
run_test_case(asm_ebnf::annotation, "@section(name=\"text\", flags=1)", true);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::line, "main: MOV R0, R1 ; copy reg\n", true);
|
||||||
|
run_test_case(asm_ebnf::line, " @deprecated NOP\n", true);
|
||||||
|
run_test_case(asm_ebnf::line, "; only a comment line\n", true);
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_full_program(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Full Program Parser ---\n";
|
||||||
|
std::string asm_code =
|
||||||
|
"#include stdio\n"
|
||||||
|
"\n"
|
||||||
|
"start:\n"
|
||||||
|
" MOV R1, 0x20 ; Load constant\n"
|
||||||
|
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
|
||||||
|
" JMP start\n";
|
||||||
|
|
||||||
|
ParseTree tree;
|
||||||
|
report(compileProgram(asm_code, tree), "full assembly program parses completely");
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_parse_tree(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Parse Tree Inspection ---\n";
|
||||||
|
const std::string asm_code =
|
||||||
|
"#include stdio\n"
|
||||||
|
"\n"
|
||||||
|
"start:\n"
|
||||||
|
" MOV R1, 0x20 ; Load constant\n"
|
||||||
|
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
|
||||||
|
" JMP start\n";
|
||||||
|
|
||||||
|
ParseTree tree;
|
||||||
|
report(compileProgram(asm_code, tree), "program becomes a tree of nodes");
|
||||||
|
|
||||||
|
ParseNode* root = tree.root();
|
||||||
|
report(root != nullptr, "tree exposes a root node");
|
||||||
|
if (root == nullptr) return;
|
||||||
|
|
||||||
|
report(root->hasTag("program"), "root node is tagged as program");
|
||||||
|
report(root->textUtf8() == asm_code, "root text covers the whole source");
|
||||||
|
report(root->childCount() > 0, "root holds child nodes");
|
||||||
|
|
||||||
|
vector<ParseNode*> lines = root->findAll("line");
|
||||||
|
report(lines.size() == 6, "program holds one node per line");
|
||||||
|
|
||||||
|
if (lines.size() == 6) {
|
||||||
|
report(lines.front()->parentNode() == root, "a line knows its parent");
|
||||||
|
report(lines.front()->depth() == 1, "lines sit one level under the root");
|
||||||
|
report(lines.front()->nextSibling() == lines[1], "sibling walk reaches the next line");
|
||||||
|
report(lines[1]->previousSibling() == lines.front(), "sibling walk goes back");
|
||||||
|
report(lines.back()->nextSibling() == nullptr, "the last line has no next sibling");
|
||||||
|
|
||||||
|
ParseNode* preprocessor = lines.front()->firstChild("preprocessor");
|
||||||
|
report(preprocessor != nullptr, "first line holds a preprocessor child");
|
||||||
|
report(preprocessor != nullptr && preprocessor->textUtf8() == "#include stdio", "preprocessor text is readable");
|
||||||
|
|
||||||
|
report(lines[2]->firstChild("line_label") != nullptr, "label line is recognized");
|
||||||
|
ParseNode* label = lines[2]->find("label");
|
||||||
|
report(label != nullptr && label->textUtf8() == "start:", "label text is readable");
|
||||||
|
|
||||||
|
ParseNode* mov = lines[3]->find("instruction");
|
||||||
|
report(mov != nullptr, "instruction is found inside its line");
|
||||||
|
if (mov != nullptr) {
|
||||||
|
report(mov->firstChild("opcode") != nullptr, "instruction exposes its opcode child");
|
||||||
|
|
||||||
|
ParseNode* operands = mov->firstChild("operand_list");
|
||||||
|
report(operands != nullptr && operands->childCount() == 2, "operand list exposes both operands");
|
||||||
|
|
||||||
|
ParseNode* hexlit = mov->find("hex_lit");
|
||||||
|
report(hexlit != nullptr && hexlit->textUtf8() == "0x20", "hex literal text is readable");
|
||||||
|
}
|
||||||
|
|
||||||
|
ParseNode* comment = lines[3]->firstChild("comment");
|
||||||
|
report(comment != nullptr && comment->textUtf8() == "; Load constant", "trailing comment is captured");
|
||||||
|
|
||||||
|
report(lines[4]->firstChild("line_annotation") != nullptr, "annotation line is recognized");
|
||||||
|
ParseNode* displacement = lines[4]->find("addrm_dis");
|
||||||
|
report(displacement != nullptr && displacement->textUtf8() == "[R2 + R3 * 8 + 4]", "displacement operand is readable");
|
||||||
|
report(tree.count("register") == 4, "every register of the program is reachable");
|
||||||
|
}
|
||||||
|
|
||||||
|
ParseTree single;
|
||||||
|
report(single.parse(asm_ebnf::instruction, "MOV R1, [R2 + 4]"), "a single instruction parses on its own");
|
||||||
|
|
||||||
|
ParseNode* instruction = single.root();
|
||||||
|
report(instruction != nullptr && instruction->hasTag("instruction"), "instruction root is tagged");
|
||||||
|
if (instruction != nullptr) {
|
||||||
|
ParseNode* operands = instruction->firstChild("operand_list");
|
||||||
|
report(operands != nullptr && operands->childCount() == 2, "operand list holds two operands");
|
||||||
|
|
||||||
|
ParseNode* second = operands == nullptr ? nullptr : operands->childAt(1);
|
||||||
|
report(second != nullptr && second->textUtf8() == "[R2 + 4]", "operand text is readable");
|
||||||
|
report(second != nullptr && second->ancestor("instruction") == instruction, "an operand can climb back to its instruction");
|
||||||
|
}
|
||||||
|
|
||||||
|
std::cout << "\n--- Parsed Tree ---\n" << tree.toString() << "\n";
|
||||||
|
}
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// Main Execution
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
int main() {
|
int main() {
|
||||||
std::string test = "UTF-STR WITH EMOJIS 😀🚀";
|
TokenFactory tf;
|
||||||
// Extracted UTF-8 chars:
|
assemblyGrammar();
|
||||||
std::u32string out;
|
|
||||||
bool r = spider::utf8::toUTF32(test, out);
|
|
||||||
|
|
||||||
std::cout << "INPUT: " << test << std::endl;
|
std::cout << "Running Token Framework Tests...\n";
|
||||||
std::cout << "RESULT: " << int(r) << std::endl;
|
|
||||||
for(spider::u32 ch : out) {
|
test_primitives_and_literals(tf);
|
||||||
std::cout << ch << " ";
|
test_choice_and_seq(tf);
|
||||||
|
test_opt_and_rep(tf);
|
||||||
|
test_literals(tf);
|
||||||
|
test_addressing_modes(tf);
|
||||||
|
test_instructions_and_lines(tf);
|
||||||
|
test_full_program(tf);
|
||||||
|
test_parse_tree(tf);
|
||||||
|
|
||||||
|
std::cout << "\n========================================\n";
|
||||||
|
std::cout << "Test Results: " << (total_tests - failed_tests) << "/" << total_tests << " passed.\n";
|
||||||
|
std::cout << "========================================\n";
|
||||||
|
|
||||||
|
if (failed_tests > 0) {
|
||||||
|
std::cerr << failed_tests << " test(s) FAILED.\n";
|
||||||
|
return 1;
|
||||||
}
|
}
|
||||||
std::cout << std::endl;
|
|
||||||
|
|
||||||
std::cout << "Happy Day!" << std::endl;
|
std::cout << "All token framework tests passed successfully!\n";
|
||||||
spider::utf8::hexdump(test.data(), test.size(), spider::pos(), std::cout);
|
|
||||||
std::cout << std::endl;
|
|
||||||
utf8sequences();
|
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1,10 +1,27 @@
|
|||||||
#pragma
|
#pragma once
|
||||||
|
|
||||||
#include <spider/compiler/common.hpp>
|
#include <spider/compiler/common.hpp>
|
||||||
|
|
||||||
|
#include <spider/compiler/text/Token.hpp>
|
||||||
|
#include <spider/compiler/text/ParseTree.hpp>
|
||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
|
|
||||||
class Token;
|
/**
|
||||||
class RootToken;
|
* @brief The token factory that owns the assembly grammar, loaded on first use.
|
||||||
|
* @details Every rule of the assembly language lives in this factory, so a caller
|
||||||
|
* can test a single rule or a whole program against the same grammar.
|
||||||
|
*/
|
||||||
|
TokenFactory& assemblyGrammar();
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Parses a whole assembly program into an inspectable tree of nodes.
|
||||||
|
* @details The grammar has to consume the entire source, so a malformed program
|
||||||
|
* is rejected instead of quietly producing a partial tree. On success the
|
||||||
|
* resulting tree can be walked like a document: program, line, label,
|
||||||
|
* instruction, operands, literals and comments.
|
||||||
|
* @returns True on success, with the parsed nodes left inside out.
|
||||||
|
*/
|
||||||
|
bool compileProgram(const std::string& source, ParseTree& out);
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2,12 +2,14 @@
|
|||||||
|
|
||||||
namespace spider::asm_ebnf {
|
namespace spider::asm_ebnf {
|
||||||
|
|
||||||
|
// Char Functions
|
||||||
|
|
||||||
bool isUTF8Alpha(u32 ch) {
|
bool isUTF8Alpha(u32 ch) {
|
||||||
return false;
|
return (u32('a') <= ch && ch <= u32('z')) || (u32('A') <= ch && ch <= u32('Z'));
|
||||||
}
|
}
|
||||||
|
|
||||||
bool isWhithespaceCharNotCrLf(u32 ch) {
|
bool isWhithespaceCharNotCrLf(u32 ch) {
|
||||||
return false;
|
return ch == u32(' ') || ch == u32('\t');
|
||||||
}
|
}
|
||||||
|
|
||||||
bool isUTF8CharNotCrLf(u32 ch) {
|
bool isUTF8CharNotCrLf(u32 ch) {
|
||||||
@@ -15,44 +17,197 @@ namespace spider::asm_ebnf {
|
|||||||
}
|
}
|
||||||
|
|
||||||
bool isUTF8CharLitCont(u32 ch) {
|
bool isUTF8CharLitCont(u32 ch) {
|
||||||
return ch != u32('\'');
|
return ch != u32('\'') && ch != u32('\\') && ch != u32('\r') && ch != u32('\n');
|
||||||
}
|
}
|
||||||
|
|
||||||
bool isUTF8StringLitCont(u32 ch) {
|
bool isUTF8StringLitCont(u32 ch) {
|
||||||
return ch != u32('"');
|
return ch != u32('"') && ch != u32('\\') && ch != u32('\r') && ch != u32('\n');
|
||||||
}
|
}
|
||||||
|
|
||||||
LitToken numbers[] = {
|
const Token* letter;
|
||||||
"0",
|
const Token* digit;
|
||||||
"1","2","3",
|
const Token* alpha_num_char;
|
||||||
"4","5","6",
|
|
||||||
"7","8","9",
|
|
||||||
};
|
|
||||||
|
|
||||||
LitToken hex_digits[][2] = {
|
const Token* hex_digit;
|
||||||
{"A", "a"},
|
const Token* octal_digit;
|
||||||
{"B", "b"},
|
const Token* binary_digit;
|
||||||
{"C", "c"},
|
|
||||||
{"D", "d"},
|
|
||||||
{"E", "e"},
|
|
||||||
{"F", "f"},
|
|
||||||
};
|
|
||||||
|
|
||||||
LitToken new_line[] = { "\r\n", "\r", "\n" };
|
const Token* ws_char;
|
||||||
|
const Token* ws_optional;
|
||||||
|
const Token* whitespace;
|
||||||
|
const Token* newline;
|
||||||
|
const Token* utf8_char;
|
||||||
|
|
||||||
LitToken symbols[] = {
|
const Token* char_escape;
|
||||||
"\\", "\'", "\"",
|
const Token* char_content;
|
||||||
"_" , ";" , "(" ,
|
const Token* char_lit;
|
||||||
"#",
|
|
||||||
"$" , "." , "+" , "-", ",", ")", "@", ":",
|
|
||||||
};
|
|
||||||
|
|
||||||
LitToken lit_letter[] = {
|
const Token* string_char;
|
||||||
"x", "c", "b"
|
const Token* string_lit;
|
||||||
};
|
|
||||||
|
|
||||||
LitToken type_letter[] = {
|
const Token* identifier;
|
||||||
"B", "S", "I", "L", "F", "D"
|
const Token* comment;
|
||||||
};
|
|
||||||
|
const Token* sign;
|
||||||
|
const Token* exponent_marker;
|
||||||
|
const Token* exponent;
|
||||||
|
|
||||||
|
const Token* decimal_lit;
|
||||||
|
const Token* float_lit;
|
||||||
|
|
||||||
|
const Token* hex_lit;
|
||||||
|
const Token* octal_lit;
|
||||||
|
const Token* binary_lit;
|
||||||
|
|
||||||
|
const Token* literal;
|
||||||
|
const Token* literal_cast;
|
||||||
|
const Token* literal_decl;
|
||||||
|
|
||||||
|
const Token* register_tok;
|
||||||
|
|
||||||
|
const Token* addrm_ind;
|
||||||
|
const Token* addrm_ptr;
|
||||||
|
const Token* addrm_idx;
|
||||||
|
const Token* addrm_sca;
|
||||||
|
const Token* addrm_dis;
|
||||||
|
|
||||||
|
const Token* addr_modes;
|
||||||
|
const Token* operand;
|
||||||
|
|
||||||
|
const Token* opcode;
|
||||||
|
const Token* operand_list;
|
||||||
|
const Token* instruction;
|
||||||
|
|
||||||
|
const Token* annotation_named;
|
||||||
|
const Token* annotation_arg;
|
||||||
|
const Token* annotation_args;
|
||||||
|
const Token* annotation_pars;
|
||||||
|
const Token* annotation;
|
||||||
|
|
||||||
|
const Token* preprocessor_val;
|
||||||
|
const Token* preprocessor;
|
||||||
|
|
||||||
|
const Token* label;
|
||||||
|
const Token* line_label;
|
||||||
|
const Token* line_annotation;
|
||||||
|
const Token* line_content;
|
||||||
|
const Token* line;
|
||||||
|
const Token* line_last;
|
||||||
|
const Token* program;
|
||||||
|
|
||||||
|
void initTokens(TokenFactory& tf) {
|
||||||
|
// (* Characters & Basic Predicates *)
|
||||||
|
letter = tf.fn(isUTF8Alpha);
|
||||||
|
digit = tf.choice("0123456789");
|
||||||
|
alpha_num_char = tf.choice({ letter, digit });
|
||||||
|
|
||||||
|
hex_digit = tf.choice("0123456789ABCDEFabcdef");
|
||||||
|
octal_digit = tf.choice("01234567");
|
||||||
|
binary_digit = tf.choice("01");
|
||||||
|
|
||||||
|
ws_char = tf.fn(isWhithespaceCharNotCrLf);
|
||||||
|
ws_optional = tf.rep(ws_char);
|
||||||
|
whitespace = tf.seq({ ws_char, tf.rep(ws_char) });
|
||||||
|
newline = tf.tag(tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }), "newline", true);
|
||||||
|
utf8_char = tf.fn(isUTF8CharNotCrLf);
|
||||||
|
|
||||||
|
char_escape = tf.seq({ tf["\\"], utf8_char });
|
||||||
|
char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
|
||||||
|
char_lit = tf.tag(tf.seq({ tf["'"], char_content, tf["'"] }), "char_lit", true);
|
||||||
|
|
||||||
|
string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
|
||||||
|
string_lit = tf.tag(tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }), "string_lit", true);
|
||||||
|
|
||||||
|
// (* Literals *)
|
||||||
|
identifier = tf.tag(tf.seq({
|
||||||
|
tf.choice({ letter, tf["_"] }),
|
||||||
|
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
|
||||||
|
}), "identifier", true);
|
||||||
|
|
||||||
|
comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true);
|
||||||
|
|
||||||
|
sign = tf.choice("+-");
|
||||||
|
exponent_marker = tf.choice("eE");
|
||||||
|
exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
|
||||||
|
|
||||||
|
decimal_lit = tf.tag(tf.seq({
|
||||||
|
tf.opt(sign),
|
||||||
|
digit,
|
||||||
|
tf.rep(digit),
|
||||||
|
tf.opt(tf.choice("BSIL"))
|
||||||
|
}), "decimal_lit", true);
|
||||||
|
|
||||||
|
float_lit = tf.tag(tf.seq({
|
||||||
|
tf.opt(sign),
|
||||||
|
tf.choice({
|
||||||
|
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
||||||
|
tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
||||||
|
tf.seq({ digit, tf.rep(digit), exponent })
|
||||||
|
}),
|
||||||
|
tf.opt(tf.choice("FD"))
|
||||||
|
}), "float_lit", true);
|
||||||
|
|
||||||
|
hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true);
|
||||||
|
octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true);
|
||||||
|
binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true);
|
||||||
|
|
||||||
|
literal = tf.tag(tf.choice({ hex_lit, octal_lit, binary_lit, float_lit, decimal_lit, string_lit, char_lit }), "literal");
|
||||||
|
literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast");
|
||||||
|
literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl");
|
||||||
|
|
||||||
|
// (* Operands *)
|
||||||
|
register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char, tf.not_(tf.choice({ alpha_num_char, tf["_"] })) }), "register", true);
|
||||||
|
|
||||||
|
addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true);
|
||||||
|
addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true);
|
||||||
|
|
||||||
|
addrm_idx = tf.tag(tf.seq({
|
||||||
|
tf["["], ws_optional, register_tok, ws_optional,
|
||||||
|
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||||
|
}), "addrm_idx", true);
|
||||||
|
|
||||||
|
addrm_sca = tf.tag(tf.seq({
|
||||||
|
tf["["], ws_optional, register_tok, ws_optional,
|
||||||
|
tf["+"], ws_optional, register_tok, ws_optional,
|
||||||
|
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||||
|
}), "addrm_sca", true);
|
||||||
|
|
||||||
|
addrm_dis = tf.tag(tf.seq({
|
||||||
|
tf["["], ws_optional, register_tok, ws_optional,
|
||||||
|
tf["+"], ws_optional, register_tok, ws_optional,
|
||||||
|
tf["*"], ws_optional, literal_decl, ws_optional,
|
||||||
|
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||||
|
}), "addrm_dis", true);
|
||||||
|
|
||||||
|
addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm");
|
||||||
|
operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand");
|
||||||
|
|
||||||
|
// (* Generalized Instructions *)
|
||||||
|
|
||||||
|
opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
|
||||||
|
operand_list = tf.tag(tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }), "operand_list", true);
|
||||||
|
instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction", true);
|
||||||
|
|
||||||
|
// (* Added Preprocessor, Annotation *)
|
||||||
|
|
||||||
|
annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named", true);
|
||||||
|
annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg", true);
|
||||||
|
annotation_args = tf.tag(tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }), "annotation_args", true);
|
||||||
|
annotation_pars = tf.tag(tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }), "annotation_pars", true);
|
||||||
|
annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation", true);
|
||||||
|
|
||||||
|
preprocessor_val = tf.choice({ identifier, literal_decl });
|
||||||
|
preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor", true);
|
||||||
|
|
||||||
|
// (* Line Structure & Program *)
|
||||||
|
|
||||||
|
label = tf.tag(tf.seq({ identifier, tf[":"] }), "label", true);
|
||||||
|
line_label = tf.tag(tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }), "line_label", true);
|
||||||
|
line_annotation = tf.tag(tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }), "line_annotation", true);
|
||||||
|
line_content = tf.tag(tf.choice({ preprocessor, line_annotation, line_label, instruction }), "line_content", true);
|
||||||
|
line = tf.tag(tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }), "line", true);
|
||||||
|
line_last = tf.tag(tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }), "line_last", true);
|
||||||
|
program = tf.tag(tf.seq({ tf.rep(line), tf.opt(line_last) }), "program", true);
|
||||||
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,6 +4,77 @@
|
|||||||
|
|
||||||
namespace spider::asm_ebnf {
|
namespace spider::asm_ebnf {
|
||||||
|
|
||||||
extern LitToken letter;
|
extern const Token* letter;
|
||||||
|
extern const Token* digit;
|
||||||
|
extern const Token* alpha_num_char;
|
||||||
|
|
||||||
|
extern const Token* hex_digit;
|
||||||
|
extern const Token* octal_digit;
|
||||||
|
extern const Token* binary_digit;
|
||||||
|
|
||||||
|
extern const Token* ws_char;
|
||||||
|
extern const Token* ws_optional;
|
||||||
|
extern const Token* whitespace;
|
||||||
|
extern const Token* newline;
|
||||||
|
extern const Token* utf8_char;
|
||||||
|
|
||||||
|
extern const Token* char_escape;
|
||||||
|
extern const Token* char_content;
|
||||||
|
extern const Token* char_lit;
|
||||||
|
|
||||||
|
extern const Token* string_char;
|
||||||
|
extern const Token* string_lit;
|
||||||
|
|
||||||
|
extern const Token* identifier;
|
||||||
|
extern const Token* comment;
|
||||||
|
|
||||||
|
extern const Token* sign;
|
||||||
|
extern const Token* exponent_marker;
|
||||||
|
extern const Token* exponent;
|
||||||
|
|
||||||
|
extern const Token* decimal_lit;
|
||||||
|
extern const Token* float_lit;
|
||||||
|
|
||||||
|
extern const Token* hex_lit;
|
||||||
|
extern const Token* octal_lit;
|
||||||
|
extern const Token* binary_lit;
|
||||||
|
|
||||||
|
extern const Token* literal;
|
||||||
|
extern const Token* literal_cast;
|
||||||
|
extern const Token* literal_decl;
|
||||||
|
|
||||||
|
extern const Token* register_tok;
|
||||||
|
|
||||||
|
extern const Token* addrm_ind;
|
||||||
|
extern const Token* addrm_ptr;
|
||||||
|
extern const Token* addrm_idx;
|
||||||
|
extern const Token* addrm_sca;
|
||||||
|
extern const Token* addrm_dis;
|
||||||
|
|
||||||
|
extern const Token* addr_modes;
|
||||||
|
extern const Token* operand;
|
||||||
|
|
||||||
|
extern const Token* opcode;
|
||||||
|
extern const Token* operand_list;
|
||||||
|
extern const Token* instruction;
|
||||||
|
|
||||||
|
extern const Token* annotation_named;
|
||||||
|
extern const Token* annotation_arg;
|
||||||
|
extern const Token* annotation_args;
|
||||||
|
extern const Token* annotation_pars;
|
||||||
|
extern const Token* annotation;
|
||||||
|
|
||||||
|
extern const Token* preprocessor_val;
|
||||||
|
extern const Token* preprocessor;
|
||||||
|
|
||||||
|
extern const Token* label;
|
||||||
|
extern const Token* line_label;
|
||||||
|
extern const Token* line_annotation;
|
||||||
|
extern const Token* line_content;
|
||||||
|
extern const Token* line;
|
||||||
|
extern const Token* line_last;
|
||||||
|
extern const Token* program;
|
||||||
|
|
||||||
|
void initTokens(TokenFactory& tf);
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,5 +1,10 @@
|
|||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
|
// Prevents conflicts if the runtime
|
||||||
|
// is included, which includes its own
|
||||||
|
// identical common.hpp
|
||||||
|
#ifndef SPIDER_RUNTIME_COMMON
|
||||||
|
|
||||||
#include <cstdint>
|
#include <cstdint>
|
||||||
#include <vector>
|
#include <vector>
|
||||||
#include <deque>
|
#include <deque>
|
||||||
@@ -62,3 +67,5 @@ namespace spider {
|
|||||||
};
|
};
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#endif
|
||||||
|
|||||||
@@ -0,0 +1,334 @@
|
|||||||
|
#include "ParseTree.hpp"
|
||||||
|
|
||||||
|
namespace spider {
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// Internal Helpers
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Renders control characters as escapes, so a node can be printed on one line.
|
||||||
|
*/
|
||||||
|
static std::string escapeText(std::string_view raw) {
|
||||||
|
std::string out;
|
||||||
|
out.reserve(raw.size());
|
||||||
|
for (char c : raw) {
|
||||||
|
switch (c) {
|
||||||
|
case '\n': out += "\\n"; break;
|
||||||
|
case '\r': out += "\\r"; break;
|
||||||
|
case '\t': out += "\\t"; break;
|
||||||
|
default: out += c; break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// ParseNode Navigation
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
isize ParseNode::depth() const {
|
||||||
|
isize levels = 0;
|
||||||
|
for (const ParseNode* n = parentNode(); n != nullptr; n = n->parentNode()) ++levels;
|
||||||
|
return levels;
|
||||||
|
}
|
||||||
|
|
||||||
|
const ParseNode* ParseNode::parentNode() const {
|
||||||
|
if (owner == nullptr || parent == ParseNode::npos) return nullptr;
|
||||||
|
return owner->resolve(parent);
|
||||||
|
}
|
||||||
|
|
||||||
|
const ParseNode* ParseNode::firstChild() const {
|
||||||
|
if (owner == nullptr || first_child == ParseNode::npos) return nullptr;
|
||||||
|
return owner->resolve(first_child);
|
||||||
|
}
|
||||||
|
|
||||||
|
const ParseNode* ParseNode::lastChild() const {
|
||||||
|
if (owner == nullptr || last_child == ParseNode::npos) return nullptr;
|
||||||
|
return owner->resolve(last_child);
|
||||||
|
}
|
||||||
|
|
||||||
|
const ParseNode* ParseNode::nextSibling() const {
|
||||||
|
if (owner == nullptr || next_sibling == ParseNode::npos) return nullptr;
|
||||||
|
return owner->resolve(next_sibling);
|
||||||
|
}
|
||||||
|
|
||||||
|
const ParseNode* ParseNode::previousSibling() const {
|
||||||
|
if (owner == nullptr || prev_sibling == ParseNode::npos) return nullptr;
|
||||||
|
return owner->resolve(prev_sibling);
|
||||||
|
}
|
||||||
|
|
||||||
|
const ParseNode* ParseNode::childAt(isize index) const {
|
||||||
|
isize seen = 0;
|
||||||
|
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||||
|
if (seen == index) return c;
|
||||||
|
++seen;
|
||||||
|
}
|
||||||
|
return nullptr;
|
||||||
|
}
|
||||||
|
|
||||||
|
ParseNode* ParseNode::parentNode() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->parentNode()); }
|
||||||
|
ParseNode* ParseNode::firstChild() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->firstChild()); }
|
||||||
|
ParseNode* ParseNode::lastChild() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->lastChild()); }
|
||||||
|
ParseNode* ParseNode::nextSibling() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->nextSibling()); }
|
||||||
|
ParseNode* ParseNode::previousSibling() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->previousSibling()); }
|
||||||
|
ParseNode* ParseNode::childAt(isize index) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->childAt(index)); }
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// ParseNode Navigation Filtered By Tag
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
const ParseNode* ParseNode::firstChild(std::string_view name) const {
|
||||||
|
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||||
|
if (c->hasTag(name)) return c;
|
||||||
|
}
|
||||||
|
return nullptr;
|
||||||
|
}
|
||||||
|
|
||||||
|
const ParseNode* ParseNode::lastChild(std::string_view name) const {
|
||||||
|
const ParseNode* found = nullptr;
|
||||||
|
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||||
|
if (c->hasTag(name)) found = c;
|
||||||
|
}
|
||||||
|
return found;
|
||||||
|
}
|
||||||
|
|
||||||
|
const ParseNode* ParseNode::nextSibling(std::string_view name) const {
|
||||||
|
for (const ParseNode* s = nextSibling(); s != nullptr; s = s->nextSibling()) {
|
||||||
|
if (s->hasTag(name)) return s;
|
||||||
|
}
|
||||||
|
return nullptr;
|
||||||
|
}
|
||||||
|
|
||||||
|
const ParseNode* ParseNode::ancestor(std::string_view name) const {
|
||||||
|
for (const ParseNode* p = parentNode(); p != nullptr; p = p->parentNode()) {
|
||||||
|
if (p->hasTag(name)) return p;
|
||||||
|
}
|
||||||
|
return nullptr;
|
||||||
|
}
|
||||||
|
|
||||||
|
ParseNode* ParseNode::firstChild(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->firstChild(name)); }
|
||||||
|
ParseNode* ParseNode::lastChild(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->lastChild(name)); }
|
||||||
|
ParseNode* ParseNode::nextSibling(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->nextSibling(name)); }
|
||||||
|
ParseNode* ParseNode::ancestor(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->ancestor(name)); }
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// ParseNode Queries
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
const ParseNode* ParseNode::find(std::string_view name) const {
|
||||||
|
if (hasTag(name)) return this;
|
||||||
|
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||||
|
if (const ParseNode* hit = c->find(name)) return hit;
|
||||||
|
}
|
||||||
|
return nullptr;
|
||||||
|
}
|
||||||
|
|
||||||
|
vector<const ParseNode*> ParseNode::findAll(std::string_view name) const {
|
||||||
|
vector<const ParseNode*> hits;
|
||||||
|
if (hasTag(name)) hits.push_back(this);
|
||||||
|
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||||
|
for (const ParseNode* hit : c->findAll(name)) hits.push_back(hit);
|
||||||
|
}
|
||||||
|
return hits;
|
||||||
|
}
|
||||||
|
|
||||||
|
isize ParseNode::count(std::string_view name) const { return findAll(name).size(); }
|
||||||
|
|
||||||
|
ParseNode* ParseNode::find(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->find(name)); }
|
||||||
|
|
||||||
|
vector<ParseNode*> ParseNode::findAll(std::string_view name) {
|
||||||
|
vector<const ParseNode*> hits = static_cast<const ParseNode*>(this)->findAll(name);
|
||||||
|
vector<ParseNode*> out;
|
||||||
|
out.reserve(hits.size());
|
||||||
|
for (const ParseNode* hit : hits) out.push_back(const_cast<ParseNode*>(hit));
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// ParseNode Rendering
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
std::string ParseNode::describe(isize depth) const {
|
||||||
|
const std::string pad(depth * 2, ' ');
|
||||||
|
|
||||||
|
if (!tag.has_value()) {
|
||||||
|
if (isLeaf()) return pad + escapeText(ownTextUtf8());
|
||||||
|
|
||||||
|
std::string out;
|
||||||
|
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||||
|
out += c->describe(depth) + "\n";
|
||||||
|
}
|
||||||
|
if (!out.empty()) out.pop_back();
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
const std::string name(*tag);
|
||||||
|
if (isLeaf()) return pad + "<" + name + ">" + escapeText(ownTextUtf8()) + "</" + name + ">";
|
||||||
|
|
||||||
|
std::string out = pad + "<" + name + ">";
|
||||||
|
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||||
|
out += "\n" + c->describe(depth + 1);
|
||||||
|
}
|
||||||
|
out += "\n" + pad + "</" + name + ">";
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
std::string ParseNode::toString() const { return describe(0); }
|
||||||
|
std::string ParseNode::toString(isize depth) const { return describe(depth); }
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// ParseTree Building
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
ParseNode* ParseTree::resolve(isize index) {
|
||||||
|
if (index == ParseNode::npos) return nullptr;
|
||||||
|
return &nodes[index];
|
||||||
|
}
|
||||||
|
|
||||||
|
const ParseNode* ParseTree::resolve(isize index) const {
|
||||||
|
if (index == ParseNode::npos) return nullptr;
|
||||||
|
return &nodes[index];
|
||||||
|
}
|
||||||
|
|
||||||
|
isize ParseTree::addNode(const TokenResult& result) {
|
||||||
|
ParseNode node;
|
||||||
|
node.owner = this;
|
||||||
|
node.tag = result.tag;
|
||||||
|
node.own = result.match;
|
||||||
|
node.text = result.flatMatch();
|
||||||
|
|
||||||
|
isize self = nodes.size();
|
||||||
|
nodes.push_back(std::move(node));
|
||||||
|
return self;
|
||||||
|
}
|
||||||
|
|
||||||
|
void ParseTree::linkChildren(isize parent_index, const vector<isize>& children) {
|
||||||
|
ParseNode& parent = nodes[parent_index];
|
||||||
|
parent.first_child = ParseNode::npos;
|
||||||
|
parent.last_child = ParseNode::npos;
|
||||||
|
parent.children_count = 0;
|
||||||
|
|
||||||
|
for (isize child : children) {
|
||||||
|
ParseNode& kid = nodes[child];
|
||||||
|
kid.parent = parent_index;
|
||||||
|
kid.prev_sibling = parent.last_child;
|
||||||
|
kid.next_sibling = ParseNode::npos;
|
||||||
|
|
||||||
|
if (parent.last_child != ParseNode::npos) {
|
||||||
|
nodes[parent.last_child].next_sibling = child;
|
||||||
|
} else {
|
||||||
|
parent.first_child = child;
|
||||||
|
}
|
||||||
|
parent.last_child = child;
|
||||||
|
parent.children_count++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
vector<isize> ParseTree::collapse(const TokenResult& result) {
|
||||||
|
vector<isize> kids;
|
||||||
|
for (const TokenResult& sub : result.child) {
|
||||||
|
for (isize kid : collapse(sub)) kids.push_back(kid);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Untagged rules are grammar scaffolding, never part of the exposed syntax.
|
||||||
|
if (!result.tag.has_value()) return kids;
|
||||||
|
|
||||||
|
const std::u32string folded = result.flatMatch();
|
||||||
|
isize self = addNode(result);
|
||||||
|
|
||||||
|
// An empty shell has nothing to show.
|
||||||
|
if (kids.empty() && folded.empty()) return {};
|
||||||
|
|
||||||
|
linkChildren(self, kids);
|
||||||
|
return { self };
|
||||||
|
}
|
||||||
|
|
||||||
|
void ParseTree::build(const TokenResult& result) {
|
||||||
|
nodes.clear();
|
||||||
|
root_index = ParseNode::npos;
|
||||||
|
if (!result.success) return;
|
||||||
|
|
||||||
|
vector<isize> tops = collapse(result);
|
||||||
|
if (tops.size() == 1) {
|
||||||
|
root_index = tops.front();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// A grammar that exposes several top level rules still gets one container.
|
||||||
|
ParseNode root;
|
||||||
|
root.owner = this;
|
||||||
|
root_index = nodes.size();
|
||||||
|
nodes.push_back(std::move(root));
|
||||||
|
linkChildren(root_index, tops);
|
||||||
|
}
|
||||||
|
|
||||||
|
bool ParseTree::parsePrefix(const Token* token, TextReader& reader) {
|
||||||
|
nodes.clear();
|
||||||
|
root_index = ParseNode::npos;
|
||||||
|
if (token == nullptr) return false;
|
||||||
|
|
||||||
|
TokenResult result = token->test(reader);
|
||||||
|
if (!result.success) return false;
|
||||||
|
|
||||||
|
build(result);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool ParseTree::parse(const Token* token, TextReader& reader) {
|
||||||
|
nodes.clear();
|
||||||
|
root_index = ParseNode::npos;
|
||||||
|
if (token == nullptr) return false;
|
||||||
|
|
||||||
|
TokenResult result = token->test(reader);
|
||||||
|
if (!result.success) return false;
|
||||||
|
|
||||||
|
// A complete parse leaves nothing behind in the reader.
|
||||||
|
if (reader.hasError()) return false;
|
||||||
|
if (reader.current().has_value()) return false;
|
||||||
|
|
||||||
|
build(result);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool ParseTree::parse(const Token* token, const std::string& source) {
|
||||||
|
StringTextReader reader(source);
|
||||||
|
return parse(token, reader);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// ParseTree Inspection
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
const ParseNode* ParseTree::find(std::string_view name) const {
|
||||||
|
const ParseNode* r = root();
|
||||||
|
return r == nullptr ? nullptr : r->find(name);
|
||||||
|
}
|
||||||
|
|
||||||
|
vector<const ParseNode*> ParseTree::findAll(std::string_view name) const {
|
||||||
|
const ParseNode* r = root();
|
||||||
|
if (r == nullptr) return {};
|
||||||
|
return r->findAll(name);
|
||||||
|
}
|
||||||
|
|
||||||
|
isize ParseTree::count(std::string_view name) const {
|
||||||
|
const ParseNode* r = root();
|
||||||
|
return r == nullptr ? 0 : r->count(name);
|
||||||
|
}
|
||||||
|
|
||||||
|
std::string ParseTree::toString() const {
|
||||||
|
const ParseNode* r = root();
|
||||||
|
return r == nullptr ? std::string() : r->toString();
|
||||||
|
}
|
||||||
|
|
||||||
|
ParseNode* ParseTree::find(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseTree*>(this)->find(name)); }
|
||||||
|
|
||||||
|
vector<ParseNode*> ParseTree::findAll(std::string_view name) {
|
||||||
|
vector<const ParseNode*> hits = static_cast<const ParseTree*>(this)->findAll(name);
|
||||||
|
vector<ParseNode*> out;
|
||||||
|
out.reserve(hits.size());
|
||||||
|
for (const ParseNode* hit : hits) out.push_back(const_cast<ParseNode*>(hit));
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
}
|
||||||
@@ -0,0 +1,295 @@
|
|||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include <spider/compiler/common.hpp>
|
||||||
|
|
||||||
|
#include <spider/compiler/text/unicode.hpp>
|
||||||
|
#include <spider/compiler/text/Token.hpp>
|
||||||
|
|
||||||
|
namespace spider {
|
||||||
|
|
||||||
|
class ParseTree;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief DOM style handle over a single node of a parsed token tree.
|
||||||
|
* @details Every node knows its parent, its children and its siblings, so a
|
||||||
|
* parsed program can be walked and inspected much like a document.
|
||||||
|
* Nodes are owned by the ParseTree that produced them and stay
|
||||||
|
* valid while that tree is alive. The tree itself is built by
|
||||||
|
* ParseTree, never by hand.
|
||||||
|
*/
|
||||||
|
class ParseNode {
|
||||||
|
friend class ParseTree;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
/** @brief Index value used for "no node" links, since isize is unsigned. */
|
||||||
|
static constexpr isize npos = static_cast<isize>(-1);
|
||||||
|
|
||||||
|
private:
|
||||||
|
|
||||||
|
ParseTree* owner = nullptr;
|
||||||
|
|
||||||
|
isize parent = npos;
|
||||||
|
isize first_child = npos;
|
||||||
|
isize last_child = npos;
|
||||||
|
isize next_sibling = npos;
|
||||||
|
isize prev_sibling = npos;
|
||||||
|
isize children_count = 0;
|
||||||
|
|
||||||
|
optional<std::string_view> tag = {};
|
||||||
|
std::u32string own = U"";
|
||||||
|
std::u32string text = U"";
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
ParseNode() = default;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief The grammar tag of this node, or nothing if untagged.
|
||||||
|
*/
|
||||||
|
const optional<std::string_view>& tagName() const { return tag; }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Checks if this node carries the given tag.
|
||||||
|
*/
|
||||||
|
bool hasTag(std::string_view name) const { return tag.has_value() && *tag == name; }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief The full source text covered by this node and all of its children.
|
||||||
|
*/
|
||||||
|
const std::u32string& fullText() const { return text; }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief The source text held by this node alone, without its children.
|
||||||
|
* @note For tagged nodes that fold their children, this is the full text.
|
||||||
|
*/
|
||||||
|
const std::u32string& ownText() const { return own; }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief UTF-8 rendering of text().
|
||||||
|
*/
|
||||||
|
std::string textUtf8() const { return unicode::toUTF8(text); }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief UTF-8 rendering of ownText().
|
||||||
|
*/
|
||||||
|
std::string ownTextUtf8() const { return unicode::toUTF8(own); }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Amount of direct children of this node.
|
||||||
|
*/
|
||||||
|
isize childCount() const { return children_count; }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief True when this node holds no text and no children.
|
||||||
|
*/
|
||||||
|
bool isLeaf() const { return children_count == 0; }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Distance from the root of the tree, zero for the root itself.
|
||||||
|
*/
|
||||||
|
isize depth() const;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Storage index of this node inside its parent, npos for the root.
|
||||||
|
*/
|
||||||
|
isize indexInParent() const { return parent; }
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------- //
|
||||||
|
// Navigation //
|
||||||
|
// ---------------------------------------------------------------- //
|
||||||
|
|
||||||
|
ParseNode* parentNode();
|
||||||
|
const ParseNode* parentNode() const;
|
||||||
|
|
||||||
|
ParseNode* firstChild();
|
||||||
|
const ParseNode* firstChild() const;
|
||||||
|
|
||||||
|
ParseNode* lastChild();
|
||||||
|
const ParseNode* lastChild() const;
|
||||||
|
|
||||||
|
ParseNode* nextSibling();
|
||||||
|
const ParseNode* nextSibling() const;
|
||||||
|
|
||||||
|
ParseNode* previousSibling();
|
||||||
|
const ParseNode* previousSibling() const;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief The index-th direct child of this node, nullptr when out of range.
|
||||||
|
*/
|
||||||
|
ParseNode* childAt(isize index);
|
||||||
|
const ParseNode* childAt(isize index) const;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------- //
|
||||||
|
// Navigation filtered by tag //
|
||||||
|
// ---------------------------------------------------------------- //
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief First direct child carrying the given tag.
|
||||||
|
*/
|
||||||
|
ParseNode* firstChild(std::string_view name);
|
||||||
|
const ParseNode* firstChild(std::string_view name) const;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Last direct child carrying the given tag.
|
||||||
|
*/
|
||||||
|
ParseNode* lastChild(std::string_view name);
|
||||||
|
const ParseNode* lastChild(std::string_view name) const;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Next sibling of this node carrying the given tag.
|
||||||
|
*/
|
||||||
|
ParseNode* nextSibling(std::string_view name);
|
||||||
|
const ParseNode* nextSibling(std::string_view name) const;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Nearest ancestor carrying the given tag, nullptr when there is none.
|
||||||
|
*/
|
||||||
|
ParseNode* ancestor(std::string_view name);
|
||||||
|
const ParseNode* ancestor(std::string_view name) const;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------- //
|
||||||
|
// Queries //
|
||||||
|
// ---------------------------------------------------------------- //
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief First node in document order carrying the given tag, this node included.
|
||||||
|
*/
|
||||||
|
ParseNode* find(std::string_view name);
|
||||||
|
const ParseNode* find(std::string_view name) const;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Every node in document order carrying the given tag, this node included.
|
||||||
|
*/
|
||||||
|
vector<ParseNode*> findAll(std::string_view name);
|
||||||
|
vector<const ParseNode*> findAll(std::string_view name) const;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Amount of nodes in document order carrying the given tag.
|
||||||
|
*/
|
||||||
|
isize count(std::string_view name) const;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief True when at least one node carries the given tag.
|
||||||
|
*/
|
||||||
|
bool contains(std::string_view name) const { return find(name) != nullptr; }
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Human readable XML-like rendering of this node and its children.
|
||||||
|
*/
|
||||||
|
std::string toString() const;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief XML-like rendering of this node alone, indented by the given depth.
|
||||||
|
*/
|
||||||
|
std::string toString(isize depth) const;
|
||||||
|
|
||||||
|
private:
|
||||||
|
|
||||||
|
std::string describe(isize depth) const;
|
||||||
|
|
||||||
|
};
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Owner of a parsed token tree, exposing a document like interface.
|
||||||
|
* @details Use parse() to turn source text into an inspectable tree. The tree
|
||||||
|
* is materialized once and then only read, so any number of consumers
|
||||||
|
* can walk the same nodes safely.
|
||||||
|
*/
|
||||||
|
class ParseTree {
|
||||||
|
friend class ParseNode;
|
||||||
|
|
||||||
|
private:
|
||||||
|
|
||||||
|
deque<ParseNode> nodes;
|
||||||
|
isize root_index = ParseNode::npos;
|
||||||
|
|
||||||
|
private:
|
||||||
|
|
||||||
|
ParseNode* resolve(isize index);
|
||||||
|
const ParseNode* resolve(isize index) const;
|
||||||
|
|
||||||
|
isize addNode(const TokenResult& result);
|
||||||
|
|
||||||
|
void linkChildren(isize parent, const vector<isize>& children);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Materializes a parse result, dropping the grammar scaffolding.
|
||||||
|
* @details Only tagged rules become nodes, so a consumer walks syntax and not
|
||||||
|
* combinators: untagged rules are pure plumbing and simply hoist their
|
||||||
|
* children upwards, and empty shells are discarded. Returns the top
|
||||||
|
* level nodes produced by this subtree.
|
||||||
|
*/
|
||||||
|
vector<isize> collapse(const TokenResult& result);
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
ParseTree() = default;
|
||||||
|
~ParseTree() = default;
|
||||||
|
|
||||||
|
// Nodes point back at their owning tree, so trees are never copied or moved.
|
||||||
|
ParseTree(const ParseTree&) = delete;
|
||||||
|
ParseTree& operator=(const ParseTree&) = delete;
|
||||||
|
ParseTree(ParseTree&&) = delete;
|
||||||
|
ParseTree& operator=(ParseTree&&) = delete;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Parses source text, requiring the grammar to consume all of it.
|
||||||
|
* @returns True on success, in which case the tree holds the parsed nodes.
|
||||||
|
*/
|
||||||
|
bool parse(const Token* token, const std::string& source);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Parses from a reader, requiring the grammar to consume all of it.
|
||||||
|
* @returns True on success, in which case the tree holds the parsed nodes.
|
||||||
|
*/
|
||||||
|
bool parse(const Token* token, TextReader& reader);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Parses the longest matching prefix of a reader.
|
||||||
|
* @returns True when at least something was matched.
|
||||||
|
*/
|
||||||
|
bool parsePrefix(const Token* token, TextReader& reader);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Adopts an already produced parse result.
|
||||||
|
*/
|
||||||
|
void build(const TokenResult& result);
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
ParseNode* root() { return resolve(root_index); }
|
||||||
|
const ParseNode* root() const { return resolve(root_index); }
|
||||||
|
|
||||||
|
bool empty() const { return root_index == ParseNode::npos; }
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
ParseNode* find(std::string_view name);
|
||||||
|
const ParseNode* find(std::string_view name) const;
|
||||||
|
|
||||||
|
vector<ParseNode*> findAll(std::string_view name);
|
||||||
|
vector<const ParseNode*> findAll(std::string_view name) const;
|
||||||
|
|
||||||
|
isize count(std::string_view name) const;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Human readable XML-like rendering of the whole tree.
|
||||||
|
*/
|
||||||
|
std::string toString() const;
|
||||||
|
|
||||||
|
};
|
||||||
|
|
||||||
|
}
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
#include "TextReader.hpp"
|
#include "TextReader.hpp"
|
||||||
|
|
||||||
#include <spider/compiler/text/utf8.hpp>
|
#include <spider/compiler/text/unicode.hpp>
|
||||||
|
|
||||||
#include <stdexcept>
|
#include <stdexcept>
|
||||||
|
|
||||||
@@ -8,11 +8,7 @@ namespace spider {
|
|||||||
|
|
||||||
// Text Reader //
|
// Text Reader //
|
||||||
|
|
||||||
TextReader::TextReader() : err(false), eof(false), bufferIndex(0) {
|
TextReader::TextReader() : err(false), eof(false), bufferIndex(0) {}
|
||||||
// Prime the buffer with the first character
|
|
||||||
// so current() is immediately valid
|
|
||||||
fillBufferTo(0);
|
|
||||||
}
|
|
||||||
|
|
||||||
TextReader::~TextReader() {}
|
TextReader::~TextReader() {}
|
||||||
|
|
||||||
@@ -32,19 +28,30 @@ namespace spider {
|
|||||||
// instead of whatever that was, convert to UTF-32
|
// instead of whatever that was, convert to UTF-32
|
||||||
// and then do an easy compare!
|
// and then do an easy compare!
|
||||||
std::u32string str;
|
std::u32string str;
|
||||||
if(!utf8::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!");
|
if(!unicode::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!");
|
||||||
return eat(str);
|
return eat(str);
|
||||||
}
|
}
|
||||||
|
|
||||||
bool TextReader::eat(const std::u32string& str) {
|
bool TextReader::eat(const std::u32string& str) {
|
||||||
|
// case 0: no str
|
||||||
|
if(str.empty()) return true;
|
||||||
|
|
||||||
|
// prepare n chars
|
||||||
|
isize index_space = str.size() - 1;
|
||||||
|
fillBufferTo(index_space);
|
||||||
|
|
||||||
|
// fast reject
|
||||||
|
if(!hasBufferTo(index_space)) return false;
|
||||||
|
|
||||||
// compare now
|
// compare now
|
||||||
isize index;
|
for(isize i = 0; i <= index_space; i++) {
|
||||||
for(index = 0; index < str.size(); index++) {
|
if(str[i] != buffer[bufferIndex + i]) {
|
||||||
if(str[index] != peekChar(index)) return false;
|
return false;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// success!
|
// success!
|
||||||
nextChar(index);
|
consumeChars(str.size());
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -68,37 +75,30 @@ namespace spider {
|
|||||||
return char(ch);
|
return char(ch);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Returns the current character.
|
|
||||||
*/
|
|
||||||
optional<u32> TextReader::current() {
|
optional<u32> TextReader::current() {
|
||||||
|
fillBufferTo(0);
|
||||||
if (bufferIndex < buffer.size()) {
|
if (bufferIndex < buffer.size()) {
|
||||||
return buffer[bufferIndex];
|
return buffer[bufferIndex];
|
||||||
}
|
}
|
||||||
return {};
|
return {};
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Reads the next character and advances the position tracker.
|
|
||||||
*/
|
|
||||||
optional<u32> TextReader::nextChar(isize n) {
|
optional<u32> TextReader::nextChar(isize n) {
|
||||||
// Ensure the character we are moving TO exists
|
// Ensure the character we are moving TO exists
|
||||||
if (fillBufferTo(n)) {
|
fillBufferTo(n); // index = n will be accessible
|
||||||
// advance n characters
|
// from [0, n] inclusive, equal to (n + 1) chars
|
||||||
while(n--) {
|
|
||||||
advance(buffer[bufferIndex]);
|
// remember partial success
|
||||||
bufferIndex++;
|
consumeChars(n); // n chars will be removed
|
||||||
}
|
|
||||||
return current();
|
// return current char
|
||||||
}
|
// current char, index = 0
|
||||||
return {};
|
return current();
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Keeps the next n-th character (n = 0 is current).
|
|
||||||
*/
|
|
||||||
optional<u32> TextReader::peekChar(isize n) {
|
optional<u32> TextReader::peekChar(isize n) {
|
||||||
if (fillBufferTo(n)) return buffer[bufferIndex + n];
|
fillBufferTo(n);
|
||||||
|
if (hasBufferTo(n)) return buffer[bufferIndex + n];
|
||||||
return {};
|
return {};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -110,12 +110,16 @@ namespace spider {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
isize TextReader::push() {
|
TextReader::State TextReader::push() {
|
||||||
return bufferIndex;
|
return { .err = err, .eof = eof, .at = at, .errmsg = errmsg, .index = bufferIndex };
|
||||||
}
|
}
|
||||||
|
|
||||||
void TextReader::pop(isize index) {
|
void TextReader::pop(TextReader::State s) {
|
||||||
bufferIndex = std::min(index, bufferIndex);
|
err = s.err;
|
||||||
|
eof = s.eof;
|
||||||
|
at = s.at;
|
||||||
|
errmsg = s.errmsg;
|
||||||
|
bufferIndex = s.index;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -139,7 +143,7 @@ namespace spider {
|
|||||||
if (err) return false;
|
if (err) return false;
|
||||||
if (eof) return false;
|
if (eof) return false;
|
||||||
|
|
||||||
isize chsize = utf8::seqlen(u8(bytes[bindex]));
|
isize chsize = unicode::seqlen(u8(bytes[bindex]));
|
||||||
if(chsize == 0) {
|
if(chsize == 0) {
|
||||||
err = true;
|
err = true;
|
||||||
errmsg = "Invalid start of UTF-8 sequence.";
|
errmsg = "Invalid start of UTF-8 sequence.";
|
||||||
@@ -154,7 +158,7 @@ namespace spider {
|
|||||||
}
|
}
|
||||||
|
|
||||||
u32 decodedChar;
|
u32 decodedChar;
|
||||||
if(!utf8::decodeArr(bytes, chsize, decodedChar)) {
|
if(!unicode::decodeArr(bytes, chsize, decodedChar)) {
|
||||||
err = true;
|
err = true;
|
||||||
errmsg = "Invalid UTF-8 sequence.";
|
errmsg = "Invalid UTF-8 sequence.";
|
||||||
return false;
|
return false;
|
||||||
@@ -163,19 +167,27 @@ namespace spider {
|
|||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Fills the buffer sequentially until it contains at least up
|
|
||||||
* to (bufferIndex + targetOffset).
|
|
||||||
*/
|
|
||||||
bool TextReader::fillBufferTo(isize targetOffset) {
|
bool TextReader::fillBufferTo(isize targetOffset) {
|
||||||
isize targetSize = bufferIndex + targetOffset + 1;
|
isize targetSize = bufferIndex + targetOffset;
|
||||||
while (buffer.size() < targetSize) {
|
while (targetSize >= buffer.size()) {
|
||||||
if(readChar()) continue;
|
if(readChar()) continue;
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
bool TextReader::hasBufferTo(isize index) {
|
||||||
|
return bufferIndex + index < buffer.size();
|
||||||
|
}
|
||||||
|
|
||||||
|
void TextReader::consumeChars(isize n) {
|
||||||
|
// advance up to specified char.
|
||||||
|
for(isize i = 0; i < n && hasBufferTo(i); i++) {
|
||||||
|
advance(buffer[bufferIndex]);
|
||||||
|
}
|
||||||
|
bufferIndex += n;
|
||||||
|
}
|
||||||
|
|
||||||
pos TextReader::getPosition() const {
|
pos TextReader::getPosition() const {
|
||||||
return at;
|
return at;
|
||||||
}
|
}
|
||||||
@@ -212,24 +224,24 @@ namespace spider {
|
|||||||
// String Reader //
|
// String Reader //
|
||||||
|
|
||||||
StringTextReader::StringTextReader(std::string initialText)
|
StringTextReader::StringTextReader(std::string initialText)
|
||||||
: buffer(std::move(initialText)),
|
: txt_buffer(std::move(initialText)),
|
||||||
stringStream(std::make_unique<std::istringstream>(buffer)) {
|
stringStream(std::make_unique<std::istringstream>(txt_buffer)) { }
|
||||||
}
|
|
||||||
|
|
||||||
std::istream& StringTextReader::getStream() {
|
std::istream& StringTextReader::getStream() {
|
||||||
return *stringStream;
|
return *stringStream;
|
||||||
}
|
}
|
||||||
|
|
||||||
void StringTextReader::set(const std::string& newText) {
|
void StringTextReader::set(const std::string& newText) {
|
||||||
buffer = newText;
|
txt_buffer = newText;
|
||||||
stringStream = std::make_unique<std::istringstream>(buffer);
|
stringStream = std::make_unique<std::istringstream>(txt_buffer);
|
||||||
}
|
|
||||||
|
|
||||||
void StringTextReader::append(const std::string& extraText) {
|
buffer.clear();
|
||||||
std::streampos pos = stringStream->tellg();
|
txt_buffer.clear();
|
||||||
buffer += extraText;
|
|
||||||
stringStream = std::make_unique<std::istringstream>(buffer);
|
bufferIndex = 0;
|
||||||
stringStream->seekg(pos);
|
err = false;
|
||||||
|
eof = false;
|
||||||
|
at = pos();
|
||||||
}
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -34,11 +34,6 @@ namespace spider {
|
|||||||
|
|
||||||
std::string errmsg;
|
std::string errmsg;
|
||||||
|
|
||||||
struct stored_char {
|
|
||||||
u8 byte_count;
|
|
||||||
u32 value;
|
|
||||||
};
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Buffer of extracted characters.
|
* Buffer of extracted characters.
|
||||||
*/
|
*/
|
||||||
@@ -52,6 +47,16 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
isize bufferIndex;
|
isize bufferIndex;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
struct State {
|
||||||
|
bool err;
|
||||||
|
bool eof;
|
||||||
|
pos at;
|
||||||
|
std::string errmsg;
|
||||||
|
isize index;
|
||||||
|
};
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
|
||||||
TextReader();
|
TextReader();
|
||||||
@@ -94,13 +99,16 @@ namespace spider {
|
|||||||
optional<u32> current();
|
optional<u32> current();
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Reads the next n-th character.
|
* Skips n number of characters and returns
|
||||||
|
* the current one in that position.
|
||||||
* n = 0 is a noop, since it's the current one.
|
* n = 0 is a noop, since it's the current one.
|
||||||
|
*
|
||||||
|
* Will advance until the EOF is reached.
|
||||||
*/
|
*/
|
||||||
optional<u32> nextChar(isize n = 1);
|
optional<u32> nextChar(isize n = 1);
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Keeps the next n-th character
|
* Returns the n-th character following the current one.
|
||||||
* n = 0 is the current one.
|
* n = 0 is the current one.
|
||||||
*/
|
*/
|
||||||
optional<u32> peekChar(isize n = 1);
|
optional<u32> peekChar(isize n = 1);
|
||||||
@@ -121,14 +129,14 @@ namespace spider {
|
|||||||
* Inside a parser, this allows to roll
|
* Inside a parser, this allows to roll
|
||||||
* back the index to a specific position.
|
* back the index to a specific position.
|
||||||
*/
|
*/
|
||||||
isize push();
|
State push();
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Sets the current buffer index.
|
* Sets the current buffer index.
|
||||||
* Inside a parser, rolls back to
|
* Inside a parser, rolls back to
|
||||||
* a previous position.
|
* a previous position.
|
||||||
*/
|
*/
|
||||||
void pop(isize index);
|
void pop(State s);
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Returns true if the end of the stream has been reached.
|
* Returns true if the end of the stream has been reached.
|
||||||
@@ -160,8 +168,29 @@ namespace spider {
|
|||||||
|
|
||||||
virtual std::istream& getStream() = 0;
|
virtual std::istream& getStream() = 0;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Fills the buffer sequentially until it the passed
|
||||||
|
* index can be safely accessed, relative to the current
|
||||||
|
* buffer position.
|
||||||
|
*
|
||||||
|
* Returns false if that index could not be reached.
|
||||||
|
* Partial success is possible, check buffer.size()!
|
||||||
|
*/
|
||||||
bool fillBufferTo(isize index);
|
bool fillBufferTo(isize index);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Verifies that the index can be safely accessed,
|
||||||
|
* relative to the current buffer position.
|
||||||
|
*/
|
||||||
|
bool hasBufferTo(isize index);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Triggers the buffer to consume this number
|
||||||
|
* of characters from the buffer. This is a reverseable
|
||||||
|
* operation.
|
||||||
|
*/
|
||||||
|
void consumeChars(isize n);
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -188,7 +217,7 @@ namespace spider {
|
|||||||
class StringTextReader : public TextReader {
|
class StringTextReader : public TextReader {
|
||||||
private:
|
private:
|
||||||
|
|
||||||
std::string buffer;
|
std::string txt_buffer;
|
||||||
std::unique_ptr<std::istringstream> stringStream;
|
std::unique_ptr<std::istringstream> stringStream;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
@@ -199,8 +228,6 @@ namespace spider {
|
|||||||
|
|
||||||
void set(const std::string& newText);
|
void set(const std::string& newText);
|
||||||
|
|
||||||
void append(const std::string& extraText);
|
|
||||||
|
|
||||||
protected:
|
protected:
|
||||||
|
|
||||||
std::istream& getStream() override;
|
std::istream& getStream() override;
|
||||||
|
|||||||
@@ -1,25 +1,94 @@
|
|||||||
#include "Token.hpp"
|
#include "Token.hpp"
|
||||||
|
|
||||||
|
#include <span>
|
||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// Token Implementation
|
// Token Factory
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
SeqToken Token::operator&(const Token& tok) {
|
Token* TokenFactory::lit(std::string_view text) {
|
||||||
return SeqToken({ tok, *this });
|
auto it = lit_cache.find(std::string(text));
|
||||||
|
if (it != lit_cache.end()) return it->second;
|
||||||
|
|
||||||
|
auto p = std::make_unique<LitToken>(text);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
lit_cache.emplace(text, t);
|
||||||
|
return t;
|
||||||
}
|
}
|
||||||
|
|
||||||
OrToken Token::operator|(const Token& tok) {
|
Token* TokenFactory::operator[](std::string_view text) {
|
||||||
return OrToken({ tok, *this });
|
return lit(text);
|
||||||
}
|
}
|
||||||
|
|
||||||
OptToken Token::operator~() {
|
Token* TokenFactory::fn(FnTokenFn predicate) {
|
||||||
return OptToken(*this);
|
uptr<Token> p = std::make_unique<FnToken>(predicate);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
}
|
}
|
||||||
|
|
||||||
RepToken Token::operator*() {
|
Token* TokenFactory::seq(const vector<const Token*>& tokens) {
|
||||||
return RepToken(*this);
|
uptr<Token> p = std::make_unique<SeqToken>(tokens);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
|
}
|
||||||
|
|
||||||
|
Token* TokenFactory::choice(std::string_view opts) {
|
||||||
|
vector<const Token*> toks;
|
||||||
|
|
||||||
|
for (char c : opts) {
|
||||||
|
std::string s = std::string(1, c);
|
||||||
|
toks.push_back(lit(s));
|
||||||
|
}
|
||||||
|
|
||||||
|
return choice(toks);
|
||||||
|
}
|
||||||
|
|
||||||
|
Token* TokenFactory::choice(const vector<const Token*>& tokens) {
|
||||||
|
uptr<Token> p = std::make_unique<OrToken>(tokens);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
|
}
|
||||||
|
|
||||||
|
Token* TokenFactory::opt(const Token* target) {
|
||||||
|
uptr<Token> p = std::make_unique<OptToken>(target);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
|
}
|
||||||
|
|
||||||
|
Token* TokenFactory::rep(const Token* target) {
|
||||||
|
uptr<Token> p = std::make_unique<RepToken>(target);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
|
}
|
||||||
|
|
||||||
|
Token* TokenFactory::not_(const Token* target) {
|
||||||
|
uptr<Token> p = std::make_unique<NotToken>(target);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
|
}
|
||||||
|
|
||||||
|
Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten) {
|
||||||
|
uptr<Token> p = std::make_unique<TagToken>(target, tagname, flatten);
|
||||||
|
auto t = p.get();
|
||||||
|
arena.emplace_back(std::move(p));
|
||||||
|
return t;
|
||||||
|
}
|
||||||
|
|
||||||
|
std::u32string TokenResult::flatMatch() const {
|
||||||
|
if (folded) return match;
|
||||||
|
|
||||||
|
std::u32string s = match;
|
||||||
|
for (const auto& c : child) s += c.flatMatch();
|
||||||
|
return s;
|
||||||
}
|
}
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
@@ -27,60 +96,56 @@ namespace spider {
|
|||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
LitToken::LitToken(std::string_view lit) {
|
LitToken::LitToken(std::string_view lit) {
|
||||||
if (!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
|
if (!unicode::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
|
||||||
}
|
}
|
||||||
|
|
||||||
LitToken::LitToken(const char* lit) : LitToken(std::string_view(lit)) {}
|
|
||||||
|
|
||||||
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
|
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
|
||||||
|
|
||||||
TokenResult LitToken::test(TextReader& ctx) const {
|
TokenResult LitToken::test(TextReader& ctx) const {
|
||||||
if(ctx.eat(literal)) return { true, literal };
|
if (ctx.eat(literal)) return { .success = true, .match = literal };
|
||||||
return { false, {} };
|
return { .success = false };
|
||||||
}
|
}
|
||||||
|
|
||||||
FnToken::FnToken(FnTokenFn chfn) : fn(chfn) {}
|
FnToken::FnToken(FnTokenFn chfn) : fn(std::move(chfn)) {}
|
||||||
|
|
||||||
TokenResult FnToken::test(TextReader& ctx) const {
|
TokenResult FnToken::test(TextReader& ctx) const {
|
||||||
std::u32string acc;
|
auto c = ctx.current();
|
||||||
fn()
|
if (c && fn(*c)) {
|
||||||
return { false, {} };
|
ctx.nextChar();
|
||||||
|
return { .success = true, .match = std::u32string(1, char32_t(*c)) };
|
||||||
|
}
|
||||||
|
return { .success = false };
|
||||||
}
|
}
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// SeqToken Implementation
|
// SeqToken Implementation
|
||||||
// ============================================================================
|
// ============================================================================30520370
|
||||||
|
|
||||||
SeqToken::SeqToken(const ilist<ref<const Token>>& list) : tokens(list) {}
|
SeqToken::SeqToken(const vector<const Token*>& _tokens) : tokens(_tokens) {}
|
||||||
|
|
||||||
TokenResult SeqToken::test(TextReader& ctx) const {
|
TokenResult SeqToken::test(TextReader& ctx) const {
|
||||||
// this is a common branch point
|
// this is a common branch point
|
||||||
std::u32string acc;
|
TokenResult r;
|
||||||
auto tri = ctx.push();
|
auto i = ctx.push();
|
||||||
|
|
||||||
// All matching steps within a sequence must pass consecutively.
|
// All matching steps within a sequence must pass consecutively.
|
||||||
for (const auto& token_ref : tokens) {
|
for (const auto& token_ref : tokens) {
|
||||||
TokenResult res = token_ref.get().test(ctx);
|
TokenResult res = token_ref->test(ctx);
|
||||||
|
|
||||||
if (!res.success) {
|
if (!res.success) {
|
||||||
// Strict ACID Transaction: Roll back context pointer entirely
|
// Strict ACID Transaction: Roll back context pointer entirely
|
||||||
// if any nested condition in the sequence fails.
|
// if any nested condition in the sequence fails.
|
||||||
ctx.pop(tri);
|
ctx.pop(i);
|
||||||
return { false, {} };
|
return { .success = false };
|
||||||
}
|
}
|
||||||
|
|
||||||
// Piecewise accumulation of individual matching sub-tokens
|
// Piecewise accumulation of individual matching sub-tokens
|
||||||
acc += res.match;
|
//r.match += res.match;
|
||||||
|
r.child.push_back(res);
|
||||||
}
|
}
|
||||||
|
|
||||||
return { true, acc };
|
r.success = true;
|
||||||
}
|
return r;
|
||||||
|
|
||||||
SeqToken SeqToken::operator&(const Token& tok) {
|
|
||||||
// Intrusive chaining optimization: Appends the next token directly into the existing
|
|
||||||
// registry vector instead of nesting structures, keeping the layout flattened.
|
|
||||||
tokens.push_back(std::cref(tok));
|
|
||||||
return *this;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -88,42 +153,35 @@ namespace spider {
|
|||||||
// OrToken Implementation
|
// OrToken Implementation
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
OrToken::OrToken(const ilist<ref<const Token>>& list) : tokens(list) {}
|
OrToken::OrToken(const vector<const Token*>& _tokens) : tokens(_tokens) {}
|
||||||
|
|
||||||
TokenResult OrToken::test(TextReader& ctx) const {
|
TokenResult OrToken::test(TextReader& ctx) const {
|
||||||
// this is a common branch point
|
|
||||||
auto tri = ctx.push();
|
|
||||||
|
|
||||||
// All matching steps within a sequence must pass consecutively.
|
// All matching steps within a sequence must pass consecutively.
|
||||||
|
auto i = ctx.push();
|
||||||
|
|
||||||
for (const auto& token_ref : tokens) {
|
for (const auto& token_ref : tokens) {
|
||||||
// Short-circuit branch: return immediately on first valid choice match
|
// Short-circuit branch: return immediately on first valid choice match
|
||||||
TokenResult res = token_ref.get().test(ctx);
|
TokenResult res = token_ref->test(ctx);
|
||||||
if (res.success) return res;
|
if (res.success) {
|
||||||
|
return res;
|
||||||
|
}
|
||||||
|
|
||||||
// Backtrack isolation: Reset the cursor position before testing the next alternative path
|
// Backtrack isolation: Reset the cursor position before testing the next alternative path
|
||||||
ctx.pop(tri);
|
ctx.pop(i);
|
||||||
}
|
}
|
||||||
|
|
||||||
return { false, {} };
|
return { .success = false };
|
||||||
}
|
}
|
||||||
|
|
||||||
OrToken OrToken::operator|(const Token& tok) {
|
|
||||||
// Intrusive grouping layout optimization:
|
|
||||||
// flattens alternative tokens at code evaluation time.
|
|
||||||
tokens.push_back(tok);
|
|
||||||
return *this;
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// OptToken Implementation
|
// OptToken Implementation
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
OptToken::OptToken(const Token& t) : target(t) {}
|
OptToken::OptToken(const Token* t) : target(t) {}
|
||||||
|
|
||||||
TokenResult OptToken::test(TextReader& ctx) const {
|
TokenResult OptToken::test(TextReader& ctx) const {
|
||||||
auto tri = ctx.push();
|
auto tri = ctx.push();
|
||||||
TokenResult res = target.test(ctx);
|
TokenResult res = target->test(ctx);
|
||||||
|
|
||||||
if (res.success) {
|
if (res.success) {
|
||||||
return res; // Option matched exactly 1 instance successfully
|
return res; // Option matched exactly 1 instance successfully
|
||||||
@@ -132,46 +190,74 @@ namespace spider {
|
|||||||
// Recovery path: If sub-rule fails, clean up the dirty state mutation
|
// Recovery path: If sub-rule fails, clean up the dirty state mutation
|
||||||
// and successfully return an empty match payload (0 instances).
|
// and successfully return an empty match payload (0 instances).
|
||||||
ctx.pop(tri);
|
ctx.pop(tri);
|
||||||
return { true, U"" };
|
return { .success = true };
|
||||||
}
|
}
|
||||||
|
|
||||||
OptToken OptToken::operator~() {
|
// ============================================================================
|
||||||
// Redundant layer trap protection: returning self
|
// NotToken Implementation
|
||||||
// prevents wrapping an Optional in an Optional
|
// ============================================================================
|
||||||
return *this;
|
|
||||||
}
|
|
||||||
|
|
||||||
|
NotToken::NotToken(const Token* t) : target(t) {}
|
||||||
|
|
||||||
|
TokenResult NotToken::test(TextReader& ctx) const {
|
||||||
|
auto i = ctx.push();
|
||||||
|
TokenResult res = target->test(ctx);
|
||||||
|
ctx.pop(i);
|
||||||
|
return { .success = !res.success };
|
||||||
|
}
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// RepToken Implementation
|
// RepToken Implementation
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
RepToken::RepToken(const Token& t) : target(t) {}
|
RepToken::RepToken(const Token* t) : target(t) {}
|
||||||
|
|
||||||
TokenResult RepToken::test(TextReader& ctx) const {
|
TokenResult RepToken::test(TextReader& ctx) const {
|
||||||
std::u32string acc;
|
TokenResult r;
|
||||||
auto tri = ctx.push();
|
|
||||||
|
for (;;) {
|
||||||
|
auto i = ctx.push();
|
||||||
|
TokenResult res = target->test(ctx);
|
||||||
|
|
||||||
for(;;) {
|
|
||||||
TokenResult res = target.test(ctx);
|
|
||||||
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
||||||
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
||||||
if(!res.success || tri == ctx.push()) {
|
if (!res.success || i.index == ctx.push().index) {
|
||||||
ctx.pop(tri);
|
ctx.pop(i);
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
acc += res.match;
|
|
||||||
|
//r.match += res.match;
|
||||||
|
r.child.push_back(res);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Repetition rules (* token) always evaluate to successful
|
// Repetition rules (* token) always evaluate to successful
|
||||||
// completion state, even with 0 matches.
|
// completion state, even with 0 matches.
|
||||||
return { true, acc };
|
r.success = true;
|
||||||
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
RepToken RepToken::operator*() {
|
// Tagged Token
|
||||||
// Redundant layer trap protection: returning self prevents wrapping
|
|
||||||
// a Repetition rule inside a Repetition rule
|
TagToken::TagToken(const Token* t, std::string_view tag, bool doflatten)
|
||||||
return *this;
|
: target(t), tag_name(tag), flatten(doflatten) {
|
||||||
|
}
|
||||||
|
|
||||||
|
TokenResult TagToken::test(TextReader& ctx) const {
|
||||||
|
auto r = target->test(ctx);
|
||||||
|
if (!r.success) return r;
|
||||||
|
|
||||||
|
// The innermost tag wins, so wrapping a rule that is already tagged, as in
|
||||||
|
// a choice of tagged rules, never hides what was actually matched.
|
||||||
|
if (!r.tag.has_value()) r.tag = tag_name;
|
||||||
|
|
||||||
|
if (flatten) {
|
||||||
|
// Fold the subtree text into this node, but keep the children so the
|
||||||
|
// parsed result stays a walkable tree instead of a flat string.
|
||||||
|
r.match = r.flatMatch();
|
||||||
|
r.folded = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
|
|
||||||
#include <spider/compiler/common.hpp>
|
#include <spider/compiler/common.hpp>
|
||||||
|
|
||||||
#include <spider/compiler/text/utf8.hpp>
|
#include <spider/compiler/text/unicode.hpp>
|
||||||
#include <spider/compiler/text/TextReader.hpp>
|
#include <spider/compiler/text/TextReader.hpp>
|
||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
@@ -13,21 +13,38 @@ namespace spider {
|
|||||||
struct TokenResult {
|
struct TokenResult {
|
||||||
|
|
||||||
/** @brief Indicates if the token composition successfully matched the input boundary. */
|
/** @brief Indicates if the token composition successfully matched the input boundary. */
|
||||||
bool success;
|
bool success = false;
|
||||||
|
|
||||||
|
optional<std::string_view> tag = {};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
|
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
|
||||||
* @note Returns empty when success is false.
|
* @note Returns empty when success is false.
|
||||||
*/
|
*/
|
||||||
std::u32string match;
|
std::u32string match = U"";
|
||||||
|
|
||||||
|
vector<TokenResult> child = {};
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Set when match already holds the folded text of the whole subtree.
|
||||||
|
* @details Tagged rules that flatten their children keep those children around
|
||||||
|
* so the parsed tree stays walkable, and flag the folded text here.
|
||||||
|
*/
|
||||||
|
bool folded = false;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief The full matched text of this node and of all of its children.
|
||||||
|
*/
|
||||||
|
std::u32string flatMatch() const;
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
// Forward declarations required by the abstract interface for operator returns.
|
class Token;
|
||||||
class SeqToken;
|
class TokenFactory;
|
||||||
class OrToken;
|
|
||||||
class OptToken;
|
using FnTokenFn = std::function<bool(u32)>;
|
||||||
class RepToken;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Pure virtual base class defining the EBNF combinator node contract.
|
* @brief Pure virtual base class defining the EBNF combinator node contract.
|
||||||
@@ -50,31 +67,50 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
virtual TokenResult test(TextReader& ctx) const = 0;
|
virtual TokenResult test(TextReader& ctx) const = 0;
|
||||||
|
|
||||||
|
};
|
||||||
|
|
||||||
|
class TokenFactory {
|
||||||
|
private:
|
||||||
|
// Arena owning all created tokens
|
||||||
|
std::vector<std::unique_ptr<Token>> arena;
|
||||||
|
|
||||||
|
// Deduplication caches
|
||||||
|
std::unordered_map<std::string, Token*> lit_cache;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
|
||||||
/**
|
TokenFactory() = default;
|
||||||
* @brief Chains this token and another sequentially.
|
|
||||||
* @return A temporary structural bridge matching both tokens sequentially.
|
|
||||||
*/
|
|
||||||
virtual SeqToken operator&(const Token& tok);
|
|
||||||
|
|
||||||
/**
|
// Prevent copying to maintain valid internal pointers
|
||||||
* @brief Combines this token and another under alternation.
|
TokenFactory(const TokenFactory&) = delete;
|
||||||
* @return A structural bridge matching either this token or the fallback selection.
|
|
||||||
*/
|
|
||||||
virtual OrToken operator|(const Token& tok);
|
|
||||||
|
|
||||||
/**
|
TokenFactory& operator=(const TokenFactory&) = delete;
|
||||||
* @brief Wraps this node in an optional layout rule.
|
|
||||||
* @return A structure matching zero or one instances of this current node.
|
|
||||||
*/
|
|
||||||
virtual OptToken operator~();
|
|
||||||
|
|
||||||
/**
|
public:
|
||||||
* @brief Wraps this node in a repetitive loop match framework.
|
|
||||||
* @return A structure matching zero or more occurrences of this current node.
|
// --- Primitive Constructors ---
|
||||||
*/
|
|
||||||
virtual RepToken operator*();
|
Token* lit(std::string_view text);
|
||||||
|
|
||||||
|
Token* operator[](std::string_view text);
|
||||||
|
|
||||||
|
Token* fn(FnTokenFn predicate);
|
||||||
|
|
||||||
|
// --- Combinator Constructors ---
|
||||||
|
|
||||||
|
Token* seq(const vector<const Token*>& tokens);
|
||||||
|
|
||||||
|
Token* choice(std::string_view opts);
|
||||||
|
|
||||||
|
Token* choice(const vector<const Token*>& tokens);
|
||||||
|
|
||||||
|
Token* opt(const Token* target);
|
||||||
|
|
||||||
|
Token* rep(const Token* target);
|
||||||
|
|
||||||
|
Token* not_(const Token* target);
|
||||||
|
|
||||||
|
Token* tag(const Token* target, std::string_view tagname, bool flatten = false);
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -93,8 +129,6 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
LitToken(std::string_view lit);
|
LitToken(std::string_view lit);
|
||||||
|
|
||||||
LitToken(const char* lit);
|
|
||||||
|
|
||||||
/** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */
|
/** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */
|
||||||
explicit LitToken(std::u32string lit);
|
explicit LitToken(std::u32string lit);
|
||||||
|
|
||||||
@@ -106,8 +140,6 @@ namespace spider {
|
|||||||
TokenResult test(TextReader& ctx) const override;
|
TokenResult test(TextReader& ctx) const override;
|
||||||
};
|
};
|
||||||
|
|
||||||
using FnTokenFn = std::function<bool(u32) >;
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Function based token
|
* @brief Function based token
|
||||||
*/
|
*/
|
||||||
@@ -135,11 +167,10 @@ namespace spider {
|
|||||||
* @brief Internal contiguous layout registry storing lightweight, zero-overhead references.
|
* @brief Internal contiguous layout registry storing lightweight, zero-overhead references.
|
||||||
* @details Avoids heap allocation penalties by referencing static instances immutably.
|
* @details Avoids heap allocation penalties by referencing static instances immutably.
|
||||||
*/
|
*/
|
||||||
vector<ref<const Token>> tokens;
|
vector<const Token*> tokens;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/** @brief Constructs a sequence container directly out of an inline brace-enclosed listing. */
|
SeqToken(const vector<const Token*>& tokens);
|
||||||
SeqToken(const ilist<ref<const Token>>& list);
|
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/**
|
/**
|
||||||
@@ -149,8 +180,6 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
TokenResult test(TextReader& ctx) const override;
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
/** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */
|
|
||||||
SeqToken operator&(const Token& tok) override;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -159,11 +188,10 @@ namespace spider {
|
|||||||
class OrToken : public Token {
|
class OrToken : public Token {
|
||||||
private:
|
private:
|
||||||
/** @brief Ordered registry of possible alternate structural paths. */
|
/** @brief Ordered registry of possible alternate structural paths. */
|
||||||
vector<ref<const Token>> tokens;
|
vector<const Token*> tokens;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/** @brief Constructs an alternation choice layout from brace-enclosed tokens. */
|
OrToken(const vector<const Token*>& tokens);
|
||||||
OrToken(const ilist<ref<const Token>>& list);
|
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/**
|
/**
|
||||||
@@ -172,8 +200,6 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
TokenResult test(TextReader& ctx) const override;
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
/** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */
|
|
||||||
OrToken operator|(const Token& tok) override;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -182,11 +208,11 @@ namespace spider {
|
|||||||
class OptToken : public Token {
|
class OptToken : public Token {
|
||||||
private:
|
private:
|
||||||
/** @brief Read-only target node reference to test optional status against. */
|
/** @brief Read-only target node reference to test optional status against. */
|
||||||
const Token& target;
|
const Token* target;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/** @brief Binds the target node rule structural dependency layout wrapper. */
|
/** @brief Binds the target node rule structural dependency layout wrapper. */
|
||||||
explicit OptToken(const Token& t);
|
explicit OptToken(const Token* t);
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/**
|
/**
|
||||||
@@ -195,8 +221,20 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
TokenResult test(TextReader& ctx) const override;
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
/** @brief Stub override providing standard compliance with the base Token interface signature. */
|
};
|
||||||
OptToken operator~() override;
|
|
||||||
|
/**
|
||||||
|
* @brief Negative lookahead combinator (NotToken).
|
||||||
|
*/
|
||||||
|
class NotToken : public Token {
|
||||||
|
private:
|
||||||
|
const Token* target;
|
||||||
|
|
||||||
|
public:
|
||||||
|
explicit NotToken(const Token* t);
|
||||||
|
|
||||||
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -205,11 +243,11 @@ namespace spider {
|
|||||||
class RepToken : public Token {
|
class RepToken : public Token {
|
||||||
private:
|
private:
|
||||||
/** @brief The base token node sequence layer evaluated in loops. */
|
/** @brief The base token node sequence layer evaluated in loops. */
|
||||||
const Token& target;
|
const Token* target;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/** @brief Binds the repeated structural blueprint node wrapper configuration. */
|
/** @brief Binds the repeated structural blueprint node wrapper configuration. */
|
||||||
explicit RepToken(const Token& t);
|
explicit RepToken(const Token* t);
|
||||||
|
|
||||||
public:
|
public:
|
||||||
/**
|
/**
|
||||||
@@ -219,8 +257,21 @@ namespace spider {
|
|||||||
*/
|
*/
|
||||||
TokenResult test(TextReader& ctx) const override;
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
/** @brief Stub override providing standard compliance with the base Token interface signature. */
|
};
|
||||||
RepToken operator*() override;
|
|
||||||
|
class TagToken : public Token {
|
||||||
|
private:
|
||||||
|
|
||||||
|
const Token* target;
|
||||||
|
std::string tag_name;
|
||||||
|
bool flatten;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
TagToken(const Token* t, std::string_view tag, bool doflatten);
|
||||||
|
|
||||||
|
TokenResult test(TextReader& ctx) const override;
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
|
|
||||||
namespace utf8 {
|
namespace unicode {
|
||||||
|
|
||||||
// --------------------- //
|
// --------------------- //
|
||||||
// UTF-8 Sequence Length //
|
// UTF-8 Sequence Length //
|
||||||
@@ -95,6 +95,55 @@ namespace spider {
|
|||||||
return _i == csize;
|
return _i == csize;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ----------------- //
|
||||||
|
// UTF-32 into UTF-8 //
|
||||||
|
// ----------------- //
|
||||||
|
|
||||||
|
inline void append_utf32_to_utf8(u32 code_point, std::string& out) {
|
||||||
|
if (code_point <= 0x7F) {
|
||||||
|
// 1-byte sequence (ASCII)
|
||||||
|
out.push_back(static_cast<char>(code_point));
|
||||||
|
} else if (code_point <= 0x7FF) {
|
||||||
|
// 2-byte sequence
|
||||||
|
out.push_back(static_cast<char>(0xC0 | ((code_point >> 6) & 0x1F)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
|
||||||
|
} else if (code_point <= 0xFFFF) {
|
||||||
|
// 3-byte sequence
|
||||||
|
// Filter out surrogate pairs (U+D800 to U+DFFF) as they are invalid Unicode scalar values
|
||||||
|
if (code_point >= 0xD800 && code_point <= 0xDFFF) {
|
||||||
|
code_point = 0xFFFD; // Replacement character
|
||||||
|
}
|
||||||
|
out.push_back(static_cast<char>(0xE0 | ((code_point >> 12) & 0x0F)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | ((code_point >> 6) & 0x3F)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
|
||||||
|
} else if (code_point <= 0x10FFFF) {
|
||||||
|
// 4-byte sequence
|
||||||
|
out.push_back(static_cast<char>(0xF0 | ((code_point >> 18) & 0x07)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | ((code_point >> 12) & 0x3F)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | ((code_point >> 6) & 0x3F)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
|
||||||
|
} else {
|
||||||
|
// Code point out of Unicode range -> insert UTF-8 replacement character U+FFFD
|
||||||
|
append_utf32_to_utf8(0xFFFD, out);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
inline std::string toUTF8(u32 cp) {
|
||||||
|
std::string s;
|
||||||
|
append_utf32_to_utf8(cp, s);
|
||||||
|
return s;
|
||||||
|
}
|
||||||
|
|
||||||
|
inline std::string toUTF8(const std::u32string& str) {
|
||||||
|
std::string s;
|
||||||
|
for(u32 ch : str) append_utf32_to_utf8(ch, s);
|
||||||
|
return s;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----------------- //
|
||||||
|
// STRINGS //
|
||||||
|
// ----------------- //
|
||||||
|
|
||||||
inline const char* getControlCharName(u8 c) {
|
inline const char* getControlCharName(u8 c) {
|
||||||
static const char* names[32] = {
|
static const char* names[32] = {
|
||||||
"NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL",
|
"NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL",
|
||||||
@@ -107,7 +156,7 @@ namespace spider {
|
|||||||
return nullptr;
|
return nullptr;
|
||||||
}
|
}
|
||||||
|
|
||||||
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {
|
inline void hexdump_utf8(const char* data, isize length, pos at, std::ostream& ostr) {
|
||||||
auto old_flags = ostr.flags();
|
auto old_flags = ostr.flags();
|
||||||
auto old_fill = ostr.fill();
|
auto old_fill = ostr.fill();
|
||||||
|
|
||||||
Reference in New Issue
Block a user