ParseTree
This commit is contained in:
@@ -1,14 +1,56 @@
|
||||
#include <iostream>
|
||||
|
||||
#include <spider/compiler/Compiler.hpp>
|
||||
#include <spider/compiler/common.hpp>
|
||||
|
||||
#include <spider/compiler/text/unicode.hpp>
|
||||
#include <spider/compiler/text/TextReader.hpp>
|
||||
|
||||
#include <spider/compiler/assembler/AsmEBNF.hpp>
|
||||
#include <spider/compiler/text/ParseTree.hpp>
|
||||
|
||||
using namespace spider;
|
||||
|
||||
// ============================================================================
|
||||
// Compiler Entry Points
|
||||
// ============================================================================
|
||||
|
||||
namespace spider {
|
||||
|
||||
TokenFactory& assemblyGrammar() {
|
||||
static TokenFactory grammar;
|
||||
static bool loaded = false;
|
||||
if (!loaded) {
|
||||
asm_ebnf::initTokens(grammar);
|
||||
loaded = true;
|
||||
}
|
||||
return grammar;
|
||||
}
|
||||
|
||||
bool compileProgram(const std::string& source, ParseTree& out) {
|
||||
return out.parse(asm_ebnf::program, source);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Test Bookkeeping
|
||||
// ============================================================================
|
||||
|
||||
// Suite wide bookkeeping, so a failing case can never hide behind a clean exit code.
|
||||
static isize total_tests = 0;
|
||||
static isize failed_tests = 0;
|
||||
|
||||
static void report(bool ok, const std::string& name) {
|
||||
++total_tests;
|
||||
if (ok) {
|
||||
std::cout << "[PASS] " << name << "\n";
|
||||
} else {
|
||||
++failed_tests;
|
||||
std::cerr << "[FAIL] " << name << "\n";
|
||||
}
|
||||
}
|
||||
|
||||
// Inline evaluator that executes the reader and prints standard output
|
||||
static bool run_test_case(const Token* token, const std::string& input, bool expectedSuccess, const std::u32string& expectedMatch = U"") {
|
||||
StringTextReader reader(input);
|
||||
@@ -17,6 +59,7 @@ static bool run_test_case(const Token* token, const std::string& input, bool exp
|
||||
bool status_ok = (res.success == expectedSuccess);
|
||||
bool match_ok = (!expectedSuccess) || expectedMatch.empty() || (res.flatMatch() == expectedMatch);
|
||||
|
||||
++total_tests;
|
||||
if (status_ok && match_ok) {
|
||||
std::cout << "[PASS] Input: \"" << input << "\" -> "
|
||||
<< (res.success ? "SUCCESS" : "FAILURE")
|
||||
@@ -30,6 +73,7 @@ static bool run_test_case(const Token* token, const std::string& input, bool exp
|
||||
if (expectedSuccess && !expectedMatch.empty()) {
|
||||
std::cerr << " Expected Match: \"" << unicode::toUTF8(expectedMatch) << "\", Got: \"" << unicode::toUTF8(res.flatMatch()) << "\"\n";
|
||||
}
|
||||
++failed_tests;
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -137,14 +181,85 @@ void test_full_program(TokenFactory& tf) {
|
||||
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
|
||||
" JMP start\n";
|
||||
|
||||
StringTextReader reader(asm_code);
|
||||
TokenResult res = asm_ebnf::program->test(reader);
|
||||
ParseTree tree;
|
||||
report(compileProgram(asm_code, tree), "full assembly program parses completely");
|
||||
}
|
||||
|
||||
if (res.success) {
|
||||
std::cout << "[PASS] Full Assembly Program parsed successfully!\n";
|
||||
} else {
|
||||
std::cerr << "[FAIL] Program parsing failed.\n";
|
||||
void test_parse_tree(TokenFactory& tf) {
|
||||
std::cout << "\n--- Testing Parse Tree Inspection ---\n";
|
||||
const std::string asm_code =
|
||||
"#include stdio\n"
|
||||
"\n"
|
||||
"start:\n"
|
||||
" MOV R1, 0x20 ; Load constant\n"
|
||||
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
|
||||
" JMP start\n";
|
||||
|
||||
ParseTree tree;
|
||||
report(compileProgram(asm_code, tree), "program becomes a tree of nodes");
|
||||
|
||||
ParseNode* root = tree.root();
|
||||
report(root != nullptr, "tree exposes a root node");
|
||||
if (root == nullptr) return;
|
||||
|
||||
report(root->hasTag("program"), "root node is tagged as program");
|
||||
report(root->textUtf8() == asm_code, "root text covers the whole source");
|
||||
report(root->childCount() > 0, "root holds child nodes");
|
||||
|
||||
vector<ParseNode*> lines = root->findAll("line");
|
||||
report(lines.size() == 6, "program holds one node per line");
|
||||
|
||||
if (lines.size() == 6) {
|
||||
report(lines.front()->parentNode() == root, "a line knows its parent");
|
||||
report(lines.front()->depth() == 1, "lines sit one level under the root");
|
||||
report(lines.front()->nextSibling() == lines[1], "sibling walk reaches the next line");
|
||||
report(lines[1]->previousSibling() == lines.front(), "sibling walk goes back");
|
||||
report(lines.back()->nextSibling() == nullptr, "the last line has no next sibling");
|
||||
|
||||
ParseNode* preprocessor = lines.front()->firstChild("preprocessor");
|
||||
report(preprocessor != nullptr, "first line holds a preprocessor child");
|
||||
report(preprocessor != nullptr && preprocessor->textUtf8() == "#include stdio", "preprocessor text is readable");
|
||||
|
||||
report(lines[2]->firstChild("line_label") != nullptr, "label line is recognized");
|
||||
ParseNode* label = lines[2]->find("label");
|
||||
report(label != nullptr && label->textUtf8() == "start:", "label text is readable");
|
||||
|
||||
ParseNode* mov = lines[3]->find("instruction");
|
||||
report(mov != nullptr, "instruction is found inside its line");
|
||||
if (mov != nullptr) {
|
||||
report(mov->firstChild("opcode") != nullptr, "instruction exposes its opcode child");
|
||||
|
||||
ParseNode* operands = mov->firstChild("operand_list");
|
||||
report(operands != nullptr && operands->childCount() == 2, "operand list exposes both operands");
|
||||
|
||||
ParseNode* hexlit = mov->find("hex_lit");
|
||||
report(hexlit != nullptr && hexlit->textUtf8() == "0x20", "hex literal text is readable");
|
||||
}
|
||||
|
||||
ParseNode* comment = lines[3]->firstChild("comment");
|
||||
report(comment != nullptr && comment->textUtf8() == "; Load constant", "trailing comment is captured");
|
||||
|
||||
report(lines[4]->firstChild("line_annotation") != nullptr, "annotation line is recognized");
|
||||
ParseNode* displacement = lines[4]->find("addrm_dis");
|
||||
report(displacement != nullptr && displacement->textUtf8() == "[R2 + R3 * 8 + 4]", "displacement operand is readable");
|
||||
report(tree.count("register") == 4, "every register of the program is reachable");
|
||||
}
|
||||
|
||||
ParseTree single;
|
||||
report(single.parse(asm_ebnf::instruction, "MOV R1, [R2 + 4]"), "a single instruction parses on its own");
|
||||
|
||||
ParseNode* instruction = single.root();
|
||||
report(instruction != nullptr && instruction->hasTag("instruction"), "instruction root is tagged");
|
||||
if (instruction != nullptr) {
|
||||
ParseNode* operands = instruction->firstChild("operand_list");
|
||||
report(operands != nullptr && operands->childCount() == 2, "operand list holds two operands");
|
||||
|
||||
ParseNode* second = operands == nullptr ? nullptr : operands->childAt(1);
|
||||
report(second != nullptr && second->textUtf8() == "[R2 + 4]", "operand text is readable");
|
||||
report(second != nullptr && second->ancestor("instruction") == instruction, "an operand can climb back to its instruction");
|
||||
}
|
||||
|
||||
std::cout << "\n--- Parsed Tree ---\n" << tree.toString() << "\n";
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
@@ -153,7 +268,7 @@ void test_full_program(TokenFactory& tf) {
|
||||
|
||||
int main() {
|
||||
TokenFactory tf;
|
||||
asm_ebnf::initTokens(tf);
|
||||
assemblyGrammar();
|
||||
|
||||
std::cout << "Running Token Framework Tests...\n";
|
||||
|
||||
@@ -164,6 +279,16 @@ int main() {
|
||||
test_addressing_modes(tf);
|
||||
test_instructions_and_lines(tf);
|
||||
test_full_program(tf);
|
||||
test_parse_tree(tf);
|
||||
|
||||
std::cout << "\n========================================\n";
|
||||
std::cout << "Test Results: " << (total_tests - failed_tests) << "/" << total_tests << " passed.\n";
|
||||
std::cout << "========================================\n";
|
||||
|
||||
if (failed_tests > 0) {
|
||||
std::cerr << failed_tests << " test(s) FAILED.\n";
|
||||
return 1;
|
||||
}
|
||||
|
||||
std::cout << "All token framework tests passed successfully!\n";
|
||||
return 0;
|
||||
|
||||
@@ -1,10 +1,27 @@
|
||||
#pragma
|
||||
#pragma once
|
||||
|
||||
#include <spider/compiler/common.hpp>
|
||||
|
||||
#include <spider/compiler/text/Token.hpp>
|
||||
#include <spider/compiler/text/ParseTree.hpp>
|
||||
|
||||
namespace spider {
|
||||
|
||||
class Token;
|
||||
class RootToken;
|
||||
/**
|
||||
* @brief The token factory that owns the assembly grammar, loaded on first use.
|
||||
* @details Every rule of the assembly language lives in this factory, so a caller
|
||||
* can test a single rule or a whole program against the same grammar.
|
||||
*/
|
||||
TokenFactory& assemblyGrammar();
|
||||
|
||||
/**
|
||||
* @brief Parses a whole assembly program into an inspectable tree of nodes.
|
||||
* @details The grammar has to consume the entire source, so a malformed program
|
||||
* is rejected instead of quietly producing a partial tree. On success the
|
||||
* resulting tree can be walked like a document: program, line, label,
|
||||
* instruction, operands, literals and comments.
|
||||
* @returns True on success, with the parsed nodes left inside out.
|
||||
*/
|
||||
bool compileProgram(const std::string& source, ParseTree& out);
|
||||
|
||||
}
|
||||
|
||||
@@ -106,17 +106,17 @@ namespace spider::asm_ebnf {
|
||||
binary_digit = tf.choice("01");
|
||||
|
||||
ws_char = tf.fn(isWhithespaceCharNotCrLf);
|
||||
ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true);
|
||||
whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true);
|
||||
newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
|
||||
ws_optional = tf.rep(ws_char);
|
||||
whitespace = tf.seq({ ws_char, tf.rep(ws_char) });
|
||||
newline = tf.tag(tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }), "newline", true);
|
||||
utf8_char = tf.fn(isUTF8CharNotCrLf);
|
||||
|
||||
char_escape = tf.seq({ tf["\\"], utf8_char });
|
||||
char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
|
||||
char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
|
||||
char_lit = tf.tag(tf.seq({ tf["'"], char_content, tf["'"] }), "char_lit", true);
|
||||
|
||||
string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
|
||||
string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
|
||||
string_lit = tf.tag(tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }), "string_lit", true);
|
||||
|
||||
// (* Literals *)
|
||||
identifier = tf.tag(tf.seq({
|
||||
@@ -185,29 +185,29 @@ namespace spider::asm_ebnf {
|
||||
// (* Generalized Instructions *)
|
||||
|
||||
opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
|
||||
operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
|
||||
instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction");
|
||||
operand_list = tf.tag(tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }), "operand_list", true);
|
||||
instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction", true);
|
||||
|
||||
// (* Added Preprocessor, Annotation *)
|
||||
|
||||
annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named");
|
||||
annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg");
|
||||
annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
|
||||
annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
|
||||
annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation");
|
||||
annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named", true);
|
||||
annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg", true);
|
||||
annotation_args = tf.tag(tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }), "annotation_args", true);
|
||||
annotation_pars = tf.tag(tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }), "annotation_pars", true);
|
||||
annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation", true);
|
||||
|
||||
preprocessor_val = tf.choice({ identifier, literal_decl });
|
||||
preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor");
|
||||
preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor", true);
|
||||
|
||||
// (* Line Structure & Program *)
|
||||
|
||||
label = tf.tag(tf.seq({ identifier, tf[":"] }), "label");
|
||||
line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||
line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||
line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
|
||||
line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
|
||||
line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
|
||||
program = tf.seq({ tf.rep(line), tf.opt(line_last) });
|
||||
label = tf.tag(tf.seq({ identifier, tf[":"] }), "label", true);
|
||||
line_label = tf.tag(tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }), "line_label", true);
|
||||
line_annotation = tf.tag(tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }), "line_annotation", true);
|
||||
line_content = tf.tag(tf.choice({ preprocessor, line_annotation, line_label, instruction }), "line_content", true);
|
||||
line = tf.tag(tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }), "line", true);
|
||||
line_last = tf.tag(tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }), "line_last", true);
|
||||
program = tf.tag(tf.seq({ tf.rep(line), tf.opt(line_last) }), "program", true);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -0,0 +1,334 @@
|
||||
#include "ParseTree.hpp"
|
||||
|
||||
namespace spider {
|
||||
|
||||
// ============================================================================
|
||||
// Internal Helpers
|
||||
// ============================================================================
|
||||
|
||||
/**
|
||||
* @brief Renders control characters as escapes, so a node can be printed on one line.
|
||||
*/
|
||||
static std::string escapeText(std::string_view raw) {
|
||||
std::string out;
|
||||
out.reserve(raw.size());
|
||||
for (char c : raw) {
|
||||
switch (c) {
|
||||
case '\n': out += "\\n"; break;
|
||||
case '\r': out += "\\r"; break;
|
||||
case '\t': out += "\\t"; break;
|
||||
default: out += c; break;
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// ParseNode Navigation
|
||||
// ============================================================================
|
||||
|
||||
isize ParseNode::depth() const {
|
||||
isize levels = 0;
|
||||
for (const ParseNode* n = parentNode(); n != nullptr; n = n->parentNode()) ++levels;
|
||||
return levels;
|
||||
}
|
||||
|
||||
const ParseNode* ParseNode::parentNode() const {
|
||||
if (owner == nullptr || parent == ParseNode::npos) return nullptr;
|
||||
return owner->resolve(parent);
|
||||
}
|
||||
|
||||
const ParseNode* ParseNode::firstChild() const {
|
||||
if (owner == nullptr || first_child == ParseNode::npos) return nullptr;
|
||||
return owner->resolve(first_child);
|
||||
}
|
||||
|
||||
const ParseNode* ParseNode::lastChild() const {
|
||||
if (owner == nullptr || last_child == ParseNode::npos) return nullptr;
|
||||
return owner->resolve(last_child);
|
||||
}
|
||||
|
||||
const ParseNode* ParseNode::nextSibling() const {
|
||||
if (owner == nullptr || next_sibling == ParseNode::npos) return nullptr;
|
||||
return owner->resolve(next_sibling);
|
||||
}
|
||||
|
||||
const ParseNode* ParseNode::previousSibling() const {
|
||||
if (owner == nullptr || prev_sibling == ParseNode::npos) return nullptr;
|
||||
return owner->resolve(prev_sibling);
|
||||
}
|
||||
|
||||
const ParseNode* ParseNode::childAt(isize index) const {
|
||||
isize seen = 0;
|
||||
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||
if (seen == index) return c;
|
||||
++seen;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ParseNode* ParseNode::parentNode() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->parentNode()); }
|
||||
ParseNode* ParseNode::firstChild() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->firstChild()); }
|
||||
ParseNode* ParseNode::lastChild() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->lastChild()); }
|
||||
ParseNode* ParseNode::nextSibling() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->nextSibling()); }
|
||||
ParseNode* ParseNode::previousSibling() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->previousSibling()); }
|
||||
ParseNode* ParseNode::childAt(isize index) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->childAt(index)); }
|
||||
|
||||
// ============================================================================
|
||||
// ParseNode Navigation Filtered By Tag
|
||||
// ============================================================================
|
||||
|
||||
const ParseNode* ParseNode::firstChild(std::string_view name) const {
|
||||
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||
if (c->hasTag(name)) return c;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const ParseNode* ParseNode::lastChild(std::string_view name) const {
|
||||
const ParseNode* found = nullptr;
|
||||
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||
if (c->hasTag(name)) found = c;
|
||||
}
|
||||
return found;
|
||||
}
|
||||
|
||||
const ParseNode* ParseNode::nextSibling(std::string_view name) const {
|
||||
for (const ParseNode* s = nextSibling(); s != nullptr; s = s->nextSibling()) {
|
||||
if (s->hasTag(name)) return s;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const ParseNode* ParseNode::ancestor(std::string_view name) const {
|
||||
for (const ParseNode* p = parentNode(); p != nullptr; p = p->parentNode()) {
|
||||
if (p->hasTag(name)) return p;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ParseNode* ParseNode::firstChild(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->firstChild(name)); }
|
||||
ParseNode* ParseNode::lastChild(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->lastChild(name)); }
|
||||
ParseNode* ParseNode::nextSibling(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->nextSibling(name)); }
|
||||
ParseNode* ParseNode::ancestor(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->ancestor(name)); }
|
||||
|
||||
// ============================================================================
|
||||
// ParseNode Queries
|
||||
// ============================================================================
|
||||
|
||||
const ParseNode* ParseNode::find(std::string_view name) const {
|
||||
if (hasTag(name)) return this;
|
||||
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||
if (const ParseNode* hit = c->find(name)) return hit;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
vector<const ParseNode*> ParseNode::findAll(std::string_view name) const {
|
||||
vector<const ParseNode*> hits;
|
||||
if (hasTag(name)) hits.push_back(this);
|
||||
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||
for (const ParseNode* hit : c->findAll(name)) hits.push_back(hit);
|
||||
}
|
||||
return hits;
|
||||
}
|
||||
|
||||
isize ParseNode::count(std::string_view name) const { return findAll(name).size(); }
|
||||
|
||||
ParseNode* ParseNode::find(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->find(name)); }
|
||||
|
||||
vector<ParseNode*> ParseNode::findAll(std::string_view name) {
|
||||
vector<const ParseNode*> hits = static_cast<const ParseNode*>(this)->findAll(name);
|
||||
vector<ParseNode*> out;
|
||||
out.reserve(hits.size());
|
||||
for (const ParseNode* hit : hits) out.push_back(const_cast<ParseNode*>(hit));
|
||||
return out;
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// ParseNode Rendering
|
||||
// ============================================================================
|
||||
|
||||
std::string ParseNode::describe(isize depth) const {
|
||||
const std::string pad(depth * 2, ' ');
|
||||
|
||||
if (!tag.has_value()) {
|
||||
if (isLeaf()) return pad + escapeText(ownTextUtf8());
|
||||
|
||||
std::string out;
|
||||
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||
out += c->describe(depth) + "\n";
|
||||
}
|
||||
if (!out.empty()) out.pop_back();
|
||||
return out;
|
||||
}
|
||||
|
||||
const std::string name(*tag);
|
||||
if (isLeaf()) return pad + "<" + name + ">" + escapeText(ownTextUtf8()) + "</" + name + ">";
|
||||
|
||||
std::string out = pad + "<" + name + ">";
|
||||
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
|
||||
out += "\n" + c->describe(depth + 1);
|
||||
}
|
||||
out += "\n" + pad + "</" + name + ">";
|
||||
return out;
|
||||
}
|
||||
|
||||
std::string ParseNode::toString() const { return describe(0); }
|
||||
std::string ParseNode::toString(isize depth) const { return describe(depth); }
|
||||
|
||||
// ============================================================================
|
||||
// ParseTree Building
|
||||
// ============================================================================
|
||||
|
||||
ParseNode* ParseTree::resolve(isize index) {
|
||||
if (index == ParseNode::npos) return nullptr;
|
||||
return &nodes[index];
|
||||
}
|
||||
|
||||
const ParseNode* ParseTree::resolve(isize index) const {
|
||||
if (index == ParseNode::npos) return nullptr;
|
||||
return &nodes[index];
|
||||
}
|
||||
|
||||
isize ParseTree::addNode(const TokenResult& result) {
|
||||
ParseNode node;
|
||||
node.owner = this;
|
||||
node.tag = result.tag;
|
||||
node.own = result.match;
|
||||
node.text = result.flatMatch();
|
||||
|
||||
isize self = nodes.size();
|
||||
nodes.push_back(std::move(node));
|
||||
return self;
|
||||
}
|
||||
|
||||
void ParseTree::linkChildren(isize parent_index, const vector<isize>& children) {
|
||||
ParseNode& parent = nodes[parent_index];
|
||||
parent.first_child = ParseNode::npos;
|
||||
parent.last_child = ParseNode::npos;
|
||||
parent.children_count = 0;
|
||||
|
||||
for (isize child : children) {
|
||||
ParseNode& kid = nodes[child];
|
||||
kid.parent = parent_index;
|
||||
kid.prev_sibling = parent.last_child;
|
||||
kid.next_sibling = ParseNode::npos;
|
||||
|
||||
if (parent.last_child != ParseNode::npos) {
|
||||
nodes[parent.last_child].next_sibling = child;
|
||||
} else {
|
||||
parent.first_child = child;
|
||||
}
|
||||
parent.last_child = child;
|
||||
parent.children_count++;
|
||||
}
|
||||
}
|
||||
|
||||
vector<isize> ParseTree::collapse(const TokenResult& result) {
|
||||
vector<isize> kids;
|
||||
for (const TokenResult& sub : result.child) {
|
||||
for (isize kid : collapse(sub)) kids.push_back(kid);
|
||||
}
|
||||
|
||||
// Untagged rules are grammar scaffolding, never part of the exposed syntax.
|
||||
if (!result.tag.has_value()) return kids;
|
||||
|
||||
const std::u32string folded = result.flatMatch();
|
||||
isize self = addNode(result);
|
||||
|
||||
// An empty shell has nothing to show.
|
||||
if (kids.empty() && folded.empty()) return {};
|
||||
|
||||
linkChildren(self, kids);
|
||||
return { self };
|
||||
}
|
||||
|
||||
void ParseTree::build(const TokenResult& result) {
|
||||
nodes.clear();
|
||||
root_index = ParseNode::npos;
|
||||
if (!result.success) return;
|
||||
|
||||
vector<isize> tops = collapse(result);
|
||||
if (tops.size() == 1) {
|
||||
root_index = tops.front();
|
||||
return;
|
||||
}
|
||||
|
||||
// A grammar that exposes several top level rules still gets one container.
|
||||
ParseNode root;
|
||||
root.owner = this;
|
||||
root_index = nodes.size();
|
||||
nodes.push_back(std::move(root));
|
||||
linkChildren(root_index, tops);
|
||||
}
|
||||
|
||||
bool ParseTree::parsePrefix(const Token* token, TextReader& reader) {
|
||||
nodes.clear();
|
||||
root_index = ParseNode::npos;
|
||||
if (token == nullptr) return false;
|
||||
|
||||
TokenResult result = token->test(reader);
|
||||
if (!result.success) return false;
|
||||
|
||||
build(result);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ParseTree::parse(const Token* token, TextReader& reader) {
|
||||
nodes.clear();
|
||||
root_index = ParseNode::npos;
|
||||
if (token == nullptr) return false;
|
||||
|
||||
TokenResult result = token->test(reader);
|
||||
if (!result.success) return false;
|
||||
|
||||
// A complete parse leaves nothing behind in the reader.
|
||||
if (reader.hasError()) return false;
|
||||
if (reader.current().has_value()) return false;
|
||||
|
||||
build(result);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool ParseTree::parse(const Token* token, const std::string& source) {
|
||||
StringTextReader reader(source);
|
||||
return parse(token, reader);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// ParseTree Inspection
|
||||
// ============================================================================
|
||||
|
||||
const ParseNode* ParseTree::find(std::string_view name) const {
|
||||
const ParseNode* r = root();
|
||||
return r == nullptr ? nullptr : r->find(name);
|
||||
}
|
||||
|
||||
vector<const ParseNode*> ParseTree::findAll(std::string_view name) const {
|
||||
const ParseNode* r = root();
|
||||
if (r == nullptr) return {};
|
||||
return r->findAll(name);
|
||||
}
|
||||
|
||||
isize ParseTree::count(std::string_view name) const {
|
||||
const ParseNode* r = root();
|
||||
return r == nullptr ? 0 : r->count(name);
|
||||
}
|
||||
|
||||
std::string ParseTree::toString() const {
|
||||
const ParseNode* r = root();
|
||||
return r == nullptr ? std::string() : r->toString();
|
||||
}
|
||||
|
||||
ParseNode* ParseTree::find(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseTree*>(this)->find(name)); }
|
||||
|
||||
vector<ParseNode*> ParseTree::findAll(std::string_view name) {
|
||||
vector<const ParseNode*> hits = static_cast<const ParseTree*>(this)->findAll(name);
|
||||
vector<ParseNode*> out;
|
||||
out.reserve(hits.size());
|
||||
for (const ParseNode* hit : hits) out.push_back(const_cast<ParseNode*>(hit));
|
||||
return out;
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,295 @@
|
||||
#pragma once
|
||||
|
||||
#include <spider/compiler/common.hpp>
|
||||
|
||||
#include <spider/compiler/text/unicode.hpp>
|
||||
#include <spider/compiler/text/Token.hpp>
|
||||
|
||||
namespace spider {
|
||||
|
||||
class ParseTree;
|
||||
|
||||
/**
|
||||
* @brief DOM style handle over a single node of a parsed token tree.
|
||||
* @details Every node knows its parent, its children and its siblings, so a
|
||||
* parsed program can be walked and inspected much like a document.
|
||||
* Nodes are owned by the ParseTree that produced them and stay
|
||||
* valid while that tree is alive. The tree itself is built by
|
||||
* ParseTree, never by hand.
|
||||
*/
|
||||
class ParseNode {
|
||||
friend class ParseTree;
|
||||
|
||||
public:
|
||||
|
||||
/** @brief Index value used for "no node" links, since isize is unsigned. */
|
||||
static constexpr isize npos = static_cast<isize>(-1);
|
||||
|
||||
private:
|
||||
|
||||
ParseTree* owner = nullptr;
|
||||
|
||||
isize parent = npos;
|
||||
isize first_child = npos;
|
||||
isize last_child = npos;
|
||||
isize next_sibling = npos;
|
||||
isize prev_sibling = npos;
|
||||
isize children_count = 0;
|
||||
|
||||
optional<std::string_view> tag = {};
|
||||
std::u32string own = U"";
|
||||
std::u32string text = U"";
|
||||
|
||||
public:
|
||||
|
||||
ParseNode() = default;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
* @brief The grammar tag of this node, or nothing if untagged.
|
||||
*/
|
||||
const optional<std::string_view>& tagName() const { return tag; }
|
||||
|
||||
/**
|
||||
* @brief Checks if this node carries the given tag.
|
||||
*/
|
||||
bool hasTag(std::string_view name) const { return tag.has_value() && *tag == name; }
|
||||
|
||||
/**
|
||||
* @brief The full source text covered by this node and all of its children.
|
||||
*/
|
||||
const std::u32string& fullText() const { return text; }
|
||||
|
||||
/**
|
||||
* @brief The source text held by this node alone, without its children.
|
||||
* @note For tagged nodes that fold their children, this is the full text.
|
||||
*/
|
||||
const std::u32string& ownText() const { return own; }
|
||||
|
||||
/**
|
||||
* @brief UTF-8 rendering of text().
|
||||
*/
|
||||
std::string textUtf8() const { return unicode::toUTF8(text); }
|
||||
|
||||
/**
|
||||
* @brief UTF-8 rendering of ownText().
|
||||
*/
|
||||
std::string ownTextUtf8() const { return unicode::toUTF8(own); }
|
||||
|
||||
/**
|
||||
* @brief Amount of direct children of this node.
|
||||
*/
|
||||
isize childCount() const { return children_count; }
|
||||
|
||||
/**
|
||||
* @brief True when this node holds no text and no children.
|
||||
*/
|
||||
bool isLeaf() const { return children_count == 0; }
|
||||
|
||||
/**
|
||||
* @brief Distance from the root of the tree, zero for the root itself.
|
||||
*/
|
||||
isize depth() const;
|
||||
|
||||
/**
|
||||
* @brief Storage index of this node inside its parent, npos for the root.
|
||||
*/
|
||||
isize indexInParent() const { return parent; }
|
||||
|
||||
public:
|
||||
|
||||
// ---------------------------------------------------------------- //
|
||||
// Navigation //
|
||||
// ---------------------------------------------------------------- //
|
||||
|
||||
ParseNode* parentNode();
|
||||
const ParseNode* parentNode() const;
|
||||
|
||||
ParseNode* firstChild();
|
||||
const ParseNode* firstChild() const;
|
||||
|
||||
ParseNode* lastChild();
|
||||
const ParseNode* lastChild() const;
|
||||
|
||||
ParseNode* nextSibling();
|
||||
const ParseNode* nextSibling() const;
|
||||
|
||||
ParseNode* previousSibling();
|
||||
const ParseNode* previousSibling() const;
|
||||
|
||||
/**
|
||||
* @brief The index-th direct child of this node, nullptr when out of range.
|
||||
*/
|
||||
ParseNode* childAt(isize index);
|
||||
const ParseNode* childAt(isize index) const;
|
||||
|
||||
public:
|
||||
|
||||
// ---------------------------------------------------------------- //
|
||||
// Navigation filtered by tag //
|
||||
// ---------------------------------------------------------------- //
|
||||
|
||||
/**
|
||||
* @brief First direct child carrying the given tag.
|
||||
*/
|
||||
ParseNode* firstChild(std::string_view name);
|
||||
const ParseNode* firstChild(std::string_view name) const;
|
||||
|
||||
/**
|
||||
* @brief Last direct child carrying the given tag.
|
||||
*/
|
||||
ParseNode* lastChild(std::string_view name);
|
||||
const ParseNode* lastChild(std::string_view name) const;
|
||||
|
||||
/**
|
||||
* @brief Next sibling of this node carrying the given tag.
|
||||
*/
|
||||
ParseNode* nextSibling(std::string_view name);
|
||||
const ParseNode* nextSibling(std::string_view name) const;
|
||||
|
||||
/**
|
||||
* @brief Nearest ancestor carrying the given tag, nullptr when there is none.
|
||||
*/
|
||||
ParseNode* ancestor(std::string_view name);
|
||||
const ParseNode* ancestor(std::string_view name) const;
|
||||
|
||||
public:
|
||||
|
||||
// ---------------------------------------------------------------- //
|
||||
// Queries //
|
||||
// ---------------------------------------------------------------- //
|
||||
|
||||
/**
|
||||
* @brief First node in document order carrying the given tag, this node included.
|
||||
*/
|
||||
ParseNode* find(std::string_view name);
|
||||
const ParseNode* find(std::string_view name) const;
|
||||
|
||||
/**
|
||||
* @brief Every node in document order carrying the given tag, this node included.
|
||||
*/
|
||||
vector<ParseNode*> findAll(std::string_view name);
|
||||
vector<const ParseNode*> findAll(std::string_view name) const;
|
||||
|
||||
/**
|
||||
* @brief Amount of nodes in document order carrying the given tag.
|
||||
*/
|
||||
isize count(std::string_view name) const;
|
||||
|
||||
/**
|
||||
* @brief True when at least one node carries the given tag.
|
||||
*/
|
||||
bool contains(std::string_view name) const { return find(name) != nullptr; }
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
* @brief Human readable XML-like rendering of this node and its children.
|
||||
*/
|
||||
std::string toString() const;
|
||||
|
||||
/**
|
||||
* @brief XML-like rendering of this node alone, indented by the given depth.
|
||||
*/
|
||||
std::string toString(isize depth) const;
|
||||
|
||||
private:
|
||||
|
||||
std::string describe(isize depth) const;
|
||||
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Owner of a parsed token tree, exposing a document like interface.
|
||||
* @details Use parse() to turn source text into an inspectable tree. The tree
|
||||
* is materialized once and then only read, so any number of consumers
|
||||
* can walk the same nodes safely.
|
||||
*/
|
||||
class ParseTree {
|
||||
friend class ParseNode;
|
||||
|
||||
private:
|
||||
|
||||
deque<ParseNode> nodes;
|
||||
isize root_index = ParseNode::npos;
|
||||
|
||||
private:
|
||||
|
||||
ParseNode* resolve(isize index);
|
||||
const ParseNode* resolve(isize index) const;
|
||||
|
||||
isize addNode(const TokenResult& result);
|
||||
|
||||
void linkChildren(isize parent, const vector<isize>& children);
|
||||
|
||||
/**
|
||||
* @brief Materializes a parse result, dropping the grammar scaffolding.
|
||||
* @details Only tagged rules become nodes, so a consumer walks syntax and not
|
||||
* combinators: untagged rules are pure plumbing and simply hoist their
|
||||
* children upwards, and empty shells are discarded. Returns the top
|
||||
* level nodes produced by this subtree.
|
||||
*/
|
||||
vector<isize> collapse(const TokenResult& result);
|
||||
|
||||
public:
|
||||
|
||||
ParseTree() = default;
|
||||
~ParseTree() = default;
|
||||
|
||||
// Nodes point back at their owning tree, so trees are never copied or moved.
|
||||
ParseTree(const ParseTree&) = delete;
|
||||
ParseTree& operator=(const ParseTree&) = delete;
|
||||
ParseTree(ParseTree&&) = delete;
|
||||
ParseTree& operator=(ParseTree&&) = delete;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
* @brief Parses source text, requiring the grammar to consume all of it.
|
||||
* @returns True on success, in which case the tree holds the parsed nodes.
|
||||
*/
|
||||
bool parse(const Token* token, const std::string& source);
|
||||
|
||||
/**
|
||||
* @brief Parses from a reader, requiring the grammar to consume all of it.
|
||||
* @returns True on success, in which case the tree holds the parsed nodes.
|
||||
*/
|
||||
bool parse(const Token* token, TextReader& reader);
|
||||
|
||||
/**
|
||||
* @brief Parses the longest matching prefix of a reader.
|
||||
* @returns True when at least something was matched.
|
||||
*/
|
||||
bool parsePrefix(const Token* token, TextReader& reader);
|
||||
|
||||
/**
|
||||
* @brief Adopts an already produced parse result.
|
||||
*/
|
||||
void build(const TokenResult& result);
|
||||
|
||||
public:
|
||||
|
||||
ParseNode* root() { return resolve(root_index); }
|
||||
const ParseNode* root() const { return resolve(root_index); }
|
||||
|
||||
bool empty() const { return root_index == ParseNode::npos; }
|
||||
|
||||
public:
|
||||
|
||||
ParseNode* find(std::string_view name);
|
||||
const ParseNode* find(std::string_view name) const;
|
||||
|
||||
vector<ParseNode*> findAll(std::string_view name);
|
||||
vector<const ParseNode*> findAll(std::string_view name) const;
|
||||
|
||||
isize count(std::string_view name) const;
|
||||
|
||||
/**
|
||||
* @brief Human readable XML-like rendering of the whole tree.
|
||||
*/
|
||||
std::string toString() const;
|
||||
|
||||
};
|
||||
|
||||
}
|
||||
@@ -83,9 +83,11 @@ namespace spider {
|
||||
return t;
|
||||
}
|
||||
|
||||
std::u32string TokenResult::flatMatch() {
|
||||
std::u32string TokenResult::flatMatch() const {
|
||||
if (folded) return match;
|
||||
|
||||
std::u32string s = match;
|
||||
for(auto c : child) s += c.flatMatch();
|
||||
for (const auto& c : child) s += c.flatMatch();
|
||||
return s;
|
||||
}
|
||||
|
||||
@@ -244,12 +246,15 @@ namespace spider {
|
||||
auto r = target->test(ctx);
|
||||
if (!r.success) return r;
|
||||
|
||||
// Set tag of this result
|
||||
r.tag = tag_name;
|
||||
// The innermost tag wins, so wrapping a rule that is already tagged, as in
|
||||
// a choice of tagged rules, never hides what was actually matched.
|
||||
if (!r.tag.has_value()) r.tag = tag_name;
|
||||
|
||||
if (flatten) {
|
||||
// Fold the subtree text into this node, but keep the children so the
|
||||
// parsed result stays a walkable tree instead of a flat string.
|
||||
r.match = r.flatMatch();
|
||||
r.child.clear();
|
||||
r.folded = true;
|
||||
}
|
||||
|
||||
return r;
|
||||
|
||||
@@ -25,9 +25,19 @@ namespace spider {
|
||||
|
||||
vector<TokenResult> child = {};
|
||||
|
||||
/**
|
||||
* @brief Set when match already holds the folded text of the whole subtree.
|
||||
* @details Tagged rules that flatten their children keep those children around
|
||||
* so the parsed tree stays walkable, and flag the folded text here.
|
||||
*/
|
||||
bool folded = false;
|
||||
|
||||
public:
|
||||
|
||||
std::u32string flatMatch();
|
||||
/**
|
||||
* @brief The full matched text of this node and of all of its children.
|
||||
*/
|
||||
std::u32string flatMatch() const;
|
||||
|
||||
};
|
||||
|
||||
|
||||
Reference in New Issue
Block a user