From 5dd0c85d78ae33454d10a1398aebb9d6b170475f Mon Sep 17 00:00:00 2001 From: Kittcannon Date: Sun, 27 Sep 2026 01:05:03 -0600 Subject: [PATCH] ParseTree --- src/spider/compiler/Compiler.cpp | 139 ++++++++- src/spider/compiler/Compiler.hpp | 23 +- src/spider/compiler/assembler/AsmEBNF.cpp | 40 +-- src/spider/compiler/text/ParseTree.cpp | 334 ++++++++++++++++++++++ src/spider/compiler/text/ParseTree.hpp | 295 +++++++++++++++++++ src/spider/compiler/text/Token.cpp | 15 +- src/spider/compiler/text/Token.hpp | 12 +- 7 files changed, 822 insertions(+), 36 deletions(-) create mode 100644 src/spider/compiler/text/ParseTree.cpp create mode 100644 src/spider/compiler/text/ParseTree.hpp diff --git a/src/spider/compiler/Compiler.cpp b/src/spider/compiler/Compiler.cpp index 589a1a2..20c1826 100644 --- a/src/spider/compiler/Compiler.cpp +++ b/src/spider/compiler/Compiler.cpp @@ -1,14 +1,56 @@ #include +#include #include #include #include #include +#include using namespace spider; +// ============================================================================ +// Compiler Entry Points +// ============================================================================ + +namespace spider { + + TokenFactory& assemblyGrammar() { + static TokenFactory grammar; + static bool loaded = false; + if (!loaded) { + asm_ebnf::initTokens(grammar); + loaded = true; + } + return grammar; + } + + bool compileProgram(const std::string& source, ParseTree& out) { + return out.parse(asm_ebnf::program, source); + } + +} + +// ============================================================================ +// Test Bookkeeping +// ============================================================================ + +// Suite wide bookkeeping, so a failing case can never hide behind a clean exit code. +static isize total_tests = 0; +static isize failed_tests = 0; + +static void report(bool ok, const std::string& name) { + ++total_tests; + if (ok) { + std::cout << "[PASS] " << name << "\n"; + } else { + ++failed_tests; + std::cerr << "[FAIL] " << name << "\n"; + } +} + // Inline evaluator that executes the reader and prints standard output static bool run_test_case(const Token* token, const std::string& input, bool expectedSuccess, const std::u32string& expectedMatch = U"") { StringTextReader reader(input); @@ -17,6 +59,7 @@ static bool run_test_case(const Token* token, const std::string& input, bool exp bool status_ok = (res.success == expectedSuccess); bool match_ok = (!expectedSuccess) || expectedMatch.empty() || (res.flatMatch() == expectedMatch); + ++total_tests; if (status_ok && match_ok) { std::cout << "[PASS] Input: \"" << input << "\" -> " << (res.success ? "SUCCESS" : "FAILURE") @@ -30,6 +73,7 @@ static bool run_test_case(const Token* token, const std::string& input, bool exp if (expectedSuccess && !expectedMatch.empty()) { std::cerr << " Expected Match: \"" << unicode::toUTF8(expectedMatch) << "\", Got: \"" << unicode::toUTF8(res.flatMatch()) << "\"\n"; } + ++failed_tests; return false; } @@ -137,14 +181,85 @@ void test_full_program(TokenFactory& tf) { " @align(16) ADD R1, [R2 + R3 * 8 + 4]\n" " JMP start\n"; - StringTextReader reader(asm_code); - TokenResult res = asm_ebnf::program->test(reader); + ParseTree tree; + report(compileProgram(asm_code, tree), "full assembly program parses completely"); +} - if (res.success) { - std::cout << "[PASS] Full Assembly Program parsed successfully!\n"; - } else { - std::cerr << "[FAIL] Program parsing failed.\n"; +void test_parse_tree(TokenFactory& tf) { + std::cout << "\n--- Testing Parse Tree Inspection ---\n"; + const std::string asm_code = + "#include stdio\n" + "\n" + "start:\n" + " MOV R1, 0x20 ; Load constant\n" + " @align(16) ADD R1, [R2 + R3 * 8 + 4]\n" + " JMP start\n"; + + ParseTree tree; + report(compileProgram(asm_code, tree), "program becomes a tree of nodes"); + + ParseNode* root = tree.root(); + report(root != nullptr, "tree exposes a root node"); + if (root == nullptr) return; + + report(root->hasTag("program"), "root node is tagged as program"); + report(root->textUtf8() == asm_code, "root text covers the whole source"); + report(root->childCount() > 0, "root holds child nodes"); + + vector lines = root->findAll("line"); + report(lines.size() == 6, "program holds one node per line"); + + if (lines.size() == 6) { + report(lines.front()->parentNode() == root, "a line knows its parent"); + report(lines.front()->depth() == 1, "lines sit one level under the root"); + report(lines.front()->nextSibling() == lines[1], "sibling walk reaches the next line"); + report(lines[1]->previousSibling() == lines.front(), "sibling walk goes back"); + report(lines.back()->nextSibling() == nullptr, "the last line has no next sibling"); + + ParseNode* preprocessor = lines.front()->firstChild("preprocessor"); + report(preprocessor != nullptr, "first line holds a preprocessor child"); + report(preprocessor != nullptr && preprocessor->textUtf8() == "#include stdio", "preprocessor text is readable"); + + report(lines[2]->firstChild("line_label") != nullptr, "label line is recognized"); + ParseNode* label = lines[2]->find("label"); + report(label != nullptr && label->textUtf8() == "start:", "label text is readable"); + + ParseNode* mov = lines[3]->find("instruction"); + report(mov != nullptr, "instruction is found inside its line"); + if (mov != nullptr) { + report(mov->firstChild("opcode") != nullptr, "instruction exposes its opcode child"); + + ParseNode* operands = mov->firstChild("operand_list"); + report(operands != nullptr && operands->childCount() == 2, "operand list exposes both operands"); + + ParseNode* hexlit = mov->find("hex_lit"); + report(hexlit != nullptr && hexlit->textUtf8() == "0x20", "hex literal text is readable"); + } + + ParseNode* comment = lines[3]->firstChild("comment"); + report(comment != nullptr && comment->textUtf8() == "; Load constant", "trailing comment is captured"); + + report(lines[4]->firstChild("line_annotation") != nullptr, "annotation line is recognized"); + ParseNode* displacement = lines[4]->find("addrm_dis"); + report(displacement != nullptr && displacement->textUtf8() == "[R2 + R3 * 8 + 4]", "displacement operand is readable"); + report(tree.count("register") == 4, "every register of the program is reachable"); } + + ParseTree single; + report(single.parse(asm_ebnf::instruction, "MOV R1, [R2 + 4]"), "a single instruction parses on its own"); + + ParseNode* instruction = single.root(); + report(instruction != nullptr && instruction->hasTag("instruction"), "instruction root is tagged"); + if (instruction != nullptr) { + ParseNode* operands = instruction->firstChild("operand_list"); + report(operands != nullptr && operands->childCount() == 2, "operand list holds two operands"); + + ParseNode* second = operands == nullptr ? nullptr : operands->childAt(1); + report(second != nullptr && second->textUtf8() == "[R2 + 4]", "operand text is readable"); + report(second != nullptr && second->ancestor("instruction") == instruction, "an operand can climb back to its instruction"); + } + + std::cout << "\n--- Parsed Tree ---\n" << tree.toString() << "\n"; } // ============================================================================ @@ -153,7 +268,7 @@ void test_full_program(TokenFactory& tf) { int main() { TokenFactory tf; - asm_ebnf::initTokens(tf); + assemblyGrammar(); std::cout << "Running Token Framework Tests...\n"; @@ -164,6 +279,16 @@ int main() { test_addressing_modes(tf); test_instructions_and_lines(tf); test_full_program(tf); + test_parse_tree(tf); + + std::cout << "\n========================================\n"; + std::cout << "Test Results: " << (total_tests - failed_tests) << "/" << total_tests << " passed.\n"; + std::cout << "========================================\n"; + + if (failed_tests > 0) { + std::cerr << failed_tests << " test(s) FAILED.\n"; + return 1; + } std::cout << "All token framework tests passed successfully!\n"; return 0; diff --git a/src/spider/compiler/Compiler.hpp b/src/spider/compiler/Compiler.hpp index 740b5cf..7fdec51 100644 --- a/src/spider/compiler/Compiler.hpp +++ b/src/spider/compiler/Compiler.hpp @@ -1,10 +1,27 @@ -#pragma +#pragma once #include +#include +#include + namespace spider { - class Token; - class RootToken; + /** + * @brief The token factory that owns the assembly grammar, loaded on first use. + * @details Every rule of the assembly language lives in this factory, so a caller + * can test a single rule or a whole program against the same grammar. + */ + TokenFactory& assemblyGrammar(); + + /** + * @brief Parses a whole assembly program into an inspectable tree of nodes. + * @details The grammar has to consume the entire source, so a malformed program + * is rejected instead of quietly producing a partial tree. On success the + * resulting tree can be walked like a document: program, line, label, + * instruction, operands, literals and comments. + * @returns True on success, with the parsed nodes left inside out. + */ + bool compileProgram(const std::string& source, ParseTree& out); } diff --git a/src/spider/compiler/assembler/AsmEBNF.cpp b/src/spider/compiler/assembler/AsmEBNF.cpp index 85413eb..b45c8be 100644 --- a/src/spider/compiler/assembler/AsmEBNF.cpp +++ b/src/spider/compiler/assembler/AsmEBNF.cpp @@ -106,17 +106,17 @@ namespace spider::asm_ebnf { binary_digit = tf.choice("01"); ws_char = tf.fn(isWhithespaceCharNotCrLf); - ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true); - whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true); - newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }); + ws_optional = tf.rep(ws_char); + whitespace = tf.seq({ ws_char, tf.rep(ws_char) }); + newline = tf.tag(tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }), "newline", true); utf8_char = tf.fn(isUTF8CharNotCrLf); char_escape = tf.seq({ tf["\\"], utf8_char }); char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) }); - char_lit = tf.seq({ tf["'"], char_content, tf["'"] }); + char_lit = tf.tag(tf.seq({ tf["'"], char_content, tf["'"] }), "char_lit", true); string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) }); - string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }); + string_lit = tf.tag(tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }), "string_lit", true); // (* Literals *) identifier = tf.tag(tf.seq({ @@ -185,29 +185,29 @@ namespace spider::asm_ebnf { // (* Generalized Instructions *) opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true); - operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }); - instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction"); + operand_list = tf.tag(tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }), "operand_list", true); + instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction", true); // (* Added Preprocessor, Annotation *) - annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named"); - annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg"); - annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }); - annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }); - annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation"); + annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named", true); + annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg", true); + annotation_args = tf.tag(tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }), "annotation_args", true); + annotation_pars = tf.tag(tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }), "annotation_pars", true); + annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation", true); preprocessor_val = tf.choice({ identifier, literal_decl }); - preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor"); + preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor", true); // (* Line Structure & Program *) - label = tf.tag(tf.seq({ identifier, tf[":"] }), "label"); - line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }); - line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }); - line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction }); - line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }); - line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }); - program = tf.seq({ tf.rep(line), tf.opt(line_last) }); + label = tf.tag(tf.seq({ identifier, tf[":"] }), "label", true); + line_label = tf.tag(tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }), "line_label", true); + line_annotation = tf.tag(tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }), "line_annotation", true); + line_content = tf.tag(tf.choice({ preprocessor, line_annotation, line_label, instruction }), "line_content", true); + line = tf.tag(tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }), "line", true); + line_last = tf.tag(tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }), "line_last", true); + program = tf.tag(tf.seq({ tf.rep(line), tf.opt(line_last) }), "program", true); } } diff --git a/src/spider/compiler/text/ParseTree.cpp b/src/spider/compiler/text/ParseTree.cpp new file mode 100644 index 0000000..cc3ba22 --- /dev/null +++ b/src/spider/compiler/text/ParseTree.cpp @@ -0,0 +1,334 @@ +#include "ParseTree.hpp" + +namespace spider { + + // ============================================================================ + // Internal Helpers + // ============================================================================ + + /** + * @brief Renders control characters as escapes, so a node can be printed on one line. + */ + static std::string escapeText(std::string_view raw) { + std::string out; + out.reserve(raw.size()); + for (char c : raw) { + switch (c) { + case '\n': out += "\\n"; break; + case '\r': out += "\\r"; break; + case '\t': out += "\\t"; break; + default: out += c; break; + } + } + return out; + } + + // ============================================================================ + // ParseNode Navigation + // ============================================================================ + + isize ParseNode::depth() const { + isize levels = 0; + for (const ParseNode* n = parentNode(); n != nullptr; n = n->parentNode()) ++levels; + return levels; + } + + const ParseNode* ParseNode::parentNode() const { + if (owner == nullptr || parent == ParseNode::npos) return nullptr; + return owner->resolve(parent); + } + + const ParseNode* ParseNode::firstChild() const { + if (owner == nullptr || first_child == ParseNode::npos) return nullptr; + return owner->resolve(first_child); + } + + const ParseNode* ParseNode::lastChild() const { + if (owner == nullptr || last_child == ParseNode::npos) return nullptr; + return owner->resolve(last_child); + } + + const ParseNode* ParseNode::nextSibling() const { + if (owner == nullptr || next_sibling == ParseNode::npos) return nullptr; + return owner->resolve(next_sibling); + } + + const ParseNode* ParseNode::previousSibling() const { + if (owner == nullptr || prev_sibling == ParseNode::npos) return nullptr; + return owner->resolve(prev_sibling); + } + + const ParseNode* ParseNode::childAt(isize index) const { + isize seen = 0; + for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) { + if (seen == index) return c; + ++seen; + } + return nullptr; + } + + ParseNode* ParseNode::parentNode() { return const_cast(static_cast(this)->parentNode()); } + ParseNode* ParseNode::firstChild() { return const_cast(static_cast(this)->firstChild()); } + ParseNode* ParseNode::lastChild() { return const_cast(static_cast(this)->lastChild()); } + ParseNode* ParseNode::nextSibling() { return const_cast(static_cast(this)->nextSibling()); } + ParseNode* ParseNode::previousSibling() { return const_cast(static_cast(this)->previousSibling()); } + ParseNode* ParseNode::childAt(isize index) { return const_cast(static_cast(this)->childAt(index)); } + + // ============================================================================ + // ParseNode Navigation Filtered By Tag + // ============================================================================ + + const ParseNode* ParseNode::firstChild(std::string_view name) const { + for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) { + if (c->hasTag(name)) return c; + } + return nullptr; + } + + const ParseNode* ParseNode::lastChild(std::string_view name) const { + const ParseNode* found = nullptr; + for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) { + if (c->hasTag(name)) found = c; + } + return found; + } + + const ParseNode* ParseNode::nextSibling(std::string_view name) const { + for (const ParseNode* s = nextSibling(); s != nullptr; s = s->nextSibling()) { + if (s->hasTag(name)) return s; + } + return nullptr; + } + + const ParseNode* ParseNode::ancestor(std::string_view name) const { + for (const ParseNode* p = parentNode(); p != nullptr; p = p->parentNode()) { + if (p->hasTag(name)) return p; + } + return nullptr; + } + + ParseNode* ParseNode::firstChild(std::string_view name) { return const_cast(static_cast(this)->firstChild(name)); } + ParseNode* ParseNode::lastChild(std::string_view name) { return const_cast(static_cast(this)->lastChild(name)); } + ParseNode* ParseNode::nextSibling(std::string_view name) { return const_cast(static_cast(this)->nextSibling(name)); } + ParseNode* ParseNode::ancestor(std::string_view name) { return const_cast(static_cast(this)->ancestor(name)); } + + // ============================================================================ + // ParseNode Queries + // ============================================================================ + + const ParseNode* ParseNode::find(std::string_view name) const { + if (hasTag(name)) return this; + for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) { + if (const ParseNode* hit = c->find(name)) return hit; + } + return nullptr; + } + + vector ParseNode::findAll(std::string_view name) const { + vector hits; + if (hasTag(name)) hits.push_back(this); + for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) { + for (const ParseNode* hit : c->findAll(name)) hits.push_back(hit); + } + return hits; + } + + isize ParseNode::count(std::string_view name) const { return findAll(name).size(); } + + ParseNode* ParseNode::find(std::string_view name) { return const_cast(static_cast(this)->find(name)); } + + vector ParseNode::findAll(std::string_view name) { + vector hits = static_cast(this)->findAll(name); + vector out; + out.reserve(hits.size()); + for (const ParseNode* hit : hits) out.push_back(const_cast(hit)); + return out; + } + + // ============================================================================ + // ParseNode Rendering + // ============================================================================ + + std::string ParseNode::describe(isize depth) const { + const std::string pad(depth * 2, ' '); + + if (!tag.has_value()) { + if (isLeaf()) return pad + escapeText(ownTextUtf8()); + + std::string out; + for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) { + out += c->describe(depth) + "\n"; + } + if (!out.empty()) out.pop_back(); + return out; + } + + const std::string name(*tag); + if (isLeaf()) return pad + "<" + name + ">" + escapeText(ownTextUtf8()) + ""; + + std::string out = pad + "<" + name + ">"; + for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) { + out += "\n" + c->describe(depth + 1); + } + out += "\n" + pad + ""; + return out; + } + + std::string ParseNode::toString() const { return describe(0); } + std::string ParseNode::toString(isize depth) const { return describe(depth); } + + // ============================================================================ + // ParseTree Building + // ============================================================================ + + ParseNode* ParseTree::resolve(isize index) { + if (index == ParseNode::npos) return nullptr; + return &nodes[index]; + } + + const ParseNode* ParseTree::resolve(isize index) const { + if (index == ParseNode::npos) return nullptr; + return &nodes[index]; + } + + isize ParseTree::addNode(const TokenResult& result) { + ParseNode node; + node.owner = this; + node.tag = result.tag; + node.own = result.match; + node.text = result.flatMatch(); + + isize self = nodes.size(); + nodes.push_back(std::move(node)); + return self; + } + + void ParseTree::linkChildren(isize parent_index, const vector& children) { + ParseNode& parent = nodes[parent_index]; + parent.first_child = ParseNode::npos; + parent.last_child = ParseNode::npos; + parent.children_count = 0; + + for (isize child : children) { + ParseNode& kid = nodes[child]; + kid.parent = parent_index; + kid.prev_sibling = parent.last_child; + kid.next_sibling = ParseNode::npos; + + if (parent.last_child != ParseNode::npos) { + nodes[parent.last_child].next_sibling = child; + } else { + parent.first_child = child; + } + parent.last_child = child; + parent.children_count++; + } + } + + vector ParseTree::collapse(const TokenResult& result) { + vector kids; + for (const TokenResult& sub : result.child) { + for (isize kid : collapse(sub)) kids.push_back(kid); + } + + // Untagged rules are grammar scaffolding, never part of the exposed syntax. + if (!result.tag.has_value()) return kids; + + const std::u32string folded = result.flatMatch(); + isize self = addNode(result); + + // An empty shell has nothing to show. + if (kids.empty() && folded.empty()) return {}; + + linkChildren(self, kids); + return { self }; + } + + void ParseTree::build(const TokenResult& result) { + nodes.clear(); + root_index = ParseNode::npos; + if (!result.success) return; + + vector tops = collapse(result); + if (tops.size() == 1) { + root_index = tops.front(); + return; + } + + // A grammar that exposes several top level rules still gets one container. + ParseNode root; + root.owner = this; + root_index = nodes.size(); + nodes.push_back(std::move(root)); + linkChildren(root_index, tops); + } + + bool ParseTree::parsePrefix(const Token* token, TextReader& reader) { + nodes.clear(); + root_index = ParseNode::npos; + if (token == nullptr) return false; + + TokenResult result = token->test(reader); + if (!result.success) return false; + + build(result); + return true; + } + + bool ParseTree::parse(const Token* token, TextReader& reader) { + nodes.clear(); + root_index = ParseNode::npos; + if (token == nullptr) return false; + + TokenResult result = token->test(reader); + if (!result.success) return false; + + // A complete parse leaves nothing behind in the reader. + if (reader.hasError()) return false; + if (reader.current().has_value()) return false; + + build(result); + return true; + } + + bool ParseTree::parse(const Token* token, const std::string& source) { + StringTextReader reader(source); + return parse(token, reader); + } + + // ============================================================================ + // ParseTree Inspection + // ============================================================================ + + const ParseNode* ParseTree::find(std::string_view name) const { + const ParseNode* r = root(); + return r == nullptr ? nullptr : r->find(name); + } + + vector ParseTree::findAll(std::string_view name) const { + const ParseNode* r = root(); + if (r == nullptr) return {}; + return r->findAll(name); + } + + isize ParseTree::count(std::string_view name) const { + const ParseNode* r = root(); + return r == nullptr ? 0 : r->count(name); + } + + std::string ParseTree::toString() const { + const ParseNode* r = root(); + return r == nullptr ? std::string() : r->toString(); + } + + ParseNode* ParseTree::find(std::string_view name) { return const_cast(static_cast(this)->find(name)); } + + vector ParseTree::findAll(std::string_view name) { + vector hits = static_cast(this)->findAll(name); + vector out; + out.reserve(hits.size()); + for (const ParseNode* hit : hits) out.push_back(const_cast(hit)); + return out; + } + +} diff --git a/src/spider/compiler/text/ParseTree.hpp b/src/spider/compiler/text/ParseTree.hpp new file mode 100644 index 0000000..034c664 --- /dev/null +++ b/src/spider/compiler/text/ParseTree.hpp @@ -0,0 +1,295 @@ +#pragma once + +#include + +#include +#include + +namespace spider { + + class ParseTree; + + /** + * @brief DOM style handle over a single node of a parsed token tree. + * @details Every node knows its parent, its children and its siblings, so a + * parsed program can be walked and inspected much like a document. + * Nodes are owned by the ParseTree that produced them and stay + * valid while that tree is alive. The tree itself is built by + * ParseTree, never by hand. + */ + class ParseNode { + friend class ParseTree; + + public: + + /** @brief Index value used for "no node" links, since isize is unsigned. */ + static constexpr isize npos = static_cast(-1); + + private: + + ParseTree* owner = nullptr; + + isize parent = npos; + isize first_child = npos; + isize last_child = npos; + isize next_sibling = npos; + isize prev_sibling = npos; + isize children_count = 0; + + optional tag = {}; + std::u32string own = U""; + std::u32string text = U""; + + public: + + ParseNode() = default; + + public: + + /** + * @brief The grammar tag of this node, or nothing if untagged. + */ + const optional& tagName() const { return tag; } + + /** + * @brief Checks if this node carries the given tag. + */ + bool hasTag(std::string_view name) const { return tag.has_value() && *tag == name; } + + /** + * @brief The full source text covered by this node and all of its children. + */ + const std::u32string& fullText() const { return text; } + + /** + * @brief The source text held by this node alone, without its children. + * @note For tagged nodes that fold their children, this is the full text. + */ + const std::u32string& ownText() const { return own; } + + /** + * @brief UTF-8 rendering of text(). + */ + std::string textUtf8() const { return unicode::toUTF8(text); } + + /** + * @brief UTF-8 rendering of ownText(). + */ + std::string ownTextUtf8() const { return unicode::toUTF8(own); } + + /** + * @brief Amount of direct children of this node. + */ + isize childCount() const { return children_count; } + + /** + * @brief True when this node holds no text and no children. + */ + bool isLeaf() const { return children_count == 0; } + + /** + * @brief Distance from the root of the tree, zero for the root itself. + */ + isize depth() const; + + /** + * @brief Storage index of this node inside its parent, npos for the root. + */ + isize indexInParent() const { return parent; } + + public: + + // ---------------------------------------------------------------- // + // Navigation // + // ---------------------------------------------------------------- // + + ParseNode* parentNode(); + const ParseNode* parentNode() const; + + ParseNode* firstChild(); + const ParseNode* firstChild() const; + + ParseNode* lastChild(); + const ParseNode* lastChild() const; + + ParseNode* nextSibling(); + const ParseNode* nextSibling() const; + + ParseNode* previousSibling(); + const ParseNode* previousSibling() const; + + /** + * @brief The index-th direct child of this node, nullptr when out of range. + */ + ParseNode* childAt(isize index); + const ParseNode* childAt(isize index) const; + + public: + + // ---------------------------------------------------------------- // + // Navigation filtered by tag // + // ---------------------------------------------------------------- // + + /** + * @brief First direct child carrying the given tag. + */ + ParseNode* firstChild(std::string_view name); + const ParseNode* firstChild(std::string_view name) const; + + /** + * @brief Last direct child carrying the given tag. + */ + ParseNode* lastChild(std::string_view name); + const ParseNode* lastChild(std::string_view name) const; + + /** + * @brief Next sibling of this node carrying the given tag. + */ + ParseNode* nextSibling(std::string_view name); + const ParseNode* nextSibling(std::string_view name) const; + + /** + * @brief Nearest ancestor carrying the given tag, nullptr when there is none. + */ + ParseNode* ancestor(std::string_view name); + const ParseNode* ancestor(std::string_view name) const; + + public: + + // ---------------------------------------------------------------- // + // Queries // + // ---------------------------------------------------------------- // + + /** + * @brief First node in document order carrying the given tag, this node included. + */ + ParseNode* find(std::string_view name); + const ParseNode* find(std::string_view name) const; + + /** + * @brief Every node in document order carrying the given tag, this node included. + */ + vector findAll(std::string_view name); + vector findAll(std::string_view name) const; + + /** + * @brief Amount of nodes in document order carrying the given tag. + */ + isize count(std::string_view name) const; + + /** + * @brief True when at least one node carries the given tag. + */ + bool contains(std::string_view name) const { return find(name) != nullptr; } + + public: + + /** + * @brief Human readable XML-like rendering of this node and its children. + */ + std::string toString() const; + + /** + * @brief XML-like rendering of this node alone, indented by the given depth. + */ + std::string toString(isize depth) const; + + private: + + std::string describe(isize depth) const; + + }; + + /** + * @brief Owner of a parsed token tree, exposing a document like interface. + * @details Use parse() to turn source text into an inspectable tree. The tree + * is materialized once and then only read, so any number of consumers + * can walk the same nodes safely. + */ + class ParseTree { + friend class ParseNode; + + private: + + deque nodes; + isize root_index = ParseNode::npos; + + private: + + ParseNode* resolve(isize index); + const ParseNode* resolve(isize index) const; + + isize addNode(const TokenResult& result); + + void linkChildren(isize parent, const vector& children); + + /** + * @brief Materializes a parse result, dropping the grammar scaffolding. + * @details Only tagged rules become nodes, so a consumer walks syntax and not + * combinators: untagged rules are pure plumbing and simply hoist their + * children upwards, and empty shells are discarded. Returns the top + * level nodes produced by this subtree. + */ + vector collapse(const TokenResult& result); + + public: + + ParseTree() = default; + ~ParseTree() = default; + + // Nodes point back at their owning tree, so trees are never copied or moved. + ParseTree(const ParseTree&) = delete; + ParseTree& operator=(const ParseTree&) = delete; + ParseTree(ParseTree&&) = delete; + ParseTree& operator=(ParseTree&&) = delete; + + public: + + /** + * @brief Parses source text, requiring the grammar to consume all of it. + * @returns True on success, in which case the tree holds the parsed nodes. + */ + bool parse(const Token* token, const std::string& source); + + /** + * @brief Parses from a reader, requiring the grammar to consume all of it. + * @returns True on success, in which case the tree holds the parsed nodes. + */ + bool parse(const Token* token, TextReader& reader); + + /** + * @brief Parses the longest matching prefix of a reader. + * @returns True when at least something was matched. + */ + bool parsePrefix(const Token* token, TextReader& reader); + + /** + * @brief Adopts an already produced parse result. + */ + void build(const TokenResult& result); + + public: + + ParseNode* root() { return resolve(root_index); } + const ParseNode* root() const { return resolve(root_index); } + + bool empty() const { return root_index == ParseNode::npos; } + + public: + + ParseNode* find(std::string_view name); + const ParseNode* find(std::string_view name) const; + + vector findAll(std::string_view name); + vector findAll(std::string_view name) const; + + isize count(std::string_view name) const; + + /** + * @brief Human readable XML-like rendering of the whole tree. + */ + std::string toString() const; + + }; + +} diff --git a/src/spider/compiler/text/Token.cpp b/src/spider/compiler/text/Token.cpp index 80555b5..93a8bf2 100644 --- a/src/spider/compiler/text/Token.cpp +++ b/src/spider/compiler/text/Token.cpp @@ -83,9 +83,11 @@ namespace spider { return t; } - std::u32string TokenResult::flatMatch() { + std::u32string TokenResult::flatMatch() const { + if (folded) return match; + std::u32string s = match; - for(auto c : child) s += c.flatMatch(); + for (const auto& c : child) s += c.flatMatch(); return s; } @@ -244,12 +246,15 @@ namespace spider { auto r = target->test(ctx); if (!r.success) return r; - // Set tag of this result - r.tag = tag_name; + // The innermost tag wins, so wrapping a rule that is already tagged, as in + // a choice of tagged rules, never hides what was actually matched. + if (!r.tag.has_value()) r.tag = tag_name; if (flatten) { + // Fold the subtree text into this node, but keep the children so the + // parsed result stays a walkable tree instead of a flat string. r.match = r.flatMatch(); - r.child.clear(); + r.folded = true; } return r; diff --git a/src/spider/compiler/text/Token.hpp b/src/spider/compiler/text/Token.hpp index b3fbdb6..b3ab281 100644 --- a/src/spider/compiler/text/Token.hpp +++ b/src/spider/compiler/text/Token.hpp @@ -25,9 +25,19 @@ namespace spider { vector child = {}; + /** + * @brief Set when match already holds the folded text of the whole subtree. + * @details Tagged rules that flatten their children keep those children around + * so the parsed tree stays walkable, and flag the folded text here. + */ + bool folded = false; + public: - std::u32string flatMatch(); + /** + * @brief The full matched text of this node and of all of its children. + */ + std::u32string flatMatch() const; };