ParseTree

This commit is contained in:
2026-09-27 01:05:03 -06:00
parent df2a7c8289
commit 5dd0c85d78
7 changed files with 822 additions and 36 deletions
+133 -8
View File
@@ -1,14 +1,56 @@
#include <iostream>
#include <spider/compiler/Compiler.hpp>
#include <spider/compiler/common.hpp>
#include <spider/compiler/text/unicode.hpp>
#include <spider/compiler/text/TextReader.hpp>
#include <spider/compiler/assembler/AsmEBNF.hpp>
#include <spider/compiler/text/ParseTree.hpp>
using namespace spider;
// ============================================================================
// Compiler Entry Points
// ============================================================================
namespace spider {
TokenFactory& assemblyGrammar() {
static TokenFactory grammar;
static bool loaded = false;
if (!loaded) {
asm_ebnf::initTokens(grammar);
loaded = true;
}
return grammar;
}
bool compileProgram(const std::string& source, ParseTree& out) {
return out.parse(asm_ebnf::program, source);
}
}
// ============================================================================
// Test Bookkeeping
// ============================================================================
// Suite wide bookkeeping, so a failing case can never hide behind a clean exit code.
static isize total_tests = 0;
static isize failed_tests = 0;
static void report(bool ok, const std::string& name) {
++total_tests;
if (ok) {
std::cout << "[PASS] " << name << "\n";
} else {
++failed_tests;
std::cerr << "[FAIL] " << name << "\n";
}
}
// Inline evaluator that executes the reader and prints standard output
static bool run_test_case(const Token* token, const std::string& input, bool expectedSuccess, const std::u32string& expectedMatch = U"") {
StringTextReader reader(input);
@@ -17,6 +59,7 @@ static bool run_test_case(const Token* token, const std::string& input, bool exp
bool status_ok = (res.success == expectedSuccess);
bool match_ok = (!expectedSuccess) || expectedMatch.empty() || (res.flatMatch() == expectedMatch);
++total_tests;
if (status_ok && match_ok) {
std::cout << "[PASS] Input: \"" << input << "\" -> "
<< (res.success ? "SUCCESS" : "FAILURE")
@@ -30,6 +73,7 @@ static bool run_test_case(const Token* token, const std::string& input, bool exp
if (expectedSuccess && !expectedMatch.empty()) {
std::cerr << " Expected Match: \"" << unicode::toUTF8(expectedMatch) << "\", Got: \"" << unicode::toUTF8(res.flatMatch()) << "\"\n";
}
++failed_tests;
return false;
}
@@ -137,14 +181,85 @@ void test_full_program(TokenFactory& tf) {
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
" JMP start\n";
StringTextReader reader(asm_code);
TokenResult res = asm_ebnf::program->test(reader);
if (res.success) {
std::cout << "[PASS] Full Assembly Program parsed successfully!\n";
} else {
std::cerr << "[FAIL] Program parsing failed.\n";
ParseTree tree;
report(compileProgram(asm_code, tree), "full assembly program parses completely");
}
void test_parse_tree(TokenFactory& tf) {
std::cout << "\n--- Testing Parse Tree Inspection ---\n";
const std::string asm_code =
"#include stdio\n"
"\n"
"start:\n"
" MOV R1, 0x20 ; Load constant\n"
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
" JMP start\n";
ParseTree tree;
report(compileProgram(asm_code, tree), "program becomes a tree of nodes");
ParseNode* root = tree.root();
report(root != nullptr, "tree exposes a root node");
if (root == nullptr) return;
report(root->hasTag("program"), "root node is tagged as program");
report(root->textUtf8() == asm_code, "root text covers the whole source");
report(root->childCount() > 0, "root holds child nodes");
vector<ParseNode*> lines = root->findAll("line");
report(lines.size() == 6, "program holds one node per line");
if (lines.size() == 6) {
report(lines.front()->parentNode() == root, "a line knows its parent");
report(lines.front()->depth() == 1, "lines sit one level under the root");
report(lines.front()->nextSibling() == lines[1], "sibling walk reaches the next line");
report(lines[1]->previousSibling() == lines.front(), "sibling walk goes back");
report(lines.back()->nextSibling() == nullptr, "the last line has no next sibling");
ParseNode* preprocessor = lines.front()->firstChild("preprocessor");
report(preprocessor != nullptr, "first line holds a preprocessor child");
report(preprocessor != nullptr && preprocessor->textUtf8() == "#include stdio", "preprocessor text is readable");
report(lines[2]->firstChild("line_label") != nullptr, "label line is recognized");
ParseNode* label = lines[2]->find("label");
report(label != nullptr && label->textUtf8() == "start:", "label text is readable");
ParseNode* mov = lines[3]->find("instruction");
report(mov != nullptr, "instruction is found inside its line");
if (mov != nullptr) {
report(mov->firstChild("opcode") != nullptr, "instruction exposes its opcode child");
ParseNode* operands = mov->firstChild("operand_list");
report(operands != nullptr && operands->childCount() == 2, "operand list exposes both operands");
ParseNode* hexlit = mov->find("hex_lit");
report(hexlit != nullptr && hexlit->textUtf8() == "0x20", "hex literal text is readable");
}
ParseNode* comment = lines[3]->firstChild("comment");
report(comment != nullptr && comment->textUtf8() == "; Load constant", "trailing comment is captured");
report(lines[4]->firstChild("line_annotation") != nullptr, "annotation line is recognized");
ParseNode* displacement = lines[4]->find("addrm_dis");
report(displacement != nullptr && displacement->textUtf8() == "[R2 + R3 * 8 + 4]", "displacement operand is readable");
report(tree.count("register") == 4, "every register of the program is reachable");
}
ParseTree single;
report(single.parse(asm_ebnf::instruction, "MOV R1, [R2 + 4]"), "a single instruction parses on its own");
ParseNode* instruction = single.root();
report(instruction != nullptr && instruction->hasTag("instruction"), "instruction root is tagged");
if (instruction != nullptr) {
ParseNode* operands = instruction->firstChild("operand_list");
report(operands != nullptr && operands->childCount() == 2, "operand list holds two operands");
ParseNode* second = operands == nullptr ? nullptr : operands->childAt(1);
report(second != nullptr && second->textUtf8() == "[R2 + 4]", "operand text is readable");
report(second != nullptr && second->ancestor("instruction") == instruction, "an operand can climb back to its instruction");
}
std::cout << "\n--- Parsed Tree ---\n" << tree.toString() << "\n";
}
// ============================================================================
@@ -153,7 +268,7 @@ void test_full_program(TokenFactory& tf) {
int main() {
TokenFactory tf;
asm_ebnf::initTokens(tf);
assemblyGrammar();
std::cout << "Running Token Framework Tests...\n";
@@ -164,6 +279,16 @@ int main() {
test_addressing_modes(tf);
test_instructions_and_lines(tf);
test_full_program(tf);
test_parse_tree(tf);
std::cout << "\n========================================\n";
std::cout << "Test Results: " << (total_tests - failed_tests) << "/" << total_tests << " passed.\n";
std::cout << "========================================\n";
if (failed_tests > 0) {
std::cerr << failed_tests << " test(s) FAILED.\n";
return 1;
}
std::cout << "All token framework tests passed successfully!\n";
return 0;
+20 -3
View File
@@ -1,10 +1,27 @@
#pragma
#pragma once
#include <spider/compiler/common.hpp>
#include <spider/compiler/text/Token.hpp>
#include <spider/compiler/text/ParseTree.hpp>
namespace spider {
class Token;
class RootToken;
/**
* @brief The token factory that owns the assembly grammar, loaded on first use.
* @details Every rule of the assembly language lives in this factory, so a caller
* can test a single rule or a whole program against the same grammar.
*/
TokenFactory& assemblyGrammar();
/**
* @brief Parses a whole assembly program into an inspectable tree of nodes.
* @details The grammar has to consume the entire source, so a malformed program
* is rejected instead of quietly producing a partial tree. On success the
* resulting tree can be walked like a document: program, line, label,
* instruction, operands, literals and comments.
* @returns True on success, with the parsed nodes left inside out.
*/
bool compileProgram(const std::string& source, ParseTree& out);
}
+20 -20
View File
@@ -106,17 +106,17 @@ namespace spider::asm_ebnf {
binary_digit = tf.choice("01");
ws_char = tf.fn(isWhithespaceCharNotCrLf);
ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true);
whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true);
newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
ws_optional = tf.rep(ws_char);
whitespace = tf.seq({ ws_char, tf.rep(ws_char) });
newline = tf.tag(tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }), "newline", true);
utf8_char = tf.fn(isUTF8CharNotCrLf);
char_escape = tf.seq({ tf["\\"], utf8_char });
char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
char_lit = tf.tag(tf.seq({ tf["'"], char_content, tf["'"] }), "char_lit", true);
string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
string_lit = tf.tag(tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }), "string_lit", true);
// (* Literals *)
identifier = tf.tag(tf.seq({
@@ -185,29 +185,29 @@ namespace spider::asm_ebnf {
// (* Generalized Instructions *)
opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction");
operand_list = tf.tag(tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }), "operand_list", true);
instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction", true);
// (* Added Preprocessor, Annotation *)
annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named");
annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg");
annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation");
annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named", true);
annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg", true);
annotation_args = tf.tag(tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }), "annotation_args", true);
annotation_pars = tf.tag(tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }), "annotation_pars", true);
annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation", true);
preprocessor_val = tf.choice({ identifier, literal_decl });
preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor");
preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor", true);
// (* Line Structure & Program *)
label = tf.tag(tf.seq({ identifier, tf[":"] }), "label");
line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
program = tf.seq({ tf.rep(line), tf.opt(line_last) });
label = tf.tag(tf.seq({ identifier, tf[":"] }), "label", true);
line_label = tf.tag(tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }), "line_label", true);
line_annotation = tf.tag(tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }), "line_annotation", true);
line_content = tf.tag(tf.choice({ preprocessor, line_annotation, line_label, instruction }), "line_content", true);
line = tf.tag(tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }), "line", true);
line_last = tf.tag(tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }), "line_last", true);
program = tf.tag(tf.seq({ tf.rep(line), tf.opt(line_last) }), "program", true);
}
}
+334
View File
@@ -0,0 +1,334 @@
#include "ParseTree.hpp"
namespace spider {
// ============================================================================
// Internal Helpers
// ============================================================================
/**
* @brief Renders control characters as escapes, so a node can be printed on one line.
*/
static std::string escapeText(std::string_view raw) {
std::string out;
out.reserve(raw.size());
for (char c : raw) {
switch (c) {
case '\n': out += "\\n"; break;
case '\r': out += "\\r"; break;
case '\t': out += "\\t"; break;
default: out += c; break;
}
}
return out;
}
// ============================================================================
// ParseNode Navigation
// ============================================================================
isize ParseNode::depth() const {
isize levels = 0;
for (const ParseNode* n = parentNode(); n != nullptr; n = n->parentNode()) ++levels;
return levels;
}
const ParseNode* ParseNode::parentNode() const {
if (owner == nullptr || parent == ParseNode::npos) return nullptr;
return owner->resolve(parent);
}
const ParseNode* ParseNode::firstChild() const {
if (owner == nullptr || first_child == ParseNode::npos) return nullptr;
return owner->resolve(first_child);
}
const ParseNode* ParseNode::lastChild() const {
if (owner == nullptr || last_child == ParseNode::npos) return nullptr;
return owner->resolve(last_child);
}
const ParseNode* ParseNode::nextSibling() const {
if (owner == nullptr || next_sibling == ParseNode::npos) return nullptr;
return owner->resolve(next_sibling);
}
const ParseNode* ParseNode::previousSibling() const {
if (owner == nullptr || prev_sibling == ParseNode::npos) return nullptr;
return owner->resolve(prev_sibling);
}
const ParseNode* ParseNode::childAt(isize index) const {
isize seen = 0;
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
if (seen == index) return c;
++seen;
}
return nullptr;
}
ParseNode* ParseNode::parentNode() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->parentNode()); }
ParseNode* ParseNode::firstChild() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->firstChild()); }
ParseNode* ParseNode::lastChild() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->lastChild()); }
ParseNode* ParseNode::nextSibling() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->nextSibling()); }
ParseNode* ParseNode::previousSibling() { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->previousSibling()); }
ParseNode* ParseNode::childAt(isize index) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->childAt(index)); }
// ============================================================================
// ParseNode Navigation Filtered By Tag
// ============================================================================
const ParseNode* ParseNode::firstChild(std::string_view name) const {
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
if (c->hasTag(name)) return c;
}
return nullptr;
}
const ParseNode* ParseNode::lastChild(std::string_view name) const {
const ParseNode* found = nullptr;
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
if (c->hasTag(name)) found = c;
}
return found;
}
const ParseNode* ParseNode::nextSibling(std::string_view name) const {
for (const ParseNode* s = nextSibling(); s != nullptr; s = s->nextSibling()) {
if (s->hasTag(name)) return s;
}
return nullptr;
}
const ParseNode* ParseNode::ancestor(std::string_view name) const {
for (const ParseNode* p = parentNode(); p != nullptr; p = p->parentNode()) {
if (p->hasTag(name)) return p;
}
return nullptr;
}
ParseNode* ParseNode::firstChild(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->firstChild(name)); }
ParseNode* ParseNode::lastChild(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->lastChild(name)); }
ParseNode* ParseNode::nextSibling(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->nextSibling(name)); }
ParseNode* ParseNode::ancestor(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->ancestor(name)); }
// ============================================================================
// ParseNode Queries
// ============================================================================
const ParseNode* ParseNode::find(std::string_view name) const {
if (hasTag(name)) return this;
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
if (const ParseNode* hit = c->find(name)) return hit;
}
return nullptr;
}
vector<const ParseNode*> ParseNode::findAll(std::string_view name) const {
vector<const ParseNode*> hits;
if (hasTag(name)) hits.push_back(this);
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
for (const ParseNode* hit : c->findAll(name)) hits.push_back(hit);
}
return hits;
}
isize ParseNode::count(std::string_view name) const { return findAll(name).size(); }
ParseNode* ParseNode::find(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseNode*>(this)->find(name)); }
vector<ParseNode*> ParseNode::findAll(std::string_view name) {
vector<const ParseNode*> hits = static_cast<const ParseNode*>(this)->findAll(name);
vector<ParseNode*> out;
out.reserve(hits.size());
for (const ParseNode* hit : hits) out.push_back(const_cast<ParseNode*>(hit));
return out;
}
// ============================================================================
// ParseNode Rendering
// ============================================================================
std::string ParseNode::describe(isize depth) const {
const std::string pad(depth * 2, ' ');
if (!tag.has_value()) {
if (isLeaf()) return pad + escapeText(ownTextUtf8());
std::string out;
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
out += c->describe(depth) + "\n";
}
if (!out.empty()) out.pop_back();
return out;
}
const std::string name(*tag);
if (isLeaf()) return pad + "<" + name + ">" + escapeText(ownTextUtf8()) + "</" + name + ">";
std::string out = pad + "<" + name + ">";
for (const ParseNode* c = firstChild(); c != nullptr; c = c->nextSibling()) {
out += "\n" + c->describe(depth + 1);
}
out += "\n" + pad + "</" + name + ">";
return out;
}
std::string ParseNode::toString() const { return describe(0); }
std::string ParseNode::toString(isize depth) const { return describe(depth); }
// ============================================================================
// ParseTree Building
// ============================================================================
ParseNode* ParseTree::resolve(isize index) {
if (index == ParseNode::npos) return nullptr;
return &nodes[index];
}
const ParseNode* ParseTree::resolve(isize index) const {
if (index == ParseNode::npos) return nullptr;
return &nodes[index];
}
isize ParseTree::addNode(const TokenResult& result) {
ParseNode node;
node.owner = this;
node.tag = result.tag;
node.own = result.match;
node.text = result.flatMatch();
isize self = nodes.size();
nodes.push_back(std::move(node));
return self;
}
void ParseTree::linkChildren(isize parent_index, const vector<isize>& children) {
ParseNode& parent = nodes[parent_index];
parent.first_child = ParseNode::npos;
parent.last_child = ParseNode::npos;
parent.children_count = 0;
for (isize child : children) {
ParseNode& kid = nodes[child];
kid.parent = parent_index;
kid.prev_sibling = parent.last_child;
kid.next_sibling = ParseNode::npos;
if (parent.last_child != ParseNode::npos) {
nodes[parent.last_child].next_sibling = child;
} else {
parent.first_child = child;
}
parent.last_child = child;
parent.children_count++;
}
}
vector<isize> ParseTree::collapse(const TokenResult& result) {
vector<isize> kids;
for (const TokenResult& sub : result.child) {
for (isize kid : collapse(sub)) kids.push_back(kid);
}
// Untagged rules are grammar scaffolding, never part of the exposed syntax.
if (!result.tag.has_value()) return kids;
const std::u32string folded = result.flatMatch();
isize self = addNode(result);
// An empty shell has nothing to show.
if (kids.empty() && folded.empty()) return {};
linkChildren(self, kids);
return { self };
}
void ParseTree::build(const TokenResult& result) {
nodes.clear();
root_index = ParseNode::npos;
if (!result.success) return;
vector<isize> tops = collapse(result);
if (tops.size() == 1) {
root_index = tops.front();
return;
}
// A grammar that exposes several top level rules still gets one container.
ParseNode root;
root.owner = this;
root_index = nodes.size();
nodes.push_back(std::move(root));
linkChildren(root_index, tops);
}
bool ParseTree::parsePrefix(const Token* token, TextReader& reader) {
nodes.clear();
root_index = ParseNode::npos;
if (token == nullptr) return false;
TokenResult result = token->test(reader);
if (!result.success) return false;
build(result);
return true;
}
bool ParseTree::parse(const Token* token, TextReader& reader) {
nodes.clear();
root_index = ParseNode::npos;
if (token == nullptr) return false;
TokenResult result = token->test(reader);
if (!result.success) return false;
// A complete parse leaves nothing behind in the reader.
if (reader.hasError()) return false;
if (reader.current().has_value()) return false;
build(result);
return true;
}
bool ParseTree::parse(const Token* token, const std::string& source) {
StringTextReader reader(source);
return parse(token, reader);
}
// ============================================================================
// ParseTree Inspection
// ============================================================================
const ParseNode* ParseTree::find(std::string_view name) const {
const ParseNode* r = root();
return r == nullptr ? nullptr : r->find(name);
}
vector<const ParseNode*> ParseTree::findAll(std::string_view name) const {
const ParseNode* r = root();
if (r == nullptr) return {};
return r->findAll(name);
}
isize ParseTree::count(std::string_view name) const {
const ParseNode* r = root();
return r == nullptr ? 0 : r->count(name);
}
std::string ParseTree::toString() const {
const ParseNode* r = root();
return r == nullptr ? std::string() : r->toString();
}
ParseNode* ParseTree::find(std::string_view name) { return const_cast<ParseNode*>(static_cast<const ParseTree*>(this)->find(name)); }
vector<ParseNode*> ParseTree::findAll(std::string_view name) {
vector<const ParseNode*> hits = static_cast<const ParseTree*>(this)->findAll(name);
vector<ParseNode*> out;
out.reserve(hits.size());
for (const ParseNode* hit : hits) out.push_back(const_cast<ParseNode*>(hit));
return out;
}
}
+295
View File
@@ -0,0 +1,295 @@
#pragma once
#include <spider/compiler/common.hpp>
#include <spider/compiler/text/unicode.hpp>
#include <spider/compiler/text/Token.hpp>
namespace spider {
class ParseTree;
/**
* @brief DOM style handle over a single node of a parsed token tree.
* @details Every node knows its parent, its children and its siblings, so a
* parsed program can be walked and inspected much like a document.
* Nodes are owned by the ParseTree that produced them and stay
* valid while that tree is alive. The tree itself is built by
* ParseTree, never by hand.
*/
class ParseNode {
friend class ParseTree;
public:
/** @brief Index value used for "no node" links, since isize is unsigned. */
static constexpr isize npos = static_cast<isize>(-1);
private:
ParseTree* owner = nullptr;
isize parent = npos;
isize first_child = npos;
isize last_child = npos;
isize next_sibling = npos;
isize prev_sibling = npos;
isize children_count = 0;
optional<std::string_view> tag = {};
std::u32string own = U"";
std::u32string text = U"";
public:
ParseNode() = default;
public:
/**
* @brief The grammar tag of this node, or nothing if untagged.
*/
const optional<std::string_view>& tagName() const { return tag; }
/**
* @brief Checks if this node carries the given tag.
*/
bool hasTag(std::string_view name) const { return tag.has_value() && *tag == name; }
/**
* @brief The full source text covered by this node and all of its children.
*/
const std::u32string& fullText() const { return text; }
/**
* @brief The source text held by this node alone, without its children.
* @note For tagged nodes that fold their children, this is the full text.
*/
const std::u32string& ownText() const { return own; }
/**
* @brief UTF-8 rendering of text().
*/
std::string textUtf8() const { return unicode::toUTF8(text); }
/**
* @brief UTF-8 rendering of ownText().
*/
std::string ownTextUtf8() const { return unicode::toUTF8(own); }
/**
* @brief Amount of direct children of this node.
*/
isize childCount() const { return children_count; }
/**
* @brief True when this node holds no text and no children.
*/
bool isLeaf() const { return children_count == 0; }
/**
* @brief Distance from the root of the tree, zero for the root itself.
*/
isize depth() const;
/**
* @brief Storage index of this node inside its parent, npos for the root.
*/
isize indexInParent() const { return parent; }
public:
// ---------------------------------------------------------------- //
// Navigation //
// ---------------------------------------------------------------- //
ParseNode* parentNode();
const ParseNode* parentNode() const;
ParseNode* firstChild();
const ParseNode* firstChild() const;
ParseNode* lastChild();
const ParseNode* lastChild() const;
ParseNode* nextSibling();
const ParseNode* nextSibling() const;
ParseNode* previousSibling();
const ParseNode* previousSibling() const;
/**
* @brief The index-th direct child of this node, nullptr when out of range.
*/
ParseNode* childAt(isize index);
const ParseNode* childAt(isize index) const;
public:
// ---------------------------------------------------------------- //
// Navigation filtered by tag //
// ---------------------------------------------------------------- //
/**
* @brief First direct child carrying the given tag.
*/
ParseNode* firstChild(std::string_view name);
const ParseNode* firstChild(std::string_view name) const;
/**
* @brief Last direct child carrying the given tag.
*/
ParseNode* lastChild(std::string_view name);
const ParseNode* lastChild(std::string_view name) const;
/**
* @brief Next sibling of this node carrying the given tag.
*/
ParseNode* nextSibling(std::string_view name);
const ParseNode* nextSibling(std::string_view name) const;
/**
* @brief Nearest ancestor carrying the given tag, nullptr when there is none.
*/
ParseNode* ancestor(std::string_view name);
const ParseNode* ancestor(std::string_view name) const;
public:
// ---------------------------------------------------------------- //
// Queries //
// ---------------------------------------------------------------- //
/**
* @brief First node in document order carrying the given tag, this node included.
*/
ParseNode* find(std::string_view name);
const ParseNode* find(std::string_view name) const;
/**
* @brief Every node in document order carrying the given tag, this node included.
*/
vector<ParseNode*> findAll(std::string_view name);
vector<const ParseNode*> findAll(std::string_view name) const;
/**
* @brief Amount of nodes in document order carrying the given tag.
*/
isize count(std::string_view name) const;
/**
* @brief True when at least one node carries the given tag.
*/
bool contains(std::string_view name) const { return find(name) != nullptr; }
public:
/**
* @brief Human readable XML-like rendering of this node and its children.
*/
std::string toString() const;
/**
* @brief XML-like rendering of this node alone, indented by the given depth.
*/
std::string toString(isize depth) const;
private:
std::string describe(isize depth) const;
};
/**
* @brief Owner of a parsed token tree, exposing a document like interface.
* @details Use parse() to turn source text into an inspectable tree. The tree
* is materialized once and then only read, so any number of consumers
* can walk the same nodes safely.
*/
class ParseTree {
friend class ParseNode;
private:
deque<ParseNode> nodes;
isize root_index = ParseNode::npos;
private:
ParseNode* resolve(isize index);
const ParseNode* resolve(isize index) const;
isize addNode(const TokenResult& result);
void linkChildren(isize parent, const vector<isize>& children);
/**
* @brief Materializes a parse result, dropping the grammar scaffolding.
* @details Only tagged rules become nodes, so a consumer walks syntax and not
* combinators: untagged rules are pure plumbing and simply hoist their
* children upwards, and empty shells are discarded. Returns the top
* level nodes produced by this subtree.
*/
vector<isize> collapse(const TokenResult& result);
public:
ParseTree() = default;
~ParseTree() = default;
// Nodes point back at their owning tree, so trees are never copied or moved.
ParseTree(const ParseTree&) = delete;
ParseTree& operator=(const ParseTree&) = delete;
ParseTree(ParseTree&&) = delete;
ParseTree& operator=(ParseTree&&) = delete;
public:
/**
* @brief Parses source text, requiring the grammar to consume all of it.
* @returns True on success, in which case the tree holds the parsed nodes.
*/
bool parse(const Token* token, const std::string& source);
/**
* @brief Parses from a reader, requiring the grammar to consume all of it.
* @returns True on success, in which case the tree holds the parsed nodes.
*/
bool parse(const Token* token, TextReader& reader);
/**
* @brief Parses the longest matching prefix of a reader.
* @returns True when at least something was matched.
*/
bool parsePrefix(const Token* token, TextReader& reader);
/**
* @brief Adopts an already produced parse result.
*/
void build(const TokenResult& result);
public:
ParseNode* root() { return resolve(root_index); }
const ParseNode* root() const { return resolve(root_index); }
bool empty() const { return root_index == ParseNode::npos; }
public:
ParseNode* find(std::string_view name);
const ParseNode* find(std::string_view name) const;
vector<ParseNode*> findAll(std::string_view name);
vector<const ParseNode*> findAll(std::string_view name) const;
isize count(std::string_view name) const;
/**
* @brief Human readable XML-like rendering of the whole tree.
*/
std::string toString() const;
};
}
+10 -5
View File
@@ -83,9 +83,11 @@ namespace spider {
return t;
}
std::u32string TokenResult::flatMatch() {
std::u32string TokenResult::flatMatch() const {
if (folded) return match;
std::u32string s = match;
for(auto c : child) s += c.flatMatch();
for (const auto& c : child) s += c.flatMatch();
return s;
}
@@ -244,12 +246,15 @@ namespace spider {
auto r = target->test(ctx);
if (!r.success) return r;
// Set tag of this result
r.tag = tag_name;
// The innermost tag wins, so wrapping a rule that is already tagged, as in
// a choice of tagged rules, never hides what was actually matched.
if (!r.tag.has_value()) r.tag = tag_name;
if (flatten) {
// Fold the subtree text into this node, but keep the children so the
// parsed result stays a walkable tree instead of a flat string.
r.match = r.flatMatch();
r.child.clear();
r.folded = true;
}
return r;
+11 -1
View File
@@ -25,9 +25,19 @@ namespace spider {
vector<TokenResult> child = {};
/**
* @brief Set when match already holds the folded text of the whole subtree.
* @details Tagged rules that flatten their children keep those children around
* so the parsed tree stays walkable, and flag the folded text here.
*/
bool folded = false;
public:
std::u32string flatMatch();
/**
* @brief The full matched text of this node and of all of its children.
*/
std::u32string flatMatch() const;
};