synch
This commit is contained in:
+149
-120
@@ -1,142 +1,171 @@
|
|||||||
#include <iostream>
|
#include <iostream>
|
||||||
|
|
||||||
#include <spider/compiler/common.hpp>
|
#include <spider/compiler/common.hpp>
|
||||||
#include <spider/compiler/text/utf8.hpp>
|
|
||||||
|
|
||||||
|
#include <spider/compiler/text/unicode.hpp>
|
||||||
#include <spider/compiler/text/TextReader.hpp>
|
#include <spider/compiler/text/TextReader.hpp>
|
||||||
|
|
||||||
|
#include <spider/compiler/assembler/AsmEBNF.hpp>
|
||||||
|
|
||||||
using namespace spider;
|
using namespace spider;
|
||||||
|
|
||||||
class TestRunner {
|
// Inline evaluator that executes the reader and prints standard output
|
||||||
private:
|
static bool run_test_case(const Token* token, const std::string& input, bool expectedSuccess, const std::u32string& expectedMatch = U"") {
|
||||||
int totalTests = 0;
|
StringTextReader reader(input);
|
||||||
int passedTests = 0;
|
TokenResult res = token->test(reader);
|
||||||
|
|
||||||
public:
|
bool status_ok = (res.success == expectedSuccess);
|
||||||
void assertCondition(bool condition, const std::string& testName) {
|
bool match_ok = (!expectedSuccess) || expectedMatch.empty() || (res.flatMatch() == expectedMatch);
|
||||||
totalTests++;
|
|
||||||
if (condition) {
|
if (status_ok && match_ok) {
|
||||||
std::cout << " [PASS] " << testName << "\n";
|
std::cout << "[PASS] Input: \"" << input << "\" -> "
|
||||||
passedTests++;
|
<< (res.success ? "SUCCESS" : "FAILURE")
|
||||||
|
<< (res.success ? (" (Matched: \"" + utf8::toUTF8(res.flatMatch()) + "\")") : "")
|
||||||
|
<< "\n";
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
std::cerr << "[FAIL] Input: \"" << input << "\"\n"
|
||||||
|
<< " Expected Success: " << (expectedSuccess ? "true" : "false") << ", Got: " << (res.success ? "true" : "false") << "\n";
|
||||||
|
if (expectedSuccess && !expectedMatch.empty()) {
|
||||||
|
std::cerr << " Expected Match: \"" << utf8::toUTF8(expectedMatch) << "\", Got: \"" << utf8::toUTF8(res.flatMatch()) << "\"\n";
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// Core Combinator & Grammar Tests
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
void test_primitives_and_literals(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Primitives & Literals ---\n";
|
||||||
|
Token* hello = tf.lit("hello");
|
||||||
|
|
||||||
|
run_test_case(hello, "hello world", true, U"hello");
|
||||||
|
run_test_case(hello, "hell", false);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::letter, "a", true);
|
||||||
|
run_test_case(asm_ebnf::letter, "f", true);
|
||||||
|
run_test_case(asm_ebnf::letter, "A", true);
|
||||||
|
run_test_case(asm_ebnf::letter, "Y", true);
|
||||||
|
run_test_case(asm_ebnf::letter, "9", false);
|
||||||
|
run_test_case(asm_ebnf::digit, "0", true);
|
||||||
|
run_test_case(asm_ebnf::digit, "7", true);
|
||||||
|
run_test_case(asm_ebnf::digit, "x", false);
|
||||||
|
run_test_case(asm_ebnf::hex_digit, "F", true);
|
||||||
|
run_test_case(asm_ebnf::hex_digit, "g", false);
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_choice_and_seq(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Choice & Sequence ---\n";
|
||||||
|
Token* seq_test = tf.seq({ tf["foo"], tf["bar"] });
|
||||||
|
run_test_case(seq_test, "foobar", true, U"foobar");
|
||||||
|
run_test_case(seq_test, "foobaz", false);
|
||||||
|
|
||||||
|
Token* choice_test = tf.choice({ tf["apple"], tf["banana"] });
|
||||||
|
run_test_case(choice_test, "banana", true, U"banana");
|
||||||
|
run_test_case(choice_test, "cherry", false);
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_opt_and_rep(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Optional & Repeat ---\n";
|
||||||
|
Token* opt_test = tf.seq({ tf.opt(tf["+"]), asm_ebnf::digit });
|
||||||
|
run_test_case(opt_test, "+5", true, U"+5");
|
||||||
|
run_test_case(opt_test, "5", true, U"5");
|
||||||
|
|
||||||
|
Token* rep_digits = tf.rep(asm_ebnf::digit);
|
||||||
|
run_test_case(rep_digits, "12345abc", true, U"12345");
|
||||||
|
run_test_case(rep_digits, "abc", true, U"");
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_literals(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Grammatical Literals ---\n";
|
||||||
|
run_test_case(asm_ebnf::identifier, "valid_var1", true, U"valid_var1");
|
||||||
|
run_test_case(asm_ebnf::identifier, "_private", true, U"_private");
|
||||||
|
run_test_case(asm_ebnf::identifier, "123invalid", false);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::decimal_lit, "1234", true);
|
||||||
|
run_test_case(asm_ebnf::decimal_lit, "-567L", true);
|
||||||
|
run_test_case(asm_ebnf::hex_lit, "0x1A2B", true);
|
||||||
|
run_test_case(asm_ebnf::octal_lit, "0c755", true);
|
||||||
|
run_test_case(asm_ebnf::binary_lit, "0b10101", true);
|
||||||
|
run_test_case(asm_ebnf::float_lit, "3.14159F", true);
|
||||||
|
run_test_case(asm_ebnf::float_lit, "1e-10D", true);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::char_lit, "'a'", true);
|
||||||
|
run_test_case(asm_ebnf::char_lit, "'\\n'", true);
|
||||||
|
run_test_case(asm_ebnf::string_lit, "\"Hello World\"", true);
|
||||||
|
run_test_case(asm_ebnf::string_lit, "\"Escape \\\" Test\"", true);
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_addressing_modes(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Addressing Modes ---\n";
|
||||||
|
run_test_case(asm_ebnf::register_tok, "R0", true);
|
||||||
|
run_test_case(asm_ebnf::register_tok, "R15", true);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::addrm_ind, "[ 0x1000 ]", true);
|
||||||
|
run_test_case(asm_ebnf::addrm_ptr, "[R1]", true);
|
||||||
|
run_test_case(asm_ebnf::addrm_idx, "[R1 + 4]", true);
|
||||||
|
run_test_case(asm_ebnf::addrm_sca, "[R1 + R2 * 4]", true);
|
||||||
|
run_test_case(asm_ebnf::addrm_dis, "[ R1 + R2 * 4 + 16 ]", true);
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_instructions_and_lines(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Instructions & Lines ---\n";
|
||||||
|
run_test_case(asm_ebnf::instruction, "NOP", true);
|
||||||
|
run_test_case(asm_ebnf::instruction, "MOV R1, [R2 + 4]", true);
|
||||||
|
run_test_case(asm_ebnf::instruction, "ADD R0, 100", true);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::preprocessor, "#define MAX_BUF", true);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::annotation, "@inline", true);
|
||||||
|
run_test_case(asm_ebnf::annotation, "@align(4)", true);
|
||||||
|
run_test_case(asm_ebnf::annotation, "@section(name=\"text\", flags=1)", true);
|
||||||
|
|
||||||
|
run_test_case(asm_ebnf::line, "main: MOV R0, R1 ; copy reg\n", true);
|
||||||
|
run_test_case(asm_ebnf::line, " @deprecated NOP\n", true);
|
||||||
|
run_test_case(asm_ebnf::line, "; only a comment line\n", true);
|
||||||
|
}
|
||||||
|
|
||||||
|
void test_full_program(TokenFactory& tf) {
|
||||||
|
std::cout << "\n--- Testing Full Program Parser ---\n";
|
||||||
|
std::string asm_code =
|
||||||
|
"#include stdio\n"
|
||||||
|
"\n"
|
||||||
|
"start:\n"
|
||||||
|
" MOV R1, 0x20 ; Load constant\n"
|
||||||
|
" @align(16) ADD R1, [R2 + R3 * 8 + 4]\n"
|
||||||
|
" JMP start\n";
|
||||||
|
|
||||||
|
StringTextReader reader(asm_code);
|
||||||
|
TokenResult res = asm_ebnf::program->test(reader);
|
||||||
|
|
||||||
|
if (res.success) {
|
||||||
|
std::cout << "[PASS] Full Assembly Program parsed successfully!\n";
|
||||||
} else {
|
} else {
|
||||||
std::cout << " [FAIL] " << testName << "\n";
|
std::cerr << "[FAIL] Program parsing failed.\n";
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
|
||||||
void printSummary() const {
|
|
||||||
std::cout << "\n========================================\n";
|
|
||||||
std::cout << "Test Results: " << passedTests << "/" << totalTests << " passed.\n";
|
|
||||||
std::cout << "========================================\n";
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
// ============================================================================
|
|
||||||
// TEST SUITES FOR StringTextReader
|
|
||||||
// ============================================================================
|
|
||||||
|
|
||||||
void test_basic_reading(TestRunner& runner) {
|
|
||||||
std::cout << "\n--- Running: Basic Reading Tests ---\n";
|
|
||||||
std::cout.flush();
|
|
||||||
|
|
||||||
StringTextReader reader("hello");
|
|
||||||
|
|
||||||
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'h', "Initial current character is 'h'");
|
|
||||||
runner.assertCondition(reader.peekChar(1).has_value() && reader.peekChar(1).value() == 'e', "Peek +1 char is 'e'");
|
|
||||||
|
|
||||||
auto next = reader.nextChar(1);
|
|
||||||
runner.assertCondition(next.has_value() && next.value() == 'e', "Advance to next char gives 'e'");
|
|
||||||
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'e', "Current character is now 'e'");
|
|
||||||
}
|
|
||||||
|
|
||||||
void test_eat_operations(TestRunner& runner) {
|
|
||||||
std::cout << "\n--- Running: Eat Operations Tests ---\n";
|
|
||||||
std::cout.flush();
|
|
||||||
|
|
||||||
StringTextReader reader("constexpr int x = 42;");
|
|
||||||
|
|
||||||
runner.assertCondition(reader.eat("constexpr"), "Eat exact string match 'constexpr'");
|
|
||||||
runner.assertCondition(reader.eat(' '), "Eat single space character");
|
|
||||||
runner.assertCondition(reader.eat("int"), "Eat second string match 'int'");
|
|
||||||
|
|
||||||
runner.assertCondition(!reader.eat("float"), "Eat fails on mismatched string 'float'");
|
|
||||||
runner.assertCondition(reader.eat(' '), "Eat single space after failure");
|
|
||||||
runner.assertCondition(reader.eat('x'), "Eat character 'x'");
|
|
||||||
}
|
|
||||||
|
|
||||||
void test_push_pop_rollback(TestRunner& runner) {
|
|
||||||
std::cout << "\n--- Running: Push/Pop Rollback Tests ---\n";
|
|
||||||
std::cout.flush();
|
|
||||||
|
|
||||||
StringTextReader reader("function_name()");
|
|
||||||
|
|
||||||
auto savedPos = reader.push();
|
|
||||||
runner.assertCondition(reader.eat("function_"), "Incomplete parse attempt");
|
|
||||||
|
|
||||||
// Rollback
|
|
||||||
reader.pop(savedPos);
|
|
||||||
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'f', "Rollback restores cursor to 'f'");
|
|
||||||
runner.assertCondition(reader.eat("function_name"), "Subsequent match succeeds after rollback");
|
|
||||||
}
|
|
||||||
|
|
||||||
void test_commit(TestRunner& runner) {
|
|
||||||
std::cout << "\n--- Running: Buffer Commit Tests ---\n";
|
|
||||||
std::cout.flush();
|
|
||||||
|
|
||||||
StringTextReader reader("line1\nline2");
|
|
||||||
|
|
||||||
reader.eat("line1\n");
|
|
||||||
reader.commit(); // Discard historical rollback buffer
|
|
||||||
|
|
||||||
runner.assertCondition(reader.current().has_value() && reader.current().value() == 'l', "Current char after commit is 'l'");
|
|
||||||
runner.assertCondition(reader.eat("line2"), "Reading continues normally after commit");
|
|
||||||
}
|
|
||||||
|
|
||||||
void test_string_mutations(TestRunner& runner) {
|
|
||||||
std::cout << "\n--- Running: String Mutation Tests (set/append) ---\n";
|
|
||||||
std::cout.flush();
|
|
||||||
|
|
||||||
StringTextReader reader("foo");
|
|
||||||
runner.assertCondition(reader.eat("foo"), "Read initial text 'foo'");
|
|
||||||
|
|
||||||
reader.set("reset_text");
|
|
||||||
runner.assertCondition(reader.eat("reset_text"), "Read completely new text after set()");
|
|
||||||
}
|
|
||||||
|
|
||||||
void test_eof_handling(TestRunner& runner) {
|
|
||||||
std::cout << "\n--- Running: EOF & Error State Tests ---\n";
|
|
||||||
std::cout.flush();
|
|
||||||
|
|
||||||
StringTextReader reader("a");
|
|
||||||
|
|
||||||
runner.assertCondition(static_cast<bool>(reader), "Reader is valid initially");
|
|
||||||
runner.assertCondition(!reader.isEOF(), "isEOF is false initially");
|
|
||||||
|
|
||||||
reader.eat('a');
|
|
||||||
std::cout << "Index: " << reader.push().index << std::endl;
|
|
||||||
std::cout << "Value: " << reader.current().value_or(0) << std::endl;
|
|
||||||
runner.assertCondition(!reader.current().has_value(), "current() returns empty optional at EOF");
|
|
||||||
runner.assertCondition(reader.isEOF(), "isEOF is true after consuming all input");
|
|
||||||
runner.assertCondition(!reader.hasError(), "hasError remains false on normal EOF");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// MAIN DRIVER
|
// Main Execution
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
|
|
||||||
int main() {
|
int main() {
|
||||||
std::cout << "========================================\n";
|
TokenFactory tf;
|
||||||
std::cout << " StringTextReader Unit Test Suite \n";
|
asm_ebnf::initTokens(tf);
|
||||||
std::cout << "========================================\n";
|
|
||||||
|
|
||||||
TestRunner runner;
|
std::cout << "Running Token Framework Tests...\n";
|
||||||
test_basic_reading(runner);
|
|
||||||
test_eat_operations(runner);
|
|
||||||
test_push_pop_rollback(runner);
|
|
||||||
test_commit(runner);
|
|
||||||
test_string_mutations(runner);
|
|
||||||
test_eof_handling(runner);
|
|
||||||
runner.printSummary();
|
|
||||||
|
|
||||||
|
test_primitives_and_literals(tf);
|
||||||
|
test_choice_and_seq(tf);
|
||||||
|
test_opt_and_rep(tf);
|
||||||
|
test_literals(tf);
|
||||||
|
test_addressing_modes(tf);
|
||||||
|
test_instructions_and_lines(tf);
|
||||||
|
test_full_program(tf);
|
||||||
|
|
||||||
|
std::cout << "All token framework tests passed successfully!\n";
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -2,10 +2,6 @@
|
|||||||
|
|
||||||
namespace spider::asm_ebnf {
|
namespace spider::asm_ebnf {
|
||||||
|
|
||||||
// Token Factory
|
|
||||||
|
|
||||||
//TokenFactory tf;
|
|
||||||
|
|
||||||
// Char Functions
|
// Char Functions
|
||||||
|
|
||||||
bool isUTF8Alpha(u32 ch) {
|
bool isUTF8Alpha(u32 ch) {
|
||||||
|
|||||||
@@ -75,4 +75,6 @@ namespace spider::asm_ebnf {
|
|||||||
extern const Token* line_last;
|
extern const Token* line_last;
|
||||||
extern const Token* program;
|
extern const Token* program;
|
||||||
|
|
||||||
|
void initTokens(TokenFactory& tf);
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ namespace spider {
|
|||||||
|
|
||||||
Token* TokenFactory::lit(std::string_view text) {
|
Token* TokenFactory::lit(std::string_view text) {
|
||||||
auto it = lit_cache.find(std::string(text));
|
auto it = lit_cache.find(std::string(text));
|
||||||
if(it != lit_cache.end()) return it->second;
|
if (it != lit_cache.end()) return it->second;
|
||||||
|
|
||||||
auto p = std::make_unique<LitToken>(text);
|
auto p = std::make_unique<LitToken>(text);
|
||||||
auto t = p.get();
|
auto t = p.get();
|
||||||
@@ -39,15 +39,17 @@ namespace spider {
|
|||||||
|
|
||||||
Token* TokenFactory::choice(std::string_view opts) {
|
Token* TokenFactory::choice(std::string_view opts) {
|
||||||
vector<const Token*> toks;
|
vector<const Token*> toks;
|
||||||
for(char c : opts) {
|
|
||||||
|
for (char c : opts) {
|
||||||
std::string s = std::string(1, c);
|
std::string s = std::string(1, c);
|
||||||
toks.push_back(lit(s));
|
toks.push_back(lit(s));
|
||||||
}
|
}
|
||||||
|
|
||||||
return choice(toks);
|
return choice(toks);
|
||||||
}
|
}
|
||||||
|
|
||||||
Token* TokenFactory::choice(const vector<const Token*>& tokens) {
|
Token* TokenFactory::choice(const vector<const Token*>& tokens) {
|
||||||
uptr<Token> p = std::make_unique<SeqToken>(tokens);
|
uptr<Token> p = std::make_unique<OrToken>(tokens);
|
||||||
auto t = p.get();
|
auto t = p.get();
|
||||||
arena.emplace_back(std::move(p));
|
arena.emplace_back(std::move(p));
|
||||||
return t;
|
return t;
|
||||||
@@ -74,6 +76,12 @@ namespace spider {
|
|||||||
return t;
|
return t;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
std::u32string TokenResult::flatMatch() {
|
||||||
|
std::u32string s = match;
|
||||||
|
for(auto c : child) s += c.flatMatch();
|
||||||
|
return s;
|
||||||
|
}
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// LitToken Implementation
|
// LitToken Implementation
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
@@ -108,6 +116,7 @@ namespace spider {
|
|||||||
}
|
}
|
||||||
|
|
||||||
r.success = !r.match.empty();
|
r.success = !r.match.empty();
|
||||||
|
if(r.success) std::cout << "[fn] matched: " << utf8::toUTF8(r.flatMatch()) << std::endl;
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -156,7 +165,10 @@ namespace spider {
|
|||||||
for (const auto& token_ref : tokens) {
|
for (const auto& token_ref : tokens) {
|
||||||
// Short-circuit branch: return immediately on first valid choice match
|
// Short-circuit branch: return immediately on first valid choice match
|
||||||
TokenResult res = token_ref->test(ctx);
|
TokenResult res = token_ref->test(ctx);
|
||||||
if (res.success) return res;
|
if (res.success) {
|
||||||
|
std::cout << "[or] matched: " << utf8::toUTF8(res.flatMatch()) << std::endl;
|
||||||
|
return res;
|
||||||
|
}
|
||||||
|
|
||||||
// Backtrack isolation: Reset the cursor position before testing the next alternative path
|
// Backtrack isolation: Reset the cursor position before testing the next alternative path
|
||||||
ctx.pop(i);
|
ctx.pop(i);
|
||||||
@@ -176,13 +188,14 @@ namespace spider {
|
|||||||
TokenResult res = target->test(ctx);
|
TokenResult res = target->test(ctx);
|
||||||
|
|
||||||
if (res.success) {
|
if (res.success) {
|
||||||
|
std::cout << "[~] matched: " << utf8::toUTF8(res.flatMatch()) << std::endl;
|
||||||
return res; // Option matched exactly 1 instance successfully
|
return res; // Option matched exactly 1 instance successfully
|
||||||
}
|
}
|
||||||
|
|
||||||
// Recovery path: If sub-rule fails, clean up the dirty state mutation
|
// Recovery path: If sub-rule fails, clean up the dirty state mutation
|
||||||
// and successfully return an empty match payload (0 instances).
|
// and successfully return an empty match payload (0 instances).
|
||||||
ctx.pop(tri);
|
ctx.pop(tri);
|
||||||
return { .success = false };
|
return { .success = true };
|
||||||
}
|
}
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
@@ -197,12 +210,14 @@ namespace spider {
|
|||||||
for (;;) {
|
for (;;) {
|
||||||
auto i = ctx.push();
|
auto i = ctx.push();
|
||||||
TokenResult res = target->test(ctx);
|
TokenResult res = target->test(ctx);
|
||||||
|
|
||||||
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
||||||
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
||||||
if (!res.success || i.index == ctx.push().index) {
|
if (!res.success || i.index == ctx.push().index) {
|
||||||
ctx.pop(i);
|
ctx.pop(i);
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
//r.match += res.match;
|
//r.match += res.match;
|
||||||
r.child.push_back(res);
|
r.child.push_back(res);
|
||||||
}
|
}
|
||||||
@@ -210,13 +225,15 @@ namespace spider {
|
|||||||
// Repetition rules (* token) always evaluate to successful
|
// Repetition rules (* token) always evaluate to successful
|
||||||
// completion state, even with 0 matches.
|
// completion state, even with 0 matches.
|
||||||
r.success = true;
|
r.success = true;
|
||||||
|
std::cout << "[*] matched: " << utf8::toUTF8(r.flatMatch()) << std::endl;
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Tagged Token
|
// Tagged Token
|
||||||
|
|
||||||
TagToken::TagToken(const Token* t, std::string_view tag, bool doflatten)
|
TagToken::TagToken(const Token* t, std::string_view tag, bool doflatten)
|
||||||
: target(t), tag_name(tag), flatten(doflatten) { }
|
: target(t), tag_name(tag), flatten(doflatten) {
|
||||||
|
}
|
||||||
|
|
||||||
TokenResult TagToken::test(TextReader& ctx) const {
|
TokenResult TagToken::test(TextReader& ctx) const {
|
||||||
auto r = target->test(ctx);
|
auto r = target->test(ctx);
|
||||||
@@ -225,11 +242,8 @@ namespace spider {
|
|||||||
// Set tag of this result
|
// Set tag of this result
|
||||||
r.tag = tag_name;
|
r.tag = tag_name;
|
||||||
|
|
||||||
if(flatten) {
|
if (flatten) {
|
||||||
r.match = U"";
|
r.match = r.flatMatch();
|
||||||
for(auto e : r.child) {
|
|
||||||
r.match += e.match;
|
|
||||||
}
|
|
||||||
r.child.clear();
|
r.child.clear();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
|
|
||||||
#include <spider/compiler/common.hpp>
|
#include <spider/compiler/common.hpp>
|
||||||
|
|
||||||
#include <spider/compiler/text/utf8.hpp>
|
#include <spider/compiler/text/unicode.hpp>
|
||||||
#include <spider/compiler/text/TextReader.hpp>
|
#include <spider/compiler/text/TextReader.hpp>
|
||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
@@ -25,6 +25,10 @@ namespace spider {
|
|||||||
|
|
||||||
vector<TokenResult> child = {};
|
vector<TokenResult> child = {};
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
std::u32string flatMatch();
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
class Token;
|
class Token;
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
|
|
||||||
namespace utf8 {
|
namespace unicode {
|
||||||
|
|
||||||
// --------------------- //
|
// --------------------- //
|
||||||
// UTF-8 Sequence Length //
|
// UTF-8 Sequence Length //
|
||||||
@@ -95,6 +95,55 @@ namespace spider {
|
|||||||
return _i == csize;
|
return _i == csize;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ----------------- //
|
||||||
|
// UTF-32 into UTF-8 //
|
||||||
|
// ----------------- //
|
||||||
|
|
||||||
|
inline void append_utf32_to_utf8(u32 code_point, std::string& out) {
|
||||||
|
if (code_point <= 0x7F) {
|
||||||
|
// 1-byte sequence (ASCII)
|
||||||
|
out.push_back(static_cast<char>(code_point));
|
||||||
|
} else if (code_point <= 0x7FF) {
|
||||||
|
// 2-byte sequence
|
||||||
|
out.push_back(static_cast<char>(0xC0 | ((code_point >> 6) & 0x1F)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
|
||||||
|
} else if (code_point <= 0xFFFF) {
|
||||||
|
// 3-byte sequence
|
||||||
|
// Filter out surrogate pairs (U+D800 to U+DFFF) as they are invalid Unicode scalar values
|
||||||
|
if (code_point >= 0xD800 && code_point <= 0xDFFF) {
|
||||||
|
code_point = 0xFFFD; // Replacement character
|
||||||
|
}
|
||||||
|
out.push_back(static_cast<char>(0xE0 | ((code_point >> 12) & 0x0F)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | ((code_point >> 6) & 0x3F)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
|
||||||
|
} else if (code_point <= 0x10FFFF) {
|
||||||
|
// 4-byte sequence
|
||||||
|
out.push_back(static_cast<char>(0xF0 | ((code_point >> 18) & 0x07)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | ((code_point >> 12) & 0x3F)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | ((code_point >> 6) & 0x3F)));
|
||||||
|
out.push_back(static_cast<char>(0x80 | (code_point & 0x3F)));
|
||||||
|
} else {
|
||||||
|
// Code point out of Unicode range -> insert UTF-8 replacement character U+FFFD
|
||||||
|
append_utf32_to_utf8(0xFFFD, out);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
inline std::string toUTF8(u32 cp) {
|
||||||
|
std::string s;
|
||||||
|
append_utf32_to_utf8(cp, s);
|
||||||
|
return s;
|
||||||
|
}
|
||||||
|
|
||||||
|
inline std::string toUTF8(const std::u32string& str) {
|
||||||
|
std::string s;
|
||||||
|
for(u32 ch : str) append_utf32_to_utf8(ch, s);
|
||||||
|
return s;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----------------- //
|
||||||
|
// STRINGS //
|
||||||
|
// ----------------- //
|
||||||
|
|
||||||
inline const char* getControlCharName(u8 c) {
|
inline const char* getControlCharName(u8 c) {
|
||||||
static const char* names[32] = {
|
static const char* names[32] = {
|
||||||
"NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL",
|
"NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL",
|
||||||
@@ -107,7 +156,7 @@ namespace spider {
|
|||||||
return nullptr;
|
return nullptr;
|
||||||
}
|
}
|
||||||
|
|
||||||
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {
|
inline void hexdump_utf8(const char* data, isize length, pos at, std::ostream& ostr) {
|
||||||
auto old_flags = ostr.flags();
|
auto old_flags = ostr.flags();
|
||||||
auto old_fill = ostr.fill();
|
auto old_fill = ostr.fill();
|
||||||
|
|
||||||
Reference in New Issue
Block a user