utf-8 hexdump implmented
This commit is contained in:
@@ -7,6 +7,56 @@ namespace spider {
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
// Test runner helper
|
||||||
|
void run_test(const std::string& name, const std::string& input) {
|
||||||
|
std::cout << "========================================\n";
|
||||||
|
std::cout << " TEST: " << name << "\n";
|
||||||
|
std::cout << "========================================\n";
|
||||||
|
|
||||||
|
spider::pos tracking_pos;
|
||||||
|
spider::utf8::hexdump(input.data(), input.size(), tracking_pos, std::cout);
|
||||||
|
std::cout << "\n";
|
||||||
|
}
|
||||||
|
|
||||||
|
void utf8sequences() {
|
||||||
|
// Permutation 1: Standard, valid mixed UTF-8 sequences (1, 2, 3, and 4 bytes)
|
||||||
|
// - 'A' -> 1 byte (U+0041)
|
||||||
|
// - '¢' (cents) -> 2 bytes (U+00A2)
|
||||||
|
// - '€' (euro) -> 3 bytes (U+20AC)
|
||||||
|
// - '𐍈' (gothic) -> 4 bytes (U+10348)
|
||||||
|
run_test("Valid Mixed Length Sequences", "A\xC2\xA2\xE2\x82\xAC\xF0\x90\x8D\x88");
|
||||||
|
|
||||||
|
// Permutation 2: Embedded Control Characters
|
||||||
|
// Should display mnemonics like (HT), (LF), (CR) without breaking formatting
|
||||||
|
run_test("ASCII Control Characters", "Text\tWith\r\nNewlines");
|
||||||
|
|
||||||
|
// Permutation 3: Invalid Lead Byte
|
||||||
|
// The byte 0xFF is structurally illegal under any UTF-8 definition.
|
||||||
|
// Expected behavior: Display single byte as INVALID LEAD, shift 1 byte over.
|
||||||
|
run_test("Invalid Lead Byte (0xFF)", "ABC\xFFXYZ");
|
||||||
|
|
||||||
|
// Permutation 4: Invalid Continuation Sequence
|
||||||
|
// A 3-byte header (\xE2) where the second byte (\x00) is a bad continuation.
|
||||||
|
// Expected behavior: Show the entire sequence up to 'm' bytes, flag as INVALID SEQUENCE.
|
||||||
|
run_test("Invalid Continuation Structure", std::string("Before \xE2\x00\xAC After", 16));
|
||||||
|
|
||||||
|
// Permutation 5: Truncated Sequence at End-of-Buffer
|
||||||
|
// A 4-byte emoji header (\xF0\x9F) but the string completely cuts off.
|
||||||
|
// Expected behavior: Display remaining space placeholders as '??' -> TRUNCATED SEQUENCE.
|
||||||
|
run_test("Truncated Sequence (Missing trailing bytes)", "Hello \xF0\x9F");
|
||||||
|
|
||||||
|
// Permutation 6: Overlong Encoding Security Vulnerability
|
||||||
|
// Attempting to write ASCII 'I' (normally 0x49) using 2 bytes: \xC1\x89
|
||||||
|
// Expected behavior: Caught by constraints checks, flagged as INVALID SEQUENCE.
|
||||||
|
run_test("Security Hack: Overlong Encoding", "Safe\xC1\x89Hack");
|
||||||
|
|
||||||
|
// Permutation 7: Out-of-bounds / Restricted Ranges
|
||||||
|
// - \xED\xA0\x80 is a UTF-16 Surrogate (U+D800)
|
||||||
|
// - \xF4\x90\x80\x80 is outside valid Unicode space (> U+10FFFF)
|
||||||
|
// Expected behavior: Flagged securely as INVALID SEQUENCE.
|
||||||
|
run_test("Security Hack: Restricted Ranges (Surrogates & Out-of-bounds)", "Surrogate: \xED\xA0\x80 MaxBounds: \xF4\x90\x80\x80");
|
||||||
}
|
}
|
||||||
|
|
||||||
int main() {
|
int main() {
|
||||||
@@ -23,5 +73,8 @@ int main() {
|
|||||||
std::cout << std::endl;
|
std::cout << std::endl;
|
||||||
|
|
||||||
std::cout << "Happy Day!" << std::endl;
|
std::cout << "Happy Day!" << std::endl;
|
||||||
|
spider::utf8::hexdump(test.data(), test.size(), spider::pos(), std::cout);
|
||||||
|
std::cout << std::endl;
|
||||||
|
utf8sequences();
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -9,6 +9,7 @@
|
|||||||
#include <memory>
|
#include <memory>
|
||||||
#include <filesystem>
|
#include <filesystem>
|
||||||
#include <set>
|
#include <set>
|
||||||
|
#include <functional>
|
||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
|
|
||||||
@@ -40,6 +41,8 @@ namespace spider {
|
|||||||
using std::optional;
|
using std::optional;
|
||||||
using std::set;
|
using std::set;
|
||||||
|
|
||||||
|
template<typename T> using ilist = std::initializer_list<T>;
|
||||||
|
template<typename T> using ref = std::reference_wrapper<T>;
|
||||||
template<typename T> using ptr = std::shared_ptr<T>;
|
template<typename T> using ptr = std::shared_ptr<T>;
|
||||||
template<typename T> using uptr = std::unique_ptr<T>;
|
template<typename T> using uptr = std::unique_ptr<T>;
|
||||||
|
|
||||||
|
|||||||
@@ -29,19 +29,21 @@ namespace spider {
|
|||||||
}
|
}
|
||||||
|
|
||||||
bool TextReader::eat(const std::string& chars) {
|
bool TextReader::eat(const std::string& chars) {
|
||||||
isize index = 0, count = 0;
|
// instead of whatever that was, convert to UTF-32
|
||||||
u32 _char;
|
// and then do an easy compare!
|
||||||
while(index < chars.length()) {
|
std::u32string str;
|
||||||
if(!utf8::charAt(chars, index, _char)) return false;
|
if(!utf8::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!");
|
||||||
if(_char != peekChar(count)) return false;
|
|
||||||
count++;
|
// compare now
|
||||||
|
isize index;
|
||||||
|
for(index = 0; index < str.size(); index++) {
|
||||||
|
if(str[index] != peekChar(index)) return false;
|
||||||
}
|
}
|
||||||
if(index == chars.length()) {
|
|
||||||
nextChar(count);
|
// success!
|
||||||
|
nextChar(index);
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
char TextReader::readByte() {
|
char TextReader::readByte() {
|
||||||
if (err) return 0;
|
if (err) return 0;
|
||||||
@@ -164,14 +166,14 @@ namespace spider {
|
|||||||
bytes[bindex] = readByte();
|
bytes[bindex] = readByte();
|
||||||
if (err) return false;
|
if (err) return false;
|
||||||
if (eof) return false;
|
if (eof) return false;
|
||||||
if (!utf8::isCont(u8(bytes[bindex]))) {
|
|
||||||
err = true;
|
|
||||||
errmsg = "Invalid continuation of UTF-8 sequence.";
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
u32 decodedChar = utf8::decodeArr(bytes, chsize);
|
u32 decodedChar;
|
||||||
|
if(!utf8::decodeArr(bytes, chsize, decodedChar)) {
|
||||||
|
err = true;
|
||||||
|
errmsg = "Invalid UTF-8 sequence.";
|
||||||
|
return false;
|
||||||
|
}
|
||||||
buffer.push_back(decodedChar);
|
buffer.push_back(decodedChar);
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,183 @@
|
|||||||
|
#include "Token.hpp"
|
||||||
|
|
||||||
|
namespace spider {
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// Token Implementation
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
SeqToken Token::operator&(const Token& tok) {
|
||||||
|
return SeqToken({ tok, *this });
|
||||||
|
}
|
||||||
|
|
||||||
|
OrToken Token::operator|(const Token& tok) {
|
||||||
|
return OrToken({ tok, *this });
|
||||||
|
}
|
||||||
|
|
||||||
|
OptToken Token::operator~() {
|
||||||
|
return OptToken(*this);
|
||||||
|
}
|
||||||
|
|
||||||
|
RepToken Token::operator*() {
|
||||||
|
return RepToken(*this);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// LitToken Implementation
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
LitToken::LitToken(std::string_view lit) {
|
||||||
|
if (!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
|
||||||
|
}
|
||||||
|
|
||||||
|
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
|
||||||
|
|
||||||
|
TokenResult LitToken::test(TokenContext& ctx) const {
|
||||||
|
// Safety check: Prevent out-of-bounds pointer slicing if the remaining
|
||||||
|
// input is smaller than the target literal.
|
||||||
|
if (ctx.cursor + literal.size() > ctx.input.size()) {
|
||||||
|
return { false, {} };
|
||||||
|
}
|
||||||
|
|
||||||
|
// Window extract optimization: Acquire a zero-copy view over the input segment
|
||||||
|
std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size());
|
||||||
|
|
||||||
|
// Direct lexicographical verification of the UTF-32 code-point sequence
|
||||||
|
if (sub == literal) {
|
||||||
|
ctx.advance(literal.size());
|
||||||
|
return { true, std::u32string(sub) }; // Deep copy payload returned per requirement
|
||||||
|
}
|
||||||
|
|
||||||
|
return { false, {} };
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// SeqToken Implementation
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
SeqToken::SeqToken(const ilist<ref<const Token>>& list) : tokens(list) {}
|
||||||
|
|
||||||
|
TokenResult SeqToken::test(TokenContext& ctx) const {
|
||||||
|
const size_t transactional_fallback_pos = ctx.cursor;
|
||||||
|
std::u32string accumulated_match;
|
||||||
|
|
||||||
|
// All matching steps within a sequence must pass consecutively.
|
||||||
|
for (const auto& token_ref : tokens) {
|
||||||
|
TokenResult res = token_ref.get().test(ctx);
|
||||||
|
|
||||||
|
if (!res.success) {
|
||||||
|
// Strict ACID Transaction: Roll back context pointer entirely
|
||||||
|
// if any nested condition in the sequence fails.
|
||||||
|
ctx.cursor = transactional_fallback_pos;
|
||||||
|
return { false, {} };
|
||||||
|
}
|
||||||
|
|
||||||
|
// Piecewise accumulation of individual matching sub-tokens
|
||||||
|
accumulated_match += res.match;
|
||||||
|
}
|
||||||
|
|
||||||
|
return { true, accumulated_match };
|
||||||
|
}
|
||||||
|
|
||||||
|
SeqToken SeqToken::operator&(const Token& tok) {
|
||||||
|
// Intrusive chaining optimization: Appends the next token directly into the existing
|
||||||
|
// registry vector instead of nesting structures, keeping the layout flattened.
|
||||||
|
tokens.push_back(std::cref(tok));
|
||||||
|
return *this;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// OrToken Implementation
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
OrToken::OrToken(const ilist<ref<const Token>>& list) : tokens(list) {}
|
||||||
|
|
||||||
|
TokenResult OrToken::test(TokenContext& ctx) const {
|
||||||
|
const size_t local_fallback_pos = ctx.cursor;
|
||||||
|
|
||||||
|
// Ordered choice evaluation: Evaluate variants sequentially.
|
||||||
|
for (const auto& token_ref : tokens) {
|
||||||
|
TokenResult res = token_ref.get().test(ctx);
|
||||||
|
|
||||||
|
if (res.success) {
|
||||||
|
return res; // Short-circuit branch: return immediately on first valid choice match
|
||||||
|
}
|
||||||
|
|
||||||
|
// Backtrack isolation: Reset the cursor position before testing the next alternative path
|
||||||
|
ctx.cursor = local_fallback_pos;
|
||||||
|
}
|
||||||
|
|
||||||
|
return { false, {} };
|
||||||
|
}
|
||||||
|
|
||||||
|
OrToken OrToken::operator|(const Token& tok) {
|
||||||
|
// Intrusive grouping layout optimization:
|
||||||
|
// flattens alternative tokens at code evaluation time.
|
||||||
|
tokens.push_back(tok);
|
||||||
|
return *this;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// OptToken Implementation
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
OptToken::OptToken(const Token& t) : target(t) {}
|
||||||
|
|
||||||
|
TokenResult OptToken::test(TokenContext& ctx) const {
|
||||||
|
const size_t local_fallback_pos = ctx.cursor;
|
||||||
|
TokenResult res = target.test(ctx);
|
||||||
|
|
||||||
|
if (res.success) {
|
||||||
|
return res; // Option matched exactly 1 instance successfully
|
||||||
|
}
|
||||||
|
|
||||||
|
// Recovery path: If sub-rule fails, clean up the dirty state mutation
|
||||||
|
// and successfully return an empty match payload (0 instances).
|
||||||
|
ctx.cursor = local_fallback_pos;
|
||||||
|
return { true, U"" };
|
||||||
|
}
|
||||||
|
|
||||||
|
OptToken OptToken::operator~() {
|
||||||
|
// Redundant layer trap protection: returning self
|
||||||
|
// prevents wrapping an Optional in an Optional
|
||||||
|
return *this;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// RepToken Implementation
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
RepToken::RepToken(const Token& t) : target(t) {}
|
||||||
|
|
||||||
|
TokenResult RepToken::test(TokenContext& ctx) const {
|
||||||
|
std::u32string accumulated_match;
|
||||||
|
|
||||||
|
// Greedily consume matches while input stream headroom remains
|
||||||
|
while (ctx.has_more()) {
|
||||||
|
const size_t pre_loop_cursor = ctx.cursor;
|
||||||
|
TokenResult res = target.test(ctx);
|
||||||
|
|
||||||
|
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
||||||
|
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
||||||
|
if (!res.success || ctx.cursor == pre_loop_cursor) {
|
||||||
|
ctx.cursor = pre_loop_cursor; // Revert cursor to last healthy match checkpoint
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
accumulated_match += res.match;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Repetition rules (* token) always evaluate to successful completion state, even with 0 matches.
|
||||||
|
return { true, accumulated_match };
|
||||||
|
}
|
||||||
|
|
||||||
|
RepToken RepToken::operator*() {
|
||||||
|
// Redundant layer trap protection: returning self prevents wrapping a Repetition rule inside a Repetition rule
|
||||||
|
return *this;
|
||||||
|
}
|
||||||
|
|
||||||
|
}
|
||||||
|
|||||||
@@ -1,8 +1,8 @@
|
|||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
#include "spider/compiler/common.hpp"
|
#include <spider/compiler/common.hpp>
|
||||||
|
|
||||||
#include "spider/compiler/text/utf8.hpp"
|
#include <spider/compiler/text/utf8.hpp>
|
||||||
|
|
||||||
namespace spider {
|
namespace spider {
|
||||||
|
|
||||||
@@ -12,146 +12,203 @@ namespace spider {
|
|||||||
size_t cursor = 0;
|
size_t cursor = 0;
|
||||||
|
|
||||||
bool has_more() const { return cursor < input.size(); }
|
bool has_more() const { return cursor < input.size(); }
|
||||||
|
|
||||||
char32_t peek() const { return input[cursor]; }
|
char32_t peek() const { return input[cursor]; }
|
||||||
|
|
||||||
void advance(size_t n = 1) { cursor += n; }
|
void advance(size_t n = 1) { cursor += n; }
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief The structural payload returned by every parsing component execution.
|
||||||
|
*/
|
||||||
struct TokenResult {
|
struct TokenResult {
|
||||||
|
|
||||||
|
/** @brief Indicates if the token composition successfully matched the input boundary. */
|
||||||
bool success;
|
bool success;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
|
||||||
|
* @note Returns empty when success is false.
|
||||||
|
*/
|
||||||
std::u32string match;
|
std::u32string match;
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Forward declarations required by the abstract interface for operator returns.
|
||||||
|
class SeqToken;
|
||||||
|
class OrToken;
|
||||||
|
class OptToken;
|
||||||
|
class RepToken;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Pure virtual base class defining the EBNF combinator node contract.
|
||||||
|
* @details Implements structural immutability for thread-safe static allocations
|
||||||
|
* while enforcing fluency using intrusive member operator overloads.
|
||||||
|
*/
|
||||||
class Token {
|
class Token {
|
||||||
public:
|
public:
|
||||||
|
|
||||||
|
/** @brief Virtual destructor ensuring clean polymorphic destruction of composite graphs. */
|
||||||
virtual ~Token() = default;
|
virtual ~Token() = default;
|
||||||
|
|
||||||
virtual TokenResult parse(TokenContext& ctx) const = 0;
|
public:
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Evaluates this token node against the provided context.
|
||||||
|
* @param ctx The current state tracker containing the UTF-32 viewing buffer and cursor.
|
||||||
|
* @return TokenResult Containing verification state and the parsed copy of matching data.
|
||||||
|
* @note Pure virtual; implementation details handle node-specific combinator semantics.
|
||||||
|
*/
|
||||||
|
virtual TokenResult test(TokenContext& ctx) const = 0;
|
||||||
|
|
||||||
|
public:
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Chains this token and another sequentially.
|
||||||
|
* @return A temporary structural bridge matching both tokens sequentially.
|
||||||
|
*/
|
||||||
|
virtual SeqToken operator&(const Token& tok);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Combines this token and another under alternation.
|
||||||
|
* @return A structural bridge matching either this token or the fallback selection.
|
||||||
|
*/
|
||||||
|
virtual OrToken operator|(const Token& tok);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Wraps this node in an optional layout rule.
|
||||||
|
* @return A structure matching zero or one instances of this current node.
|
||||||
|
*/
|
||||||
|
virtual OptToken operator~();
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Wraps this node in a repetitive loop match framework.
|
||||||
|
* @return A structure matching zero or more occurrences of this current node.
|
||||||
|
*/
|
||||||
|
virtual RepToken operator*();
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Leaf terminal parser validating precise exact matching UTF-32 sequences.
|
||||||
|
*/
|
||||||
class LitToken : public Token {
|
class LitToken : public Token {
|
||||||
private:
|
private:
|
||||||
|
/** @brief The reference literal sequence being inspected. */
|
||||||
std::u32string literal;
|
std::u32string literal;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
/**
|
||||||
|
* @brief Constructs a literal rule by transforming a standard UTF-8 string view.
|
||||||
|
* @throws std::runtime_error If incoming character boundaries contain invalid UTF-8 formatting.
|
||||||
|
*/
|
||||||
|
explicit LitToken(std::string_view lit);
|
||||||
|
|
||||||
explicit LitToken(std::string_view lit) {
|
/** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */
|
||||||
if(!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
|
explicit LitToken(std::u32string lit);
|
||||||
}
|
|
||||||
|
|
||||||
explicit LitToken(std::u32string lit) : literal(std::move(lit)) {}
|
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
/**
|
||||||
TokenResult parse(TokenContext& ctx) const override {
|
* @brief Validates match of the backing u32string exactly at the context cursor pointer.
|
||||||
if (ctx.cursor + literal.size() > ctx.input.size()) return { false, {} };
|
* @details Advances the context cursor precisely by literal length on success; zero state mutation on failure.
|
||||||
std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size());
|
*/
|
||||||
if (sub == literal) {
|
TokenResult test(TokenContext& ctx) const override;
|
||||||
ctx.advance(literal.size());
|
|
||||||
return { true, std::u32string(sub) };
|
|
||||||
}
|
|
||||||
return { false, {} };
|
|
||||||
}
|
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
// AndToken references existing static tokens rather than managing their lifecycles
|
/**
|
||||||
class AndToken : public Token {
|
* @brief Evaluates an unrolled sequence of ordered grammatical rules sequentially (AndToken).
|
||||||
|
*/
|
||||||
|
class SeqToken : public Token {
|
||||||
private:
|
private:
|
||||||
|
/**
|
||||||
const Token& lhs;
|
* @brief Internal contiguous layout registry storing lightweight, zero-overhead references.
|
||||||
const Token& rhs;
|
* @details Avoids heap allocation penalties by referencing static instances immutably.
|
||||||
|
*/
|
||||||
|
vector<ref<const Token>> tokens;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
/** @brief Constructs a sequence container directly out of an inline brace-enclosed listing. */
|
||||||
AndToken(const Token& l, const Token& r) : lhs(l), rhs(r) {}
|
SeqToken(const ilist<ref<const Token>>& list);
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
/**
|
||||||
|
* @brief Iterates the internal sequence, enforcing that every child node must pass sequentially.
|
||||||
|
* @details Implements a strict transaction boundary: if any internal element fails, the index
|
||||||
|
* backtracks entirely to its starting cursor value before returning failure.
|
||||||
|
*/
|
||||||
|
TokenResult test(TokenContext& ctx) const override;
|
||||||
|
|
||||||
TokenResult parse(TokenContext& ctx) const override {
|
/** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */
|
||||||
size_t start_pos = ctx.cursor;
|
SeqToken operator&(const Token& tok) override;
|
||||||
if (!lhs.parse(ctx).success) return { false, {} };
|
|
||||||
if (!rhs.parse(ctx).success) {
|
|
||||||
ctx.cursor = start_pos; // Backtrack
|
|
||||||
return { false, {} };
|
|
||||||
}
|
|
||||||
return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) };
|
|
||||||
}
|
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Implements standard alternation selection matching rules (OrToken).
|
||||||
|
*/
|
||||||
class OrToken : public Token {
|
class OrToken : public Token {
|
||||||
private:
|
private:
|
||||||
|
/** @brief Ordered registry of possible alternate structural paths. */
|
||||||
const Token& lhs;
|
vector<ref<const Token>> tokens;
|
||||||
const Token& rhs;
|
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
/** @brief Constructs an alternation choice layout from brace-enclosed tokens. */
|
||||||
OrToken(const Token& l, const Token& r) : lhs(l), rhs(r) {}
|
OrToken(const ilist<ref<const Token>>& list);
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
/**
|
||||||
|
* @brief Scans through alternatives, resolving immediately on the first candidate that passes.
|
||||||
|
* @details Safely rolls back changes to the context cursor point between failed alternative attempts.
|
||||||
|
*/
|
||||||
|
TokenResult test(TokenContext& ctx) const override;
|
||||||
|
|
||||||
TokenResult parse(TokenContext& ctx) const override {
|
/** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */
|
||||||
size_t start_pos = ctx.cursor;
|
OrToken operator|(const Token& tok) override;
|
||||||
auto res = lhs.parse(ctx);
|
|
||||||
if (res.success) return res;
|
|
||||||
ctx.cursor = start_pos; // Backtrack
|
|
||||||
return rhs.parse(ctx);
|
|
||||||
}
|
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Structurally represents an optional token condition rule layer (OptToken).
|
||||||
|
*/
|
||||||
class OptToken : public Token {
|
class OptToken : public Token {
|
||||||
private:
|
private:
|
||||||
|
/** @brief Read-only target node reference to test optional status against. */
|
||||||
const Token& target;
|
const Token& target;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
/** @brief Binds the target node rule structural dependency layout wrapper. */
|
||||||
explicit OptToken(const Token& t) : target(t) {}
|
explicit OptToken(const Token& t);
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
/**
|
||||||
|
* @brief Evaluates target presence. Returns success true regardless of sub-rule evaluation outcome.
|
||||||
|
* @details If the nested rule fails, the context cursor rolls back to initial state, returning empty matches.
|
||||||
|
*/
|
||||||
|
TokenResult test(TokenContext& ctx) const override;
|
||||||
|
|
||||||
TokenResult parse(TokenContext& ctx) const override {
|
/** @brief Stub override providing standard compliance with the base Token interface signature. */
|
||||||
size_t start_pos = ctx.cursor;
|
OptToken operator~() override;
|
||||||
if (target.parse(ctx).success) return {
|
|
||||||
true,
|
|
||||||
std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos))
|
|
||||||
};
|
|
||||||
ctx.cursor = start_pos;
|
|
||||||
return { true, std::u32string(ctx.input.substr(start_pos, 0)) };
|
|
||||||
}
|
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @brief Evaluates zero-to-infinite loops of a single repetitive match signature (RepToken).
|
||||||
|
*/
|
||||||
class RepToken : public Token {
|
class RepToken : public Token {
|
||||||
private:
|
private:
|
||||||
|
/** @brief The base token node sequence layer evaluated in loops. */
|
||||||
const Token& target;
|
const Token& target;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
/** @brief Binds the repeated structural blueprint node wrapper configuration. */
|
||||||
explicit RepToken(const Token& t) : target(t) {}
|
explicit RepToken(const Token& t);
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
/**
|
||||||
|
* @brief Greedily attempts to match target repeatedly until an execution failure boundary is hit.
|
||||||
|
* @details Includes internal safety loops checking cursor delta advancement to guarantee infinite
|
||||||
|
* empty-matching sub-loops do not cause thread lockups.
|
||||||
|
*/
|
||||||
|
TokenResult test(TokenContext& ctx) const override;
|
||||||
|
|
||||||
TokenResult parse(TokenContext& ctx) const override {
|
/** @brief Stub override providing standard compliance with the base Token interface signature. */
|
||||||
size_t start_pos = ctx.cursor;
|
RepToken operator*() override;
|
||||||
while (ctx.has_more()) {
|
|
||||||
size_t loop_start = ctx.cursor;
|
|
||||||
if (!target.parse(ctx).success || ctx.cursor == loop_start) {
|
|
||||||
ctx.cursor = loop_start;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) };
|
|
||||||
}
|
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -26,24 +26,13 @@ namespace spider {
|
|||||||
return (c & 0xC0) == 0x80;
|
return (c & 0xC0) == 0x80;
|
||||||
}
|
}
|
||||||
|
|
||||||
constexpr isize isValidSeq(const char* src, isize len) {
|
|
||||||
if (len == 0) return 0;
|
|
||||||
isize m = seqlen(u8(src[0]));
|
|
||||||
if (m == 0 || m > len) return 0;
|
|
||||||
for (isize i = 1; i < m; i++) {
|
|
||||||
if (!isCont(u8(src[i]))) return 0;
|
|
||||||
}
|
|
||||||
return m;
|
|
||||||
}
|
|
||||||
|
|
||||||
// ----------------- //
|
// ----------------- //
|
||||||
// UTF-8 into UTF-32 //
|
// UTF-8 into UTF-32 //
|
||||||
// ----------------- //
|
// ----------------- //
|
||||||
|
|
||||||
inline isize decode(const char* src, isize len, u32& out) {
|
inline bool decodeArr(const char* src, isize chlen, u32& out) {
|
||||||
// check input is valid
|
// Check character length
|
||||||
isize charlen = isValidSeq(src, len);
|
if (chlen < 1 || chlen > 4) return false;
|
||||||
if (charlen == 0) return 0;
|
|
||||||
|
|
||||||
// map of masks, starts at 1
|
// map of masks, starts at 1
|
||||||
static constexpr u8 firstMask[5] = {
|
static constexpr u8 firstMask[5] = {
|
||||||
@@ -54,63 +43,165 @@ namespace spider {
|
|||||||
0x07 // 11110xxx
|
0x07 // 11110xxx
|
||||||
};
|
};
|
||||||
|
|
||||||
// assemble the char
|
u32 result = u8(src[0]) & firstMask[chlen];
|
||||||
out = u8(src[0]) & firstMask[charlen];
|
|
||||||
for (isize i = 1; i < charlen; ++i) {
|
|
||||||
out <<= 6;
|
|
||||||
out |= u8(src[i]) & 0x3F;
|
|
||||||
}
|
|
||||||
return charlen;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* A simpler version, which consider it already
|
|
||||||
* having a validated input array
|
|
||||||
*/
|
|
||||||
inline u32 decodeArr(const char* src, isize chlen) {
|
|
||||||
// map of masks, starts at 1
|
|
||||||
static constexpr u8 firstMask[5] = {
|
|
||||||
0x00, // unused
|
|
||||||
0x7F, // 0xxxxxxx
|
|
||||||
0x1F, // 110xxxxx
|
|
||||||
0x0F, // 1110xxxx
|
|
||||||
0x07 // 11110xxx
|
|
||||||
};
|
|
||||||
|
|
||||||
// assemble the char
|
|
||||||
u32 out = u8(src[0]) & firstMask[chlen];
|
|
||||||
for (isize i = 1; i < chlen; ++i) {
|
for (isize i = 1; i < chlen; ++i) {
|
||||||
out <<= 6;
|
if (!isCont(u8(src[i]))) return false;
|
||||||
out |= u8(src[i]) & 0x3F;
|
result = (result << 6) | (u8(src[i]) & 0x3F);
|
||||||
}
|
|
||||||
return out;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
inline bool charAt(std::string_view str, isize& index, u32& out) {
|
// Security & Constraints Validation
|
||||||
isize chlen = isValidSeq(str.begin(), str.size());
|
// So, as it turns out, you can define UTF-8
|
||||||
if(chlen == 0) return false;
|
// characters to be longer than they should.
|
||||||
out = decodeArr(str.begin(), chlen);
|
// This protects against the (albeit niche) hack
|
||||||
index += chlen;
|
// of sending sensitive chars like "\", "<", etc.
|
||||||
|
// and have a parser not target them.
|
||||||
|
// So take this: ALWAYS use UTF-32 to make comparisons!
|
||||||
|
// Byte-sequences are NOT to be trusted NEVER.
|
||||||
|
// Still, this ensures it's DOUBLE correct.
|
||||||
|
if (chlen == 2 && result < 0x80) return false; // Overlong 2-byte
|
||||||
|
if (chlen == 3 && result < 0x0800) return false; // Overlong 3-byte
|
||||||
|
if (chlen == 4 && result < 0x10000) return false; // Overlong 4-byte
|
||||||
|
if (result >= 0xD800 && result <= 0xDFFF) return false; // UTF-16 Surrogates
|
||||||
|
if (result > 0x10FFFF) return false; // Out of Unicode bounds
|
||||||
|
|
||||||
|
out = result;
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline isize decode(const char* src, isize len, u32& out) {
|
||||||
|
if (len <= 0) return 0;
|
||||||
|
|
||||||
|
isize m = seqlen(u8(src[0]));
|
||||||
|
if (m == 0 || m > len) return 0;
|
||||||
|
|
||||||
|
if (decodeArr(src, m, out)) return m;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
inline bool toUTF32(std::string_view str, std::u32string& out) {
|
inline bool toUTF32(std::string_view str, std::u32string& out) {
|
||||||
isize _i = 0, _size;
|
isize _i = 0;
|
||||||
|
isize csize = str.size();
|
||||||
|
const char* data = str.data();
|
||||||
|
|
||||||
auto cptr = str.cbegin();
|
while (_i < csize) {
|
||||||
auto csize = str.size();
|
u32 codepoint;
|
||||||
|
isize _size = decode(data + _i, csize - _i, codepoint);
|
||||||
|
if (_size == 0) return false;
|
||||||
|
|
||||||
while(_i < csize) {
|
out += char32_t(codepoint);
|
||||||
_size = isValidSeq(cptr + _i, csize - _i);
|
|
||||||
if(_size == 0) return false;
|
|
||||||
out += decodeArr(cptr + _i, _size);
|
|
||||||
_i += _size;
|
_i += _size;
|
||||||
}
|
}
|
||||||
|
|
||||||
return _i == csize;
|
return _i == csize;
|
||||||
}
|
}
|
||||||
|
|
||||||
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {}
|
inline const char* getControlCharName(u8 c) {
|
||||||
|
static const char* names[32] = {
|
||||||
|
"NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL",
|
||||||
|
"BS", "HT", "LF", "VT", "FF", "CR", "SO", "SI",
|
||||||
|
"DLE", "DC1", "DC2", "DC3", "DC4", "NAK", "SYN", "ETB",
|
||||||
|
"CAN", "EM", "SUB", "ESC", "FS", "GS", "RS", "US"
|
||||||
|
};
|
||||||
|
if (c < 32) return names[c];
|
||||||
|
if (c == 127) return "DEL";
|
||||||
|
return nullptr;
|
||||||
|
}
|
||||||
|
|
||||||
|
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {
|
||||||
|
auto old_flags = ostr.flags();
|
||||||
|
auto old_fill = ostr.fill();
|
||||||
|
|
||||||
|
isize i = 0;
|
||||||
|
while (i < length) {
|
||||||
|
u8 lead = u8(data[i]);
|
||||||
|
isize m = seqlen(lead);
|
||||||
|
|
||||||
|
// Byte, Line, Col
|
||||||
|
ostr << std::setfill('0') << std::hex << std::uppercase;
|
||||||
|
ostr << "0x" << std::setw(8) << (at.byteoff + i);
|
||||||
|
|
||||||
|
ostr << std::setfill(' ') << std::dec;
|
||||||
|
ostr << " (" << std::setw(3) << at.line << ", " << std::setw(3) << at.col << ") : ";
|
||||||
|
|
||||||
|
// 1. Invalid Lead Byte handling
|
||||||
|
if (m == 0) {
|
||||||
|
ostr << std::setfill('0') << std::hex << std::uppercase;
|
||||||
|
ostr << "0x" << std::setw(2) << u32(lead);
|
||||||
|
ostr << std::setw(15) << std::setfill(' ') << " "; // Pad to match normal spacing
|
||||||
|
ostr << " : INVALID LEAD\n";
|
||||||
|
if (lead == '\n') { at.line++; at.col = 1; } else { at.col++; }
|
||||||
|
i++;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// 2. Truncated Sequence handling (Not enough bytes left in buffer)
|
||||||
|
if ((i + m) > length) {
|
||||||
|
isize available = length - i;
|
||||||
|
|
||||||
|
// Print what we can see
|
||||||
|
for (isize j = 0; j < available; ++j) {
|
||||||
|
ostr << "0x" << std::setw(2) << std::setfill('0') << std::hex << int(u8(data[i + j])) << " ";
|
||||||
|
}
|
||||||
|
// Fill missing bytes with ??
|
||||||
|
for (isize j = available; j < m; ++j) {
|
||||||
|
ostr << "0x?? ";
|
||||||
|
}
|
||||||
|
|
||||||
|
// Pad the rest of the column width
|
||||||
|
isize spaces_to_print = 20 - (m * 5);
|
||||||
|
if(spaces_to_print) ostr << std::setw(i32(spaces_to_print)) << std::setfill(' ') << " ";
|
||||||
|
ostr << ": TRUNC SEQ\n";
|
||||||
|
|
||||||
|
// Update row tracking based on what we actually processed
|
||||||
|
for (isize j = 0; j < available; ++j) {
|
||||||
|
if (u8(data[i + j]) == '\n') { at.line++; at.col = 1; } else { at.col++; }
|
||||||
|
}
|
||||||
|
i += available;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// 3. Fully Available Sequence Processing
|
||||||
|
u32 codepoint = 0;
|
||||||
|
bool valid = decodeArr(data + i, m, codepoint);
|
||||||
|
|
||||||
|
for (isize j = 0; j < m; ++j) {
|
||||||
|
ostr << "0x" << std::setw(2) << std::setfill('0') << std::hex << int(u8(data[i + j])) << " ";
|
||||||
|
}
|
||||||
|
|
||||||
|
isize spaces_to_print = 20 - (m * 5);
|
||||||
|
if(spaces_to_print) ostr << std::setw(i32(spaces_to_print)) << std::setfill(' ') << " ";
|
||||||
|
|
||||||
|
if (valid) {
|
||||||
|
ostr << ": U+" << std::setw(5) << std::setfill('0') << std::uppercase << std::hex << codepoint << " ";
|
||||||
|
|
||||||
|
// Check for control characters
|
||||||
|
const char* ctrlName = getControlCharName(lead); // Multi-byte UTF-8 can't be ASCII control chars
|
||||||
|
if (m == 1 && ctrlName != nullptr) {
|
||||||
|
ostr << "(" << ctrlName << ")\n";
|
||||||
|
} else {
|
||||||
|
ostr.write(data + i, i64(m));
|
||||||
|
ostr << "\n";
|
||||||
|
}
|
||||||
|
|
||||||
|
// Track position updates
|
||||||
|
if (m == 1 && lead == '\n') {
|
||||||
|
at.line++;
|
||||||
|
at.col = 1;
|
||||||
|
} else {
|
||||||
|
at.col++;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
ostr << ": INVALID SEQ\n";
|
||||||
|
at.col++;
|
||||||
|
}
|
||||||
|
|
||||||
|
i += m;
|
||||||
|
}
|
||||||
|
|
||||||
|
ostr.flags(old_flags);
|
||||||
|
ostr.fill(old_fill);
|
||||||
|
at.byteoff += length;
|
||||||
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user