utf-8 hexdump implmented
This commit is contained in:
@@ -7,6 +7,56 @@ namespace spider {
|
||||
|
||||
|
||||
|
||||
}
|
||||
|
||||
// Test runner helper
|
||||
void run_test(const std::string& name, const std::string& input) {
|
||||
std::cout << "========================================\n";
|
||||
std::cout << " TEST: " << name << "\n";
|
||||
std::cout << "========================================\n";
|
||||
|
||||
spider::pos tracking_pos;
|
||||
spider::utf8::hexdump(input.data(), input.size(), tracking_pos, std::cout);
|
||||
std::cout << "\n";
|
||||
}
|
||||
|
||||
void utf8sequences() {
|
||||
// Permutation 1: Standard, valid mixed UTF-8 sequences (1, 2, 3, and 4 bytes)
|
||||
// - 'A' -> 1 byte (U+0041)
|
||||
// - '¢' (cents) -> 2 bytes (U+00A2)
|
||||
// - '€' (euro) -> 3 bytes (U+20AC)
|
||||
// - '𐍈' (gothic) -> 4 bytes (U+10348)
|
||||
run_test("Valid Mixed Length Sequences", "A\xC2\xA2\xE2\x82\xAC\xF0\x90\x8D\x88");
|
||||
|
||||
// Permutation 2: Embedded Control Characters
|
||||
// Should display mnemonics like (HT), (LF), (CR) without breaking formatting
|
||||
run_test("ASCII Control Characters", "Text\tWith\r\nNewlines");
|
||||
|
||||
// Permutation 3: Invalid Lead Byte
|
||||
// The byte 0xFF is structurally illegal under any UTF-8 definition.
|
||||
// Expected behavior: Display single byte as INVALID LEAD, shift 1 byte over.
|
||||
run_test("Invalid Lead Byte (0xFF)", "ABC\xFFXYZ");
|
||||
|
||||
// Permutation 4: Invalid Continuation Sequence
|
||||
// A 3-byte header (\xE2) where the second byte (\x00) is a bad continuation.
|
||||
// Expected behavior: Show the entire sequence up to 'm' bytes, flag as INVALID SEQUENCE.
|
||||
run_test("Invalid Continuation Structure", std::string("Before \xE2\x00\xAC After", 16));
|
||||
|
||||
// Permutation 5: Truncated Sequence at End-of-Buffer
|
||||
// A 4-byte emoji header (\xF0\x9F) but the string completely cuts off.
|
||||
// Expected behavior: Display remaining space placeholders as '??' -> TRUNCATED SEQUENCE.
|
||||
run_test("Truncated Sequence (Missing trailing bytes)", "Hello \xF0\x9F");
|
||||
|
||||
// Permutation 6: Overlong Encoding Security Vulnerability
|
||||
// Attempting to write ASCII 'I' (normally 0x49) using 2 bytes: \xC1\x89
|
||||
// Expected behavior: Caught by constraints checks, flagged as INVALID SEQUENCE.
|
||||
run_test("Security Hack: Overlong Encoding", "Safe\xC1\x89Hack");
|
||||
|
||||
// Permutation 7: Out-of-bounds / Restricted Ranges
|
||||
// - \xED\xA0\x80 is a UTF-16 Surrogate (U+D800)
|
||||
// - \xF4\x90\x80\x80 is outside valid Unicode space (> U+10FFFF)
|
||||
// Expected behavior: Flagged securely as INVALID SEQUENCE.
|
||||
run_test("Security Hack: Restricted Ranges (Surrogates & Out-of-bounds)", "Surrogate: \xED\xA0\x80 MaxBounds: \xF4\x90\x80\x80");
|
||||
}
|
||||
|
||||
int main() {
|
||||
@@ -23,5 +73,8 @@ int main() {
|
||||
std::cout << std::endl;
|
||||
|
||||
std::cout << "Happy Day!" << std::endl;
|
||||
spider::utf8::hexdump(test.data(), test.size(), spider::pos(), std::cout);
|
||||
std::cout << std::endl;
|
||||
utf8sequences();
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
#include <memory>
|
||||
#include <filesystem>
|
||||
#include <set>
|
||||
#include <functional>
|
||||
|
||||
namespace spider {
|
||||
|
||||
@@ -40,6 +41,8 @@ namespace spider {
|
||||
using std::optional;
|
||||
using std::set;
|
||||
|
||||
template<typename T> using ilist = std::initializer_list<T>;
|
||||
template<typename T> using ref = std::reference_wrapper<T>;
|
||||
template<typename T> using ptr = std::shared_ptr<T>;
|
||||
template<typename T> using uptr = std::unique_ptr<T>;
|
||||
|
||||
|
||||
@@ -29,18 +29,20 @@ namespace spider {
|
||||
}
|
||||
|
||||
bool TextReader::eat(const std::string& chars) {
|
||||
isize index = 0, count = 0;
|
||||
u32 _char;
|
||||
while(index < chars.length()) {
|
||||
if(!utf8::charAt(chars, index, _char)) return false;
|
||||
if(_char != peekChar(count)) return false;
|
||||
count++;
|
||||
// instead of whatever that was, convert to UTF-32
|
||||
// and then do an easy compare!
|
||||
std::u32string str;
|
||||
if(!utf8::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!");
|
||||
|
||||
// compare now
|
||||
isize index;
|
||||
for(index = 0; index < str.size(); index++) {
|
||||
if(str[index] != peekChar(index)) return false;
|
||||
}
|
||||
if(index == chars.length()) {
|
||||
nextChar(count);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
|
||||
// success!
|
||||
nextChar(index);
|
||||
return true;
|
||||
}
|
||||
|
||||
char TextReader::readByte() {
|
||||
@@ -164,14 +166,14 @@ namespace spider {
|
||||
bytes[bindex] = readByte();
|
||||
if (err) return false;
|
||||
if (eof) return false;
|
||||
if (!utf8::isCont(u8(bytes[bindex]))) {
|
||||
err = true;
|
||||
errmsg = "Invalid continuation of UTF-8 sequence.";
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
u32 decodedChar = utf8::decodeArr(bytes, chsize);
|
||||
u32 decodedChar;
|
||||
if(!utf8::decodeArr(bytes, chsize, decodedChar)) {
|
||||
err = true;
|
||||
errmsg = "Invalid UTF-8 sequence.";
|
||||
return false;
|
||||
}
|
||||
buffer.push_back(decodedChar);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,183 @@
|
||||
#include "Token.hpp"
|
||||
|
||||
namespace spider {
|
||||
|
||||
// ============================================================================
|
||||
// Token Implementation
|
||||
// ============================================================================
|
||||
|
||||
SeqToken Token::operator&(const Token& tok) {
|
||||
return SeqToken({ tok, *this });
|
||||
}
|
||||
|
||||
OrToken Token::operator|(const Token& tok) {
|
||||
return OrToken({ tok, *this });
|
||||
}
|
||||
|
||||
OptToken Token::operator~() {
|
||||
return OptToken(*this);
|
||||
}
|
||||
|
||||
RepToken Token::operator*() {
|
||||
return RepToken(*this);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// LitToken Implementation
|
||||
// ============================================================================
|
||||
|
||||
LitToken::LitToken(std::string_view lit) {
|
||||
if (!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
|
||||
}
|
||||
|
||||
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
|
||||
|
||||
TokenResult LitToken::test(TokenContext& ctx) const {
|
||||
// Safety check: Prevent out-of-bounds pointer slicing if the remaining
|
||||
// input is smaller than the target literal.
|
||||
if (ctx.cursor + literal.size() > ctx.input.size()) {
|
||||
return { false, {} };
|
||||
}
|
||||
|
||||
// Window extract optimization: Acquire a zero-copy view over the input segment
|
||||
std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size());
|
||||
|
||||
// Direct lexicographical verification of the UTF-32 code-point sequence
|
||||
if (sub == literal) {
|
||||
ctx.advance(literal.size());
|
||||
return { true, std::u32string(sub) }; // Deep copy payload returned per requirement
|
||||
}
|
||||
|
||||
return { false, {} };
|
||||
}
|
||||
|
||||
|
||||
// ============================================================================
|
||||
// SeqToken Implementation
|
||||
// ============================================================================
|
||||
|
||||
SeqToken::SeqToken(const ilist<ref<const Token>>& list) : tokens(list) {}
|
||||
|
||||
TokenResult SeqToken::test(TokenContext& ctx) const {
|
||||
const size_t transactional_fallback_pos = ctx.cursor;
|
||||
std::u32string accumulated_match;
|
||||
|
||||
// All matching steps within a sequence must pass consecutively.
|
||||
for (const auto& token_ref : tokens) {
|
||||
TokenResult res = token_ref.get().test(ctx);
|
||||
|
||||
if (!res.success) {
|
||||
// Strict ACID Transaction: Roll back context pointer entirely
|
||||
// if any nested condition in the sequence fails.
|
||||
ctx.cursor = transactional_fallback_pos;
|
||||
return { false, {} };
|
||||
}
|
||||
|
||||
// Piecewise accumulation of individual matching sub-tokens
|
||||
accumulated_match += res.match;
|
||||
}
|
||||
|
||||
return { true, accumulated_match };
|
||||
}
|
||||
|
||||
SeqToken SeqToken::operator&(const Token& tok) {
|
||||
// Intrusive chaining optimization: Appends the next token directly into the existing
|
||||
// registry vector instead of nesting structures, keeping the layout flattened.
|
||||
tokens.push_back(std::cref(tok));
|
||||
return *this;
|
||||
}
|
||||
|
||||
|
||||
// ============================================================================
|
||||
// OrToken Implementation
|
||||
// ============================================================================
|
||||
|
||||
OrToken::OrToken(const ilist<ref<const Token>>& list) : tokens(list) {}
|
||||
|
||||
TokenResult OrToken::test(TokenContext& ctx) const {
|
||||
const size_t local_fallback_pos = ctx.cursor;
|
||||
|
||||
// Ordered choice evaluation: Evaluate variants sequentially.
|
||||
for (const auto& token_ref : tokens) {
|
||||
TokenResult res = token_ref.get().test(ctx);
|
||||
|
||||
if (res.success) {
|
||||
return res; // Short-circuit branch: return immediately on first valid choice match
|
||||
}
|
||||
|
||||
// Backtrack isolation: Reset the cursor position before testing the next alternative path
|
||||
ctx.cursor = local_fallback_pos;
|
||||
}
|
||||
|
||||
return { false, {} };
|
||||
}
|
||||
|
||||
OrToken OrToken::operator|(const Token& tok) {
|
||||
// Intrusive grouping layout optimization:
|
||||
// flattens alternative tokens at code evaluation time.
|
||||
tokens.push_back(tok);
|
||||
return *this;
|
||||
}
|
||||
|
||||
|
||||
// ============================================================================
|
||||
// OptToken Implementation
|
||||
// ============================================================================
|
||||
|
||||
OptToken::OptToken(const Token& t) : target(t) {}
|
||||
|
||||
TokenResult OptToken::test(TokenContext& ctx) const {
|
||||
const size_t local_fallback_pos = ctx.cursor;
|
||||
TokenResult res = target.test(ctx);
|
||||
|
||||
if (res.success) {
|
||||
return res; // Option matched exactly 1 instance successfully
|
||||
}
|
||||
|
||||
// Recovery path: If sub-rule fails, clean up the dirty state mutation
|
||||
// and successfully return an empty match payload (0 instances).
|
||||
ctx.cursor = local_fallback_pos;
|
||||
return { true, U"" };
|
||||
}
|
||||
|
||||
OptToken OptToken::operator~() {
|
||||
// Redundant layer trap protection: returning self
|
||||
// prevents wrapping an Optional in an Optional
|
||||
return *this;
|
||||
}
|
||||
|
||||
|
||||
// ============================================================================
|
||||
// RepToken Implementation
|
||||
// ============================================================================
|
||||
|
||||
RepToken::RepToken(const Token& t) : target(t) {}
|
||||
|
||||
TokenResult RepToken::test(TokenContext& ctx) const {
|
||||
std::u32string accumulated_match;
|
||||
|
||||
// Greedily consume matches while input stream headroom remains
|
||||
while (ctx.has_more()) {
|
||||
const size_t pre_loop_cursor = ctx.cursor;
|
||||
TokenResult res = target.test(ctx);
|
||||
|
||||
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
|
||||
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
|
||||
if (!res.success || ctx.cursor == pre_loop_cursor) {
|
||||
ctx.cursor = pre_loop_cursor; // Revert cursor to last healthy match checkpoint
|
||||
break;
|
||||
}
|
||||
|
||||
accumulated_match += res.match;
|
||||
}
|
||||
|
||||
// Repetition rules (* token) always evaluate to successful completion state, even with 0 matches.
|
||||
return { true, accumulated_match };
|
||||
}
|
||||
|
||||
RepToken RepToken::operator*() {
|
||||
// Redundant layer trap protection: returning self prevents wrapping a Repetition rule inside a Repetition rule
|
||||
return *this;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
#pragma once
|
||||
|
||||
#include "spider/compiler/common.hpp"
|
||||
#include <spider/compiler/common.hpp>
|
||||
|
||||
#include "spider/compiler/text/utf8.hpp"
|
||||
#include <spider/compiler/text/utf8.hpp>
|
||||
|
||||
namespace spider {
|
||||
|
||||
@@ -12,146 +12,203 @@ namespace spider {
|
||||
size_t cursor = 0;
|
||||
|
||||
bool has_more() const { return cursor < input.size(); }
|
||||
|
||||
char32_t peek() const { return input[cursor]; }
|
||||
|
||||
void advance(size_t n = 1) { cursor += n; }
|
||||
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief The structural payload returned by every parsing component execution.
|
||||
*/
|
||||
struct TokenResult {
|
||||
|
||||
/** @brief Indicates if the token composition successfully matched the input boundary. */
|
||||
bool success;
|
||||
|
||||
/**
|
||||
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
|
||||
* @note Returns empty when success is false.
|
||||
*/
|
||||
std::u32string match;
|
||||
|
||||
};
|
||||
|
||||
// Forward declarations required by the abstract interface for operator returns.
|
||||
class SeqToken;
|
||||
class OrToken;
|
||||
class OptToken;
|
||||
class RepToken;
|
||||
|
||||
/**
|
||||
* @brief Pure virtual base class defining the EBNF combinator node contract.
|
||||
* @details Implements structural immutability for thread-safe static allocations
|
||||
* while enforcing fluency using intrusive member operator overloads.
|
||||
*/
|
||||
class Token {
|
||||
public:
|
||||
|
||||
/** @brief Virtual destructor ensuring clean polymorphic destruction of composite graphs. */
|
||||
virtual ~Token() = default;
|
||||
|
||||
virtual TokenResult parse(TokenContext& ctx) const = 0;
|
||||
public:
|
||||
|
||||
/**
|
||||
* @brief Evaluates this token node against the provided context.
|
||||
* @param ctx The current state tracker containing the UTF-32 viewing buffer and cursor.
|
||||
* @return TokenResult Containing verification state and the parsed copy of matching data.
|
||||
* @note Pure virtual; implementation details handle node-specific combinator semantics.
|
||||
*/
|
||||
virtual TokenResult test(TokenContext& ctx) const = 0;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
* @brief Chains this token and another sequentially.
|
||||
* @return A temporary structural bridge matching both tokens sequentially.
|
||||
*/
|
||||
virtual SeqToken operator&(const Token& tok);
|
||||
|
||||
/**
|
||||
* @brief Combines this token and another under alternation.
|
||||
* @return A structural bridge matching either this token or the fallback selection.
|
||||
*/
|
||||
virtual OrToken operator|(const Token& tok);
|
||||
|
||||
/**
|
||||
* @brief Wraps this node in an optional layout rule.
|
||||
* @return A structure matching zero or one instances of this current node.
|
||||
*/
|
||||
virtual OptToken operator~();
|
||||
|
||||
/**
|
||||
* @brief Wraps this node in a repetitive loop match framework.
|
||||
* @return A structure matching zero or more occurrences of this current node.
|
||||
*/
|
||||
virtual RepToken operator*();
|
||||
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Leaf terminal parser validating precise exact matching UTF-32 sequences.
|
||||
*/
|
||||
class LitToken : public Token {
|
||||
private:
|
||||
|
||||
/** @brief The reference literal sequence being inspected. */
|
||||
std::u32string literal;
|
||||
|
||||
public:
|
||||
/**
|
||||
* @brief Constructs a literal rule by transforming a standard UTF-8 string view.
|
||||
* @throws std::runtime_error If incoming character boundaries contain invalid UTF-8 formatting.
|
||||
*/
|
||||
explicit LitToken(std::string_view lit);
|
||||
|
||||
explicit LitToken(std::string_view lit) {
|
||||
if(!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
|
||||
}
|
||||
|
||||
explicit LitToken(std::u32string lit) : literal(std::move(lit)) {}
|
||||
/** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */
|
||||
explicit LitToken(std::u32string lit);
|
||||
|
||||
public:
|
||||
|
||||
TokenResult parse(TokenContext& ctx) const override {
|
||||
if (ctx.cursor + literal.size() > ctx.input.size()) return { false, {} };
|
||||
std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size());
|
||||
if (sub == literal) {
|
||||
ctx.advance(literal.size());
|
||||
return { true, std::u32string(sub) };
|
||||
}
|
||||
return { false, {} };
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Validates match of the backing u32string exactly at the context cursor pointer.
|
||||
* @details Advances the context cursor precisely by literal length on success; zero state mutation on failure.
|
||||
*/
|
||||
TokenResult test(TokenContext& ctx) const override;
|
||||
};
|
||||
|
||||
// AndToken references existing static tokens rather than managing their lifecycles
|
||||
class AndToken : public Token {
|
||||
/**
|
||||
* @brief Evaluates an unrolled sequence of ordered grammatical rules sequentially (AndToken).
|
||||
*/
|
||||
class SeqToken : public Token {
|
||||
private:
|
||||
|
||||
const Token& lhs;
|
||||
const Token& rhs;
|
||||
/**
|
||||
* @brief Internal contiguous layout registry storing lightweight, zero-overhead references.
|
||||
* @details Avoids heap allocation penalties by referencing static instances immutably.
|
||||
*/
|
||||
vector<ref<const Token>> tokens;
|
||||
|
||||
public:
|
||||
|
||||
AndToken(const Token& l, const Token& r) : lhs(l), rhs(r) {}
|
||||
/** @brief Constructs a sequence container directly out of an inline brace-enclosed listing. */
|
||||
SeqToken(const ilist<ref<const Token>>& list);
|
||||
|
||||
public:
|
||||
/**
|
||||
* @brief Iterates the internal sequence, enforcing that every child node must pass sequentially.
|
||||
* @details Implements a strict transaction boundary: if any internal element fails, the index
|
||||
* backtracks entirely to its starting cursor value before returning failure.
|
||||
*/
|
||||
TokenResult test(TokenContext& ctx) const override;
|
||||
|
||||
TokenResult parse(TokenContext& ctx) const override {
|
||||
size_t start_pos = ctx.cursor;
|
||||
if (!lhs.parse(ctx).success) return { false, {} };
|
||||
if (!rhs.parse(ctx).success) {
|
||||
ctx.cursor = start_pos; // Backtrack
|
||||
return { false, {} };
|
||||
}
|
||||
return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) };
|
||||
}
|
||||
/** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */
|
||||
SeqToken operator&(const Token& tok) override;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Implements standard alternation selection matching rules (OrToken).
|
||||
*/
|
||||
class OrToken : public Token {
|
||||
private:
|
||||
|
||||
const Token& lhs;
|
||||
const Token& rhs;
|
||||
/** @brief Ordered registry of possible alternate structural paths. */
|
||||
vector<ref<const Token>> tokens;
|
||||
|
||||
public:
|
||||
|
||||
OrToken(const Token& l, const Token& r) : lhs(l), rhs(r) {}
|
||||
/** @brief Constructs an alternation choice layout from brace-enclosed tokens. */
|
||||
OrToken(const ilist<ref<const Token>>& list);
|
||||
|
||||
public:
|
||||
/**
|
||||
* @brief Scans through alternatives, resolving immediately on the first candidate that passes.
|
||||
* @details Safely rolls back changes to the context cursor point between failed alternative attempts.
|
||||
*/
|
||||
TokenResult test(TokenContext& ctx) const override;
|
||||
|
||||
TokenResult parse(TokenContext& ctx) const override {
|
||||
size_t start_pos = ctx.cursor;
|
||||
auto res = lhs.parse(ctx);
|
||||
if (res.success) return res;
|
||||
ctx.cursor = start_pos; // Backtrack
|
||||
return rhs.parse(ctx);
|
||||
}
|
||||
|
||||
/** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */
|
||||
OrToken operator|(const Token& tok) override;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Structurally represents an optional token condition rule layer (OptToken).
|
||||
*/
|
||||
class OptToken : public Token {
|
||||
private:
|
||||
|
||||
/** @brief Read-only target node reference to test optional status against. */
|
||||
const Token& target;
|
||||
|
||||
public:
|
||||
|
||||
explicit OptToken(const Token& t) : target(t) {}
|
||||
/** @brief Binds the target node rule structural dependency layout wrapper. */
|
||||
explicit OptToken(const Token& t);
|
||||
|
||||
public:
|
||||
/**
|
||||
* @brief Evaluates target presence. Returns success true regardless of sub-rule evaluation outcome.
|
||||
* @details If the nested rule fails, the context cursor rolls back to initial state, returning empty matches.
|
||||
*/
|
||||
TokenResult test(TokenContext& ctx) const override;
|
||||
|
||||
TokenResult parse(TokenContext& ctx) const override {
|
||||
size_t start_pos = ctx.cursor;
|
||||
if (target.parse(ctx).success) return {
|
||||
true,
|
||||
std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos))
|
||||
};
|
||||
ctx.cursor = start_pos;
|
||||
return { true, std::u32string(ctx.input.substr(start_pos, 0)) };
|
||||
}
|
||||
|
||||
/** @brief Stub override providing standard compliance with the base Token interface signature. */
|
||||
OptToken operator~() override;
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Evaluates zero-to-infinite loops of a single repetitive match signature (RepToken).
|
||||
*/
|
||||
class RepToken : public Token {
|
||||
private:
|
||||
|
||||
/** @brief The base token node sequence layer evaluated in loops. */
|
||||
const Token& target;
|
||||
|
||||
public:
|
||||
|
||||
explicit RepToken(const Token& t) : target(t) {}
|
||||
/** @brief Binds the repeated structural blueprint node wrapper configuration. */
|
||||
explicit RepToken(const Token& t);
|
||||
|
||||
public:
|
||||
/**
|
||||
* @brief Greedily attempts to match target repeatedly until an execution failure boundary is hit.
|
||||
* @details Includes internal safety loops checking cursor delta advancement to guarantee infinite
|
||||
* empty-matching sub-loops do not cause thread lockups.
|
||||
*/
|
||||
TokenResult test(TokenContext& ctx) const override;
|
||||
|
||||
TokenResult parse(TokenContext& ctx) const override {
|
||||
size_t start_pos = ctx.cursor;
|
||||
while (ctx.has_more()) {
|
||||
size_t loop_start = ctx.cursor;
|
||||
if (!target.parse(ctx).success || ctx.cursor == loop_start) {
|
||||
ctx.cursor = loop_start;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) };
|
||||
}
|
||||
|
||||
/** @brief Stub override providing standard compliance with the base Token interface signature. */
|
||||
RepToken operator*() override;
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
@@ -26,24 +26,13 @@ namespace spider {
|
||||
return (c & 0xC0) == 0x80;
|
||||
}
|
||||
|
||||
constexpr isize isValidSeq(const char* src, isize len) {
|
||||
if (len == 0) return 0;
|
||||
isize m = seqlen(u8(src[0]));
|
||||
if (m == 0 || m > len) return 0;
|
||||
for (isize i = 1; i < m; i++) {
|
||||
if (!isCont(u8(src[i]))) return 0;
|
||||
}
|
||||
return m;
|
||||
}
|
||||
|
||||
// ----------------- //
|
||||
// UTF-8 into UTF-32 //
|
||||
// ----------------- //
|
||||
|
||||
inline isize decode(const char* src, isize len, u32& out) {
|
||||
// check input is valid
|
||||
isize charlen = isValidSeq(src, len);
|
||||
if (charlen == 0) return 0;
|
||||
inline bool decodeArr(const char* src, isize chlen, u32& out) {
|
||||
// Check character length
|
||||
if (chlen < 1 || chlen > 4) return false;
|
||||
|
||||
// map of masks, starts at 1
|
||||
static constexpr u8 firstMask[5] = {
|
||||
@@ -54,63 +43,165 @@ namespace spider {
|
||||
0x07 // 11110xxx
|
||||
};
|
||||
|
||||
// assemble the char
|
||||
out = u8(src[0]) & firstMask[charlen];
|
||||
for (isize i = 1; i < charlen; ++i) {
|
||||
out <<= 6;
|
||||
out |= u8(src[i]) & 0x3F;
|
||||
}
|
||||
return charlen;
|
||||
}
|
||||
|
||||
/**
|
||||
* A simpler version, which consider it already
|
||||
* having a validated input array
|
||||
*/
|
||||
inline u32 decodeArr(const char* src, isize chlen) {
|
||||
// map of masks, starts at 1
|
||||
static constexpr u8 firstMask[5] = {
|
||||
0x00, // unused
|
||||
0x7F, // 0xxxxxxx
|
||||
0x1F, // 110xxxxx
|
||||
0x0F, // 1110xxxx
|
||||
0x07 // 11110xxx
|
||||
};
|
||||
|
||||
// assemble the char
|
||||
u32 out = u8(src[0]) & firstMask[chlen];
|
||||
u32 result = u8(src[0]) & firstMask[chlen];
|
||||
for (isize i = 1; i < chlen; ++i) {
|
||||
out <<= 6;
|
||||
out |= u8(src[i]) & 0x3F;
|
||||
if (!isCont(u8(src[i]))) return false;
|
||||
result = (result << 6) | (u8(src[i]) & 0x3F);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
inline bool charAt(std::string_view str, isize& index, u32& out) {
|
||||
isize chlen = isValidSeq(str.begin(), str.size());
|
||||
if(chlen == 0) return false;
|
||||
out = decodeArr(str.begin(), chlen);
|
||||
index += chlen;
|
||||
// Security & Constraints Validation
|
||||
// So, as it turns out, you can define UTF-8
|
||||
// characters to be longer than they should.
|
||||
// This protects against the (albeit niche) hack
|
||||
// of sending sensitive chars like "\", "<", etc.
|
||||
// and have a parser not target them.
|
||||
// So take this: ALWAYS use UTF-32 to make comparisons!
|
||||
// Byte-sequences are NOT to be trusted NEVER.
|
||||
// Still, this ensures it's DOUBLE correct.
|
||||
if (chlen == 2 && result < 0x80) return false; // Overlong 2-byte
|
||||
if (chlen == 3 && result < 0x0800) return false; // Overlong 3-byte
|
||||
if (chlen == 4 && result < 0x10000) return false; // Overlong 4-byte
|
||||
if (result >= 0xD800 && result <= 0xDFFF) return false; // UTF-16 Surrogates
|
||||
if (result > 0x10FFFF) return false; // Out of Unicode bounds
|
||||
|
||||
out = result;
|
||||
return true;
|
||||
}
|
||||
|
||||
inline isize decode(const char* src, isize len, u32& out) {
|
||||
if (len <= 0) return 0;
|
||||
|
||||
isize m = seqlen(u8(src[0]));
|
||||
if (m == 0 || m > len) return 0;
|
||||
|
||||
if (decodeArr(src, m, out)) return m;
|
||||
return 0;
|
||||
}
|
||||
|
||||
inline bool toUTF32(std::string_view str, std::u32string& out) {
|
||||
isize _i = 0, _size;
|
||||
isize _i = 0;
|
||||
isize csize = str.size();
|
||||
const char* data = str.data();
|
||||
|
||||
auto cptr = str.cbegin();
|
||||
auto csize = str.size();
|
||||
while (_i < csize) {
|
||||
u32 codepoint;
|
||||
isize _size = decode(data + _i, csize - _i, codepoint);
|
||||
if (_size == 0) return false;
|
||||
|
||||
while(_i < csize) {
|
||||
_size = isValidSeq(cptr + _i, csize - _i);
|
||||
if(_size == 0) return false;
|
||||
out += decodeArr(cptr + _i, _size);
|
||||
out += char32_t(codepoint);
|
||||
_i += _size;
|
||||
}
|
||||
|
||||
return _i == csize;
|
||||
}
|
||||
|
||||
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {}
|
||||
inline const char* getControlCharName(u8 c) {
|
||||
static const char* names[32] = {
|
||||
"NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL",
|
||||
"BS", "HT", "LF", "VT", "FF", "CR", "SO", "SI",
|
||||
"DLE", "DC1", "DC2", "DC3", "DC4", "NAK", "SYN", "ETB",
|
||||
"CAN", "EM", "SUB", "ESC", "FS", "GS", "RS", "US"
|
||||
};
|
||||
if (c < 32) return names[c];
|
||||
if (c == 127) return "DEL";
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {
|
||||
auto old_flags = ostr.flags();
|
||||
auto old_fill = ostr.fill();
|
||||
|
||||
isize i = 0;
|
||||
while (i < length) {
|
||||
u8 lead = u8(data[i]);
|
||||
isize m = seqlen(lead);
|
||||
|
||||
// Byte, Line, Col
|
||||
ostr << std::setfill('0') << std::hex << std::uppercase;
|
||||
ostr << "0x" << std::setw(8) << (at.byteoff + i);
|
||||
|
||||
ostr << std::setfill(' ') << std::dec;
|
||||
ostr << " (" << std::setw(3) << at.line << ", " << std::setw(3) << at.col << ") : ";
|
||||
|
||||
// 1. Invalid Lead Byte handling
|
||||
if (m == 0) {
|
||||
ostr << std::setfill('0') << std::hex << std::uppercase;
|
||||
ostr << "0x" << std::setw(2) << u32(lead);
|
||||
ostr << std::setw(15) << std::setfill(' ') << " "; // Pad to match normal spacing
|
||||
ostr << " : INVALID LEAD\n";
|
||||
if (lead == '\n') { at.line++; at.col = 1; } else { at.col++; }
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
|
||||
// 2. Truncated Sequence handling (Not enough bytes left in buffer)
|
||||
if ((i + m) > length) {
|
||||
isize available = length - i;
|
||||
|
||||
// Print what we can see
|
||||
for (isize j = 0; j < available; ++j) {
|
||||
ostr << "0x" << std::setw(2) << std::setfill('0') << std::hex << int(u8(data[i + j])) << " ";
|
||||
}
|
||||
// Fill missing bytes with ??
|
||||
for (isize j = available; j < m; ++j) {
|
||||
ostr << "0x?? ";
|
||||
}
|
||||
|
||||
// Pad the rest of the column width
|
||||
isize spaces_to_print = 20 - (m * 5);
|
||||
if(spaces_to_print) ostr << std::setw(i32(spaces_to_print)) << std::setfill(' ') << " ";
|
||||
ostr << ": TRUNC SEQ\n";
|
||||
|
||||
// Update row tracking based on what we actually processed
|
||||
for (isize j = 0; j < available; ++j) {
|
||||
if (u8(data[i + j]) == '\n') { at.line++; at.col = 1; } else { at.col++; }
|
||||
}
|
||||
i += available;
|
||||
continue;
|
||||
}
|
||||
|
||||
// 3. Fully Available Sequence Processing
|
||||
u32 codepoint = 0;
|
||||
bool valid = decodeArr(data + i, m, codepoint);
|
||||
|
||||
for (isize j = 0; j < m; ++j) {
|
||||
ostr << "0x" << std::setw(2) << std::setfill('0') << std::hex << int(u8(data[i + j])) << " ";
|
||||
}
|
||||
|
||||
isize spaces_to_print = 20 - (m * 5);
|
||||
if(spaces_to_print) ostr << std::setw(i32(spaces_to_print)) << std::setfill(' ') << " ";
|
||||
|
||||
if (valid) {
|
||||
ostr << ": U+" << std::setw(5) << std::setfill('0') << std::uppercase << std::hex << codepoint << " ";
|
||||
|
||||
// Check for control characters
|
||||
const char* ctrlName = getControlCharName(lead); // Multi-byte UTF-8 can't be ASCII control chars
|
||||
if (m == 1 && ctrlName != nullptr) {
|
||||
ostr << "(" << ctrlName << ")\n";
|
||||
} else {
|
||||
ostr.write(data + i, i64(m));
|
||||
ostr << "\n";
|
||||
}
|
||||
|
||||
// Track position updates
|
||||
if (m == 1 && lead == '\n') {
|
||||
at.line++;
|
||||
at.col = 1;
|
||||
} else {
|
||||
at.col++;
|
||||
}
|
||||
} else {
|
||||
ostr << ": INVALID SEQ\n";
|
||||
at.col++;
|
||||
}
|
||||
|
||||
i += m;
|
||||
}
|
||||
|
||||
ostr.flags(old_flags);
|
||||
ostr.fill(old_fill);
|
||||
at.byteoff += length;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user