diff --git a/src/spider/compiler/Compiler.cpp b/src/spider/compiler/Compiler.cpp index ed21b53..bd74202 100644 --- a/src/spider/compiler/Compiler.cpp +++ b/src/spider/compiler/Compiler.cpp @@ -7,6 +7,56 @@ namespace spider { +} + +// Test runner helper +void run_test(const std::string& name, const std::string& input) { + std::cout << "========================================\n"; + std::cout << " TEST: " << name << "\n"; + std::cout << "========================================\n"; + + spider::pos tracking_pos; + spider::utf8::hexdump(input.data(), input.size(), tracking_pos, std::cout); + std::cout << "\n"; +} + +void utf8sequences() { + // Permutation 1: Standard, valid mixed UTF-8 sequences (1, 2, 3, and 4 bytes) + // - 'A' -> 1 byte (U+0041) + // - '¢' (cents) -> 2 bytes (U+00A2) + // - '€' (euro) -> 3 bytes (U+20AC) + // - '𐍈' (gothic) -> 4 bytes (U+10348) + run_test("Valid Mixed Length Sequences", "A\xC2\xA2\xE2\x82\xAC\xF0\x90\x8D\x88"); + + // Permutation 2: Embedded Control Characters + // Should display mnemonics like (HT), (LF), (CR) without breaking formatting + run_test("ASCII Control Characters", "Text\tWith\r\nNewlines"); + + // Permutation 3: Invalid Lead Byte + // The byte 0xFF is structurally illegal under any UTF-8 definition. + // Expected behavior: Display single byte as INVALID LEAD, shift 1 byte over. + run_test("Invalid Lead Byte (0xFF)", "ABC\xFFXYZ"); + + // Permutation 4: Invalid Continuation Sequence + // A 3-byte header (\xE2) where the second byte (\x00) is a bad continuation. + // Expected behavior: Show the entire sequence up to 'm' bytes, flag as INVALID SEQUENCE. + run_test("Invalid Continuation Structure", std::string("Before \xE2\x00\xAC After", 16)); + + // Permutation 5: Truncated Sequence at End-of-Buffer + // A 4-byte emoji header (\xF0\x9F) but the string completely cuts off. + // Expected behavior: Display remaining space placeholders as '??' -> TRUNCATED SEQUENCE. + run_test("Truncated Sequence (Missing trailing bytes)", "Hello \xF0\x9F"); + + // Permutation 6: Overlong Encoding Security Vulnerability + // Attempting to write ASCII 'I' (normally 0x49) using 2 bytes: \xC1\x89 + // Expected behavior: Caught by constraints checks, flagged as INVALID SEQUENCE. + run_test("Security Hack: Overlong Encoding", "Safe\xC1\x89Hack"); + + // Permutation 7: Out-of-bounds / Restricted Ranges + // - \xED\xA0\x80 is a UTF-16 Surrogate (U+D800) + // - \xF4\x90\x80\x80 is outside valid Unicode space (> U+10FFFF) + // Expected behavior: Flagged securely as INVALID SEQUENCE. + run_test("Security Hack: Restricted Ranges (Surrogates & Out-of-bounds)", "Surrogate: \xED\xA0\x80 MaxBounds: \xF4\x90\x80\x80"); } int main() { @@ -23,5 +73,8 @@ int main() { std::cout << std::endl; std::cout << "Happy Day!" << std::endl; + spider::utf8::hexdump(test.data(), test.size(), spider::pos(), std::cout); + std::cout << std::endl; + utf8sequences(); return 0; } diff --git a/src/spider/compiler/common.hpp b/src/spider/compiler/common.hpp index 47109f3..b61e38b 100644 --- a/src/spider/compiler/common.hpp +++ b/src/spider/compiler/common.hpp @@ -9,6 +9,7 @@ #include #include #include +#include namespace spider { @@ -40,6 +41,8 @@ namespace spider { using std::optional; using std::set; + template using ilist = std::initializer_list; + template using ref = std::reference_wrapper; template using ptr = std::shared_ptr; template using uptr = std::unique_ptr; diff --git a/src/spider/compiler/text/TextReader.cpp b/src/spider/compiler/text/TextReader.cpp index 599b735..8c70eec 100644 --- a/src/spider/compiler/text/TextReader.cpp +++ b/src/spider/compiler/text/TextReader.cpp @@ -29,18 +29,20 @@ namespace spider { } bool TextReader::eat(const std::string& chars) { - isize index = 0, count = 0; - u32 _char; - while(index < chars.length()) { - if(!utf8::charAt(chars, index, _char)) return false; - if(_char != peekChar(count)) return false; - count++; + // instead of whatever that was, convert to UTF-32 + // and then do an easy compare! + std::u32string str; + if(!utf8::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!"); + + // compare now + isize index; + for(index = 0; index < str.size(); index++) { + if(str[index] != peekChar(index)) return false; } - if(index == chars.length()) { - nextChar(count); - return true; - } - return false; + + // success! + nextChar(index); + return true; } char TextReader::readByte() { @@ -164,14 +166,14 @@ namespace spider { bytes[bindex] = readByte(); if (err) return false; if (eof) return false; - if (!utf8::isCont(u8(bytes[bindex]))) { - err = true; - errmsg = "Invalid continuation of UTF-8 sequence."; - return false; - } } - u32 decodedChar = utf8::decodeArr(bytes, chsize); + u32 decodedChar; + if(!utf8::decodeArr(bytes, chsize, decodedChar)) { + err = true; + errmsg = "Invalid UTF-8 sequence."; + return false; + } buffer.push_back(decodedChar); return true; } diff --git a/src/spider/compiler/text/Token.cpp b/src/spider/compiler/text/Token.cpp index e69de29..638ed2f 100644 --- a/src/spider/compiler/text/Token.cpp +++ b/src/spider/compiler/text/Token.cpp @@ -0,0 +1,183 @@ +#include "Token.hpp" + +namespace spider { + + // ============================================================================ + // Token Implementation + // ============================================================================ + + SeqToken Token::operator&(const Token& tok) { + return SeqToken({ tok, *this }); + } + + OrToken Token::operator|(const Token& tok) { + return OrToken({ tok, *this }); + } + + OptToken Token::operator~() { + return OptToken(*this); + } + + RepToken Token::operator*() { + return RepToken(*this); + } + + // ============================================================================ + // LitToken Implementation + // ============================================================================ + + LitToken::LitToken(std::string_view lit) { + if (!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!"); + } + + LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {} + + TokenResult LitToken::test(TokenContext& ctx) const { + // Safety check: Prevent out-of-bounds pointer slicing if the remaining + // input is smaller than the target literal. + if (ctx.cursor + literal.size() > ctx.input.size()) { + return { false, {} }; + } + + // Window extract optimization: Acquire a zero-copy view over the input segment + std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size()); + + // Direct lexicographical verification of the UTF-32 code-point sequence + if (sub == literal) { + ctx.advance(literal.size()); + return { true, std::u32string(sub) }; // Deep copy payload returned per requirement + } + + return { false, {} }; + } + + + // ============================================================================ + // SeqToken Implementation + // ============================================================================ + + SeqToken::SeqToken(const ilist>& list) : tokens(list) {} + + TokenResult SeqToken::test(TokenContext& ctx) const { + const size_t transactional_fallback_pos = ctx.cursor; + std::u32string accumulated_match; + + // All matching steps within a sequence must pass consecutively. + for (const auto& token_ref : tokens) { + TokenResult res = token_ref.get().test(ctx); + + if (!res.success) { + // Strict ACID Transaction: Roll back context pointer entirely + // if any nested condition in the sequence fails. + ctx.cursor = transactional_fallback_pos; + return { false, {} }; + } + + // Piecewise accumulation of individual matching sub-tokens + accumulated_match += res.match; + } + + return { true, accumulated_match }; + } + + SeqToken SeqToken::operator&(const Token& tok) { + // Intrusive chaining optimization: Appends the next token directly into the existing + // registry vector instead of nesting structures, keeping the layout flattened. + tokens.push_back(std::cref(tok)); + return *this; + } + + + // ============================================================================ + // OrToken Implementation + // ============================================================================ + + OrToken::OrToken(const ilist>& list) : tokens(list) {} + + TokenResult OrToken::test(TokenContext& ctx) const { + const size_t local_fallback_pos = ctx.cursor; + + // Ordered choice evaluation: Evaluate variants sequentially. + for (const auto& token_ref : tokens) { + TokenResult res = token_ref.get().test(ctx); + + if (res.success) { + return res; // Short-circuit branch: return immediately on first valid choice match + } + + // Backtrack isolation: Reset the cursor position before testing the next alternative path + ctx.cursor = local_fallback_pos; + } + + return { false, {} }; + } + + OrToken OrToken::operator|(const Token& tok) { + // Intrusive grouping layout optimization: + // flattens alternative tokens at code evaluation time. + tokens.push_back(tok); + return *this; + } + + + // ============================================================================ + // OptToken Implementation + // ============================================================================ + + OptToken::OptToken(const Token& t) : target(t) {} + + TokenResult OptToken::test(TokenContext& ctx) const { + const size_t local_fallback_pos = ctx.cursor; + TokenResult res = target.test(ctx); + + if (res.success) { + return res; // Option matched exactly 1 instance successfully + } + + // Recovery path: If sub-rule fails, clean up the dirty state mutation + // and successfully return an empty match payload (0 instances). + ctx.cursor = local_fallback_pos; + return { true, U"" }; + } + + OptToken OptToken::operator~() { + // Redundant layer trap protection: returning self + // prevents wrapping an Optional in an Optional + return *this; + } + + + // ============================================================================ + // RepToken Implementation + // ============================================================================ + + RepToken::RepToken(const Token& t) : target(t) {} + + TokenResult RepToken::test(TokenContext& ctx) const { + std::u32string accumulated_match; + + // Greedily consume matches while input stream headroom remains + while (ctx.has_more()) { + const size_t pre_loop_cursor = ctx.cursor; + TokenResult res = target.test(ctx); + + // Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching + // rule succeeded without advancing the buffer index (prevents dynamic parsing lockups). + if (!res.success || ctx.cursor == pre_loop_cursor) { + ctx.cursor = pre_loop_cursor; // Revert cursor to last healthy match checkpoint + break; + } + + accumulated_match += res.match; + } + + // Repetition rules (* token) always evaluate to successful completion state, even with 0 matches. + return { true, accumulated_match }; + } + + RepToken RepToken::operator*() { + // Redundant layer trap protection: returning self prevents wrapping a Repetition rule inside a Repetition rule + return *this; + } + +} diff --git a/src/spider/compiler/text/Token.hpp b/src/spider/compiler/text/Token.hpp index c7ce813..76a4d50 100644 --- a/src/spider/compiler/text/Token.hpp +++ b/src/spider/compiler/text/Token.hpp @@ -1,8 +1,8 @@ #pragma once -#include "spider/compiler/common.hpp" +#include -#include "spider/compiler/text/utf8.hpp" +#include namespace spider { @@ -12,146 +12,203 @@ namespace spider { size_t cursor = 0; bool has_more() const { return cursor < input.size(); } - char32_t peek() const { return input[cursor]; } - void advance(size_t n = 1) { cursor += n; } }; + /** + * @brief The structural payload returned by every parsing component execution. + */ struct TokenResult { + + /** @brief Indicates if the token composition successfully matched the input boundary. */ bool success; + + /** + * @brief Holds the deep-copied UTF-32 matching substring upon victory. + * @note Returns empty when success is false. + */ std::u32string match; + }; + // Forward declarations required by the abstract interface for operator returns. + class SeqToken; + class OrToken; + class OptToken; + class RepToken; + + /** + * @brief Pure virtual base class defining the EBNF combinator node contract. + * @details Implements structural immutability for thread-safe static allocations + * while enforcing fluency using intrusive member operator overloads. + */ class Token { public: + /** @brief Virtual destructor ensuring clean polymorphic destruction of composite graphs. */ virtual ~Token() = default; - virtual TokenResult parse(TokenContext& ctx) const = 0; + public: + + /** + * @brief Evaluates this token node against the provided context. + * @param ctx The current state tracker containing the UTF-32 viewing buffer and cursor. + * @return TokenResult Containing verification state and the parsed copy of matching data. + * @note Pure virtual; implementation details handle node-specific combinator semantics. + */ + virtual TokenResult test(TokenContext& ctx) const = 0; + + public: + + /** + * @brief Chains this token and another sequentially. + * @return A temporary structural bridge matching both tokens sequentially. + */ + virtual SeqToken operator&(const Token& tok); + + /** + * @brief Combines this token and another under alternation. + * @return A structural bridge matching either this token or the fallback selection. + */ + virtual OrToken operator|(const Token& tok); + + /** + * @brief Wraps this node in an optional layout rule. + * @return A structure matching zero or one instances of this current node. + */ + virtual OptToken operator~(); + + /** + * @brief Wraps this node in a repetitive loop match framework. + * @return A structure matching zero or more occurrences of this current node. + */ + virtual RepToken operator*(); }; + /** + * @brief Leaf terminal parser validating precise exact matching UTF-32 sequences. + */ class LitToken : public Token { private: - + /** @brief The reference literal sequence being inspected. */ std::u32string literal; public: + /** + * @brief Constructs a literal rule by transforming a standard UTF-8 string view. + * @throws std::runtime_error If incoming character boundaries contain invalid UTF-8 formatting. + */ + explicit LitToken(std::string_view lit); - explicit LitToken(std::string_view lit) { - if(!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!"); - } - - explicit LitToken(std::u32string lit) : literal(std::move(lit)) {} + /** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */ + explicit LitToken(std::u32string lit); public: - - TokenResult parse(TokenContext& ctx) const override { - if (ctx.cursor + literal.size() > ctx.input.size()) return { false, {} }; - std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size()); - if (sub == literal) { - ctx.advance(literal.size()); - return { true, std::u32string(sub) }; - } - return { false, {} }; - } - + /** + * @brief Validates match of the backing u32string exactly at the context cursor pointer. + * @details Advances the context cursor precisely by literal length on success; zero state mutation on failure. + */ + TokenResult test(TokenContext& ctx) const override; }; - // AndToken references existing static tokens rather than managing their lifecycles - class AndToken : public Token { + /** + * @brief Evaluates an unrolled sequence of ordered grammatical rules sequentially (AndToken). + */ + class SeqToken : public Token { private: - - const Token& lhs; - const Token& rhs; + /** + * @brief Internal contiguous layout registry storing lightweight, zero-overhead references. + * @details Avoids heap allocation penalties by referencing static instances immutably. + */ + vector> tokens; public: - - AndToken(const Token& l, const Token& r) : lhs(l), rhs(r) {} + /** @brief Constructs a sequence container directly out of an inline brace-enclosed listing. */ + SeqToken(const ilist>& list); public: + /** + * @brief Iterates the internal sequence, enforcing that every child node must pass sequentially. + * @details Implements a strict transaction boundary: if any internal element fails, the index + * backtracks entirely to its starting cursor value before returning failure. + */ + TokenResult test(TokenContext& ctx) const override; - TokenResult parse(TokenContext& ctx) const override { - size_t start_pos = ctx.cursor; - if (!lhs.parse(ctx).success) return { false, {} }; - if (!rhs.parse(ctx).success) { - ctx.cursor = start_pos; // Backtrack - return { false, {} }; - } - return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) }; - } + /** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */ + SeqToken operator&(const Token& tok) override; }; + /** + * @brief Implements standard alternation selection matching rules (OrToken). + */ class OrToken : public Token { private: - - const Token& lhs; - const Token& rhs; + /** @brief Ordered registry of possible alternate structural paths. */ + vector> tokens; public: - - OrToken(const Token& l, const Token& r) : lhs(l), rhs(r) {} + /** @brief Constructs an alternation choice layout from brace-enclosed tokens. */ + OrToken(const ilist>& list); public: + /** + * @brief Scans through alternatives, resolving immediately on the first candidate that passes. + * @details Safely rolls back changes to the context cursor point between failed alternative attempts. + */ + TokenResult test(TokenContext& ctx) const override; - TokenResult parse(TokenContext& ctx) const override { - size_t start_pos = ctx.cursor; - auto res = lhs.parse(ctx); - if (res.success) return res; - ctx.cursor = start_pos; // Backtrack - return rhs.parse(ctx); - } - + /** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */ + OrToken operator|(const Token& tok) override; }; + /** + * @brief Structurally represents an optional token condition rule layer (OptToken). + */ class OptToken : public Token { private: - + /** @brief Read-only target node reference to test optional status against. */ const Token& target; public: - - explicit OptToken(const Token& t) : target(t) {} + /** @brief Binds the target node rule structural dependency layout wrapper. */ + explicit OptToken(const Token& t); public: + /** + * @brief Evaluates target presence. Returns success true regardless of sub-rule evaluation outcome. + * @details If the nested rule fails, the context cursor rolls back to initial state, returning empty matches. + */ + TokenResult test(TokenContext& ctx) const override; - TokenResult parse(TokenContext& ctx) const override { - size_t start_pos = ctx.cursor; - if (target.parse(ctx).success) return { - true, - std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) - }; - ctx.cursor = start_pos; - return { true, std::u32string(ctx.input.substr(start_pos, 0)) }; - } - + /** @brief Stub override providing standard compliance with the base Token interface signature. */ + OptToken operator~() override; }; + /** + * @brief Evaluates zero-to-infinite loops of a single repetitive match signature (RepToken). + */ class RepToken : public Token { private: - + /** @brief The base token node sequence layer evaluated in loops. */ const Token& target; public: - - explicit RepToken(const Token& t) : target(t) {} + /** @brief Binds the repeated structural blueprint node wrapper configuration. */ + explicit RepToken(const Token& t); public: + /** + * @brief Greedily attempts to match target repeatedly until an execution failure boundary is hit. + * @details Includes internal safety loops checking cursor delta advancement to guarantee infinite + * empty-matching sub-loops do not cause thread lockups. + */ + TokenResult test(TokenContext& ctx) const override; - TokenResult parse(TokenContext& ctx) const override { - size_t start_pos = ctx.cursor; - while (ctx.has_more()) { - size_t loop_start = ctx.cursor; - if (!target.parse(ctx).success || ctx.cursor == loop_start) { - ctx.cursor = loop_start; - break; - } - } - return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) }; - } - + /** @brief Stub override providing standard compliance with the base Token interface signature. */ + RepToken operator*() override; }; } diff --git a/src/spider/compiler/text/utf8.hpp b/src/spider/compiler/text/utf8.hpp index d230442..87b698a 100644 --- a/src/spider/compiler/text/utf8.hpp +++ b/src/spider/compiler/text/utf8.hpp @@ -26,24 +26,13 @@ namespace spider { return (c & 0xC0) == 0x80; } - constexpr isize isValidSeq(const char* src, isize len) { - if (len == 0) return 0; - isize m = seqlen(u8(src[0])); - if (m == 0 || m > len) return 0; - for (isize i = 1; i < m; i++) { - if (!isCont(u8(src[i]))) return 0; - } - return m; - } - // ----------------- // // UTF-8 into UTF-32 // // ----------------- // - - inline isize decode(const char* src, isize len, u32& out) { - // check input is valid - isize charlen = isValidSeq(src, len); - if (charlen == 0) return 0; + + inline bool decodeArr(const char* src, isize chlen, u32& out) { + // Check character length + if (chlen < 1 || chlen > 4) return false; // map of masks, starts at 1 static constexpr u8 firstMask[5] = { @@ -54,63 +43,165 @@ namespace spider { 0x07 // 11110xxx }; - // assemble the char - out = u8(src[0]) & firstMask[charlen]; - for (isize i = 1; i < charlen; ++i) { - out <<= 6; - out |= u8(src[i]) & 0x3F; - } - return charlen; - } - - /** - * A simpler version, which consider it already - * having a validated input array - */ - inline u32 decodeArr(const char* src, isize chlen) { - // map of masks, starts at 1 - static constexpr u8 firstMask[5] = { - 0x00, // unused - 0x7F, // 0xxxxxxx - 0x1F, // 110xxxxx - 0x0F, // 1110xxxx - 0x07 // 11110xxx - }; - - // assemble the char - u32 out = u8(src[0]) & firstMask[chlen]; + u32 result = u8(src[0]) & firstMask[chlen]; for (isize i = 1; i < chlen; ++i) { - out <<= 6; - out |= u8(src[i]) & 0x3F; + if (!isCont(u8(src[i]))) return false; + result = (result << 6) | (u8(src[i]) & 0x3F); } - return out; - } - inline bool charAt(std::string_view str, isize& index, u32& out) { - isize chlen = isValidSeq(str.begin(), str.size()); - if(chlen == 0) return false; - out = decodeArr(str.begin(), chlen); - index += chlen; + // Security & Constraints Validation + // So, as it turns out, you can define UTF-8 + // characters to be longer than they should. + // This protects against the (albeit niche) hack + // of sending sensitive chars like "\", "<", etc. + // and have a parser not target them. + // So take this: ALWAYS use UTF-32 to make comparisons! + // Byte-sequences are NOT to be trusted NEVER. + // Still, this ensures it's DOUBLE correct. + if (chlen == 2 && result < 0x80) return false; // Overlong 2-byte + if (chlen == 3 && result < 0x0800) return false; // Overlong 3-byte + if (chlen == 4 && result < 0x10000) return false; // Overlong 4-byte + if (result >= 0xD800 && result <= 0xDFFF) return false; // UTF-16 Surrogates + if (result > 0x10FFFF) return false; // Out of Unicode bounds + + out = result; return true; } - inline bool toUTF32(std::string_view str, std::u32string& out) { - isize _i = 0, _size; - - auto cptr = str.cbegin(); - auto csize = str.size(); + inline isize decode(const char* src, isize len, u32& out) { + if (len <= 0) return 0; - while(_i < csize) { - _size = isValidSeq(cptr + _i, csize - _i); - if(_size == 0) return false; - out += decodeArr(cptr + _i, _size); + isize m = seqlen(u8(src[0])); + if (m == 0 || m > len) return 0; + + if (decodeArr(src, m, out)) return m; + return 0; + } + + inline bool toUTF32(std::string_view str, std::u32string& out) { + isize _i = 0; + isize csize = str.size(); + const char* data = str.data(); + + while (_i < csize) { + u32 codepoint; + isize _size = decode(data + _i, csize - _i, codepoint); + if (_size == 0) return false; + + out += char32_t(codepoint); _i += _size; } - + return _i == csize; } - inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {} + inline const char* getControlCharName(u8 c) { + static const char* names[32] = { + "NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL", + "BS", "HT", "LF", "VT", "FF", "CR", "SO", "SI", + "DLE", "DC1", "DC2", "DC3", "DC4", "NAK", "SYN", "ETB", + "CAN", "EM", "SUB", "ESC", "FS", "GS", "RS", "US" + }; + if (c < 32) return names[c]; + if (c == 127) return "DEL"; + return nullptr; + } + + inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) { + auto old_flags = ostr.flags(); + auto old_fill = ostr.fill(); + + isize i = 0; + while (i < length) { + u8 lead = u8(data[i]); + isize m = seqlen(lead); + + // Byte, Line, Col + ostr << std::setfill('0') << std::hex << std::uppercase; + ostr << "0x" << std::setw(8) << (at.byteoff + i); + + ostr << std::setfill(' ') << std::dec; + ostr << " (" << std::setw(3) << at.line << ", " << std::setw(3) << at.col << ") : "; + + // 1. Invalid Lead Byte handling + if (m == 0) { + ostr << std::setfill('0') << std::hex << std::uppercase; + ostr << "0x" << std::setw(2) << u32(lead); + ostr << std::setw(15) << std::setfill(' ') << " "; // Pad to match normal spacing + ostr << " : INVALID LEAD\n"; + if (lead == '\n') { at.line++; at.col = 1; } else { at.col++; } + i++; + continue; + } + + // 2. Truncated Sequence handling (Not enough bytes left in buffer) + if ((i + m) > length) { + isize available = length - i; + + // Print what we can see + for (isize j = 0; j < available; ++j) { + ostr << "0x" << std::setw(2) << std::setfill('0') << std::hex << int(u8(data[i + j])) << " "; + } + // Fill missing bytes with ?? + for (isize j = available; j < m; ++j) { + ostr << "0x?? "; + } + + // Pad the rest of the column width + isize spaces_to_print = 20 - (m * 5); + if(spaces_to_print) ostr << std::setw(i32(spaces_to_print)) << std::setfill(' ') << " "; + ostr << ": TRUNC SEQ\n"; + + // Update row tracking based on what we actually processed + for (isize j = 0; j < available; ++j) { + if (u8(data[i + j]) == '\n') { at.line++; at.col = 1; } else { at.col++; } + } + i += available; + continue; + } + + // 3. Fully Available Sequence Processing + u32 codepoint = 0; + bool valid = decodeArr(data + i, m, codepoint); + + for (isize j = 0; j < m; ++j) { + ostr << "0x" << std::setw(2) << std::setfill('0') << std::hex << int(u8(data[i + j])) << " "; + } + + isize spaces_to_print = 20 - (m * 5); + if(spaces_to_print) ostr << std::setw(i32(spaces_to_print)) << std::setfill(' ') << " "; + + if (valid) { + ostr << ": U+" << std::setw(5) << std::setfill('0') << std::uppercase << std::hex << codepoint << " "; + + // Check for control characters + const char* ctrlName = getControlCharName(lead); // Multi-byte UTF-8 can't be ASCII control chars + if (m == 1 && ctrlName != nullptr) { + ostr << "(" << ctrlName << ")\n"; + } else { + ostr.write(data + i, i64(m)); + ostr << "\n"; + } + + // Track position updates + if (m == 1 && lead == '\n') { + at.line++; + at.col = 1; + } else { + at.col++; + } + } else { + ostr << ": INVALID SEQ\n"; + at.col++; + } + + i += m; + } + + ostr.flags(old_flags); + ostr.fill(old_fill); + at.byteoff += length; + } }