utf-8 hexdump implmented

This commit is contained in:
2026-07-19 19:56:17 -06:00
parent 09d117debf
commit b30a59accd
6 changed files with 545 additions and 156 deletions
+53
View File
@@ -7,6 +7,56 @@ namespace spider {
}
// Test runner helper
void run_test(const std::string& name, const std::string& input) {
std::cout << "========================================\n";
std::cout << " TEST: " << name << "\n";
std::cout << "========================================\n";
spider::pos tracking_pos;
spider::utf8::hexdump(input.data(), input.size(), tracking_pos, std::cout);
std::cout << "\n";
}
void utf8sequences() {
// Permutation 1: Standard, valid mixed UTF-8 sequences (1, 2, 3, and 4 bytes)
// - 'A' -> 1 byte (U+0041)
// - '¢' (cents) -> 2 bytes (U+00A2)
// - '€' (euro) -> 3 bytes (U+20AC)
// - '𐍈' (gothic) -> 4 bytes (U+10348)
run_test("Valid Mixed Length Sequences", "A\xC2\xA2\xE2\x82\xAC\xF0\x90\x8D\x88");
// Permutation 2: Embedded Control Characters
// Should display mnemonics like (HT), (LF), (CR) without breaking formatting
run_test("ASCII Control Characters", "Text\tWith\r\nNewlines");
// Permutation 3: Invalid Lead Byte
// The byte 0xFF is structurally illegal under any UTF-8 definition.
// Expected behavior: Display single byte as INVALID LEAD, shift 1 byte over.
run_test("Invalid Lead Byte (0xFF)", "ABC\xFFXYZ");
// Permutation 4: Invalid Continuation Sequence
// A 3-byte header (\xE2) where the second byte (\x00) is a bad continuation.
// Expected behavior: Show the entire sequence up to 'm' bytes, flag as INVALID SEQUENCE.
run_test("Invalid Continuation Structure", std::string("Before \xE2\x00\xAC After", 16));
// Permutation 5: Truncated Sequence at End-of-Buffer
// A 4-byte emoji header (\xF0\x9F) but the string completely cuts off.
// Expected behavior: Display remaining space placeholders as '??' -> TRUNCATED SEQUENCE.
run_test("Truncated Sequence (Missing trailing bytes)", "Hello \xF0\x9F");
// Permutation 6: Overlong Encoding Security Vulnerability
// Attempting to write ASCII 'I' (normally 0x49) using 2 bytes: \xC1\x89
// Expected behavior: Caught by constraints checks, flagged as INVALID SEQUENCE.
run_test("Security Hack: Overlong Encoding", "Safe\xC1\x89Hack");
// Permutation 7: Out-of-bounds / Restricted Ranges
// - \xED\xA0\x80 is a UTF-16 Surrogate (U+D800)
// - \xF4\x90\x80\x80 is outside valid Unicode space (> U+10FFFF)
// Expected behavior: Flagged securely as INVALID SEQUENCE.
run_test("Security Hack: Restricted Ranges (Surrogates & Out-of-bounds)", "Surrogate: \xED\xA0\x80 MaxBounds: \xF4\x90\x80\x80");
}
int main() {
@@ -23,5 +73,8 @@ int main() {
std::cout << std::endl;
std::cout << "Happy Day!" << std::endl;
spider::utf8::hexdump(test.data(), test.size(), spider::pos(), std::cout);
std::cout << std::endl;
utf8sequences();
return 0;
}
+3
View File
@@ -9,6 +9,7 @@
#include <memory>
#include <filesystem>
#include <set>
#include <functional>
namespace spider {
@@ -40,6 +41,8 @@ namespace spider {
using std::optional;
using std::set;
template<typename T> using ilist = std::initializer_list<T>;
template<typename T> using ref = std::reference_wrapper<T>;
template<typename T> using ptr = std::shared_ptr<T>;
template<typename T> using uptr = std::unique_ptr<T>;
+19 -17
View File
@@ -29,18 +29,20 @@ namespace spider {
}
bool TextReader::eat(const std::string& chars) {
isize index = 0, count = 0;
u32 _char;
while(index < chars.length()) {
if(!utf8::charAt(chars, index, _char)) return false;
if(_char != peekChar(count)) return false;
count++;
// instead of whatever that was, convert to UTF-32
// and then do an easy compare!
std::u32string str;
if(!utf8::toUTF32(chars, str)) throw std::runtime_error("Specified invalid UTF-8 string!");
// compare now
isize index;
for(index = 0; index < str.size(); index++) {
if(str[index] != peekChar(index)) return false;
}
if(index == chars.length()) {
nextChar(count);
return true;
}
return false;
// success!
nextChar(index);
return true;
}
char TextReader::readByte() {
@@ -164,14 +166,14 @@ namespace spider {
bytes[bindex] = readByte();
if (err) return false;
if (eof) return false;
if (!utf8::isCont(u8(bytes[bindex]))) {
err = true;
errmsg = "Invalid continuation of UTF-8 sequence.";
return false;
}
}
u32 decodedChar = utf8::decodeArr(bytes, chsize);
u32 decodedChar;
if(!utf8::decodeArr(bytes, chsize, decodedChar)) {
err = true;
errmsg = "Invalid UTF-8 sequence.";
return false;
}
buffer.push_back(decodedChar);
return true;
}
+183
View File
@@ -0,0 +1,183 @@
#include "Token.hpp"
namespace spider {
// ============================================================================
// Token Implementation
// ============================================================================
SeqToken Token::operator&(const Token& tok) {
return SeqToken({ tok, *this });
}
OrToken Token::operator|(const Token& tok) {
return OrToken({ tok, *this });
}
OptToken Token::operator~() {
return OptToken(*this);
}
RepToken Token::operator*() {
return RepToken(*this);
}
// ============================================================================
// LitToken Implementation
// ============================================================================
LitToken::LitToken(std::string_view lit) {
if (!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
}
LitToken::LitToken(std::u32string lit) : literal(std::move(lit)) {}
TokenResult LitToken::test(TokenContext& ctx) const {
// Safety check: Prevent out-of-bounds pointer slicing if the remaining
// input is smaller than the target literal.
if (ctx.cursor + literal.size() > ctx.input.size()) {
return { false, {} };
}
// Window extract optimization: Acquire a zero-copy view over the input segment
std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size());
// Direct lexicographical verification of the UTF-32 code-point sequence
if (sub == literal) {
ctx.advance(literal.size());
return { true, std::u32string(sub) }; // Deep copy payload returned per requirement
}
return { false, {} };
}
// ============================================================================
// SeqToken Implementation
// ============================================================================
SeqToken::SeqToken(const ilist<ref<const Token>>& list) : tokens(list) {}
TokenResult SeqToken::test(TokenContext& ctx) const {
const size_t transactional_fallback_pos = ctx.cursor;
std::u32string accumulated_match;
// All matching steps within a sequence must pass consecutively.
for (const auto& token_ref : tokens) {
TokenResult res = token_ref.get().test(ctx);
if (!res.success) {
// Strict ACID Transaction: Roll back context pointer entirely
// if any nested condition in the sequence fails.
ctx.cursor = transactional_fallback_pos;
return { false, {} };
}
// Piecewise accumulation of individual matching sub-tokens
accumulated_match += res.match;
}
return { true, accumulated_match };
}
SeqToken SeqToken::operator&(const Token& tok) {
// Intrusive chaining optimization: Appends the next token directly into the existing
// registry vector instead of nesting structures, keeping the layout flattened.
tokens.push_back(std::cref(tok));
return *this;
}
// ============================================================================
// OrToken Implementation
// ============================================================================
OrToken::OrToken(const ilist<ref<const Token>>& list) : tokens(list) {}
TokenResult OrToken::test(TokenContext& ctx) const {
const size_t local_fallback_pos = ctx.cursor;
// Ordered choice evaluation: Evaluate variants sequentially.
for (const auto& token_ref : tokens) {
TokenResult res = token_ref.get().test(ctx);
if (res.success) {
return res; // Short-circuit branch: return immediately on first valid choice match
}
// Backtrack isolation: Reset the cursor position before testing the next alternative path
ctx.cursor = local_fallback_pos;
}
return { false, {} };
}
OrToken OrToken::operator|(const Token& tok) {
// Intrusive grouping layout optimization:
// flattens alternative tokens at code evaluation time.
tokens.push_back(tok);
return *this;
}
// ============================================================================
// OptToken Implementation
// ============================================================================
OptToken::OptToken(const Token& t) : target(t) {}
TokenResult OptToken::test(TokenContext& ctx) const {
const size_t local_fallback_pos = ctx.cursor;
TokenResult res = target.test(ctx);
if (res.success) {
return res; // Option matched exactly 1 instance successfully
}
// Recovery path: If sub-rule fails, clean up the dirty state mutation
// and successfully return an empty match payload (0 instances).
ctx.cursor = local_fallback_pos;
return { true, U"" };
}
OptToken OptToken::operator~() {
// Redundant layer trap protection: returning self
// prevents wrapping an Optional in an Optional
return *this;
}
// ============================================================================
// RepToken Implementation
// ============================================================================
RepToken::RepToken(const Token& t) : target(t) {}
TokenResult RepToken::test(TokenContext& ctx) const {
std::u32string accumulated_match;
// Greedily consume matches while input stream headroom remains
while (ctx.has_more()) {
const size_t pre_loop_cursor = ctx.cursor;
TokenResult res = target.test(ctx);
// Guard Clause: Break loop if sub-parser signals failure, or if an empty-matching
// rule succeeded without advancing the buffer index (prevents dynamic parsing lockups).
if (!res.success || ctx.cursor == pre_loop_cursor) {
ctx.cursor = pre_loop_cursor; // Revert cursor to last healthy match checkpoint
break;
}
accumulated_match += res.match;
}
// Repetition rules (* token) always evaluate to successful completion state, even with 0 matches.
return { true, accumulated_match };
}
RepToken RepToken::operator*() {
// Redundant layer trap protection: returning self prevents wrapping a Repetition rule inside a Repetition rule
return *this;
}
}
+136 -79
View File
@@ -1,8 +1,8 @@
#pragma once
#include "spider/compiler/common.hpp"
#include <spider/compiler/common.hpp>
#include "spider/compiler/text/utf8.hpp"
#include <spider/compiler/text/utf8.hpp>
namespace spider {
@@ -12,146 +12,203 @@ namespace spider {
size_t cursor = 0;
bool has_more() const { return cursor < input.size(); }
char32_t peek() const { return input[cursor]; }
void advance(size_t n = 1) { cursor += n; }
};
/**
* @brief The structural payload returned by every parsing component execution.
*/
struct TokenResult {
/** @brief Indicates if the token composition successfully matched the input boundary. */
bool success;
/**
* @brief Holds the deep-copied UTF-32 matching substring upon victory.
* @note Returns empty when success is false.
*/
std::u32string match;
};
// Forward declarations required by the abstract interface for operator returns.
class SeqToken;
class OrToken;
class OptToken;
class RepToken;
/**
* @brief Pure virtual base class defining the EBNF combinator node contract.
* @details Implements structural immutability for thread-safe static allocations
* while enforcing fluency using intrusive member operator overloads.
*/
class Token {
public:
/** @brief Virtual destructor ensuring clean polymorphic destruction of composite graphs. */
virtual ~Token() = default;
virtual TokenResult parse(TokenContext& ctx) const = 0;
public:
/**
* @brief Evaluates this token node against the provided context.
* @param ctx The current state tracker containing the UTF-32 viewing buffer and cursor.
* @return TokenResult Containing verification state and the parsed copy of matching data.
* @note Pure virtual; implementation details handle node-specific combinator semantics.
*/
virtual TokenResult test(TokenContext& ctx) const = 0;
public:
/**
* @brief Chains this token and another sequentially.
* @return A temporary structural bridge matching both tokens sequentially.
*/
virtual SeqToken operator&(const Token& tok);
/**
* @brief Combines this token and another under alternation.
* @return A structural bridge matching either this token or the fallback selection.
*/
virtual OrToken operator|(const Token& tok);
/**
* @brief Wraps this node in an optional layout rule.
* @return A structure matching zero or one instances of this current node.
*/
virtual OptToken operator~();
/**
* @brief Wraps this node in a repetitive loop match framework.
* @return A structure matching zero or more occurrences of this current node.
*/
virtual RepToken operator*();
};
/**
* @brief Leaf terminal parser validating precise exact matching UTF-32 sequences.
*/
class LitToken : public Token {
private:
/** @brief The reference literal sequence being inspected. */
std::u32string literal;
public:
/**
* @brief Constructs a literal rule by transforming a standard UTF-8 string view.
* @throws std::runtime_error If incoming character boundaries contain invalid UTF-8 formatting.
*/
explicit LitToken(std::string_view lit);
explicit LitToken(std::string_view lit) {
if(!utf8::toUTF32(lit, literal)) throw std::runtime_error("Illegal UTF8 literal!");
}
explicit LitToken(std::u32string lit) : literal(std::move(lit)) {}
/** @brief Direct zero-conversion construction using an existing native UTF-32 literal. */
explicit LitToken(std::u32string lit);
public:
TokenResult parse(TokenContext& ctx) const override {
if (ctx.cursor + literal.size() > ctx.input.size()) return { false, {} };
std::u32string_view sub = ctx.input.substr(ctx.cursor, literal.size());
if (sub == literal) {
ctx.advance(literal.size());
return { true, std::u32string(sub) };
}
return { false, {} };
}
/**
* @brief Validates match of the backing u32string exactly at the context cursor pointer.
* @details Advances the context cursor precisely by literal length on success; zero state mutation on failure.
*/
TokenResult test(TokenContext& ctx) const override;
};
// AndToken references existing static tokens rather than managing their lifecycles
class AndToken : public Token {
/**
* @brief Evaluates an unrolled sequence of ordered grammatical rules sequentially (AndToken).
*/
class SeqToken : public Token {
private:
const Token& lhs;
const Token& rhs;
/**
* @brief Internal contiguous layout registry storing lightweight, zero-overhead references.
* @details Avoids heap allocation penalties by referencing static instances immutably.
*/
vector<ref<const Token>> tokens;
public:
AndToken(const Token& l, const Token& r) : lhs(l), rhs(r) {}
/** @brief Constructs a sequence container directly out of an inline brace-enclosed listing. */
SeqToken(const ilist<ref<const Token>>& list);
public:
/**
* @brief Iterates the internal sequence, enforcing that every child node must pass sequentially.
* @details Implements a strict transaction boundary: if any internal element fails, the index
* backtracks entirely to its starting cursor value before returning failure.
*/
TokenResult test(TokenContext& ctx) const override;
TokenResult parse(TokenContext& ctx) const override {
size_t start_pos = ctx.cursor;
if (!lhs.parse(ctx).success) return { false, {} };
if (!rhs.parse(ctx).success) {
ctx.cursor = start_pos; // Backtrack
return { false, {} };
}
return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) };
}
/** @brief Intrusive override for fluid cascading sequencing (`rule1 & rule2 & rule3`). */
SeqToken operator&(const Token& tok) override;
};
/**
* @brief Implements standard alternation selection matching rules (OrToken).
*/
class OrToken : public Token {
private:
const Token& lhs;
const Token& rhs;
/** @brief Ordered registry of possible alternate structural paths. */
vector<ref<const Token>> tokens;
public:
OrToken(const Token& l, const Token& r) : lhs(l), rhs(r) {}
/** @brief Constructs an alternation choice layout from brace-enclosed tokens. */
OrToken(const ilist<ref<const Token>>& list);
public:
/**
* @brief Scans through alternatives, resolving immediately on the first candidate that passes.
* @details Safely rolls back changes to the context cursor point between failed alternative attempts.
*/
TokenResult test(TokenContext& ctx) const override;
TokenResult parse(TokenContext& ctx) const override {
size_t start_pos = ctx.cursor;
auto res = lhs.parse(ctx);
if (res.success) return res;
ctx.cursor = start_pos; // Backtrack
return rhs.parse(ctx);
}
/** @brief Intrusive override for cascading alternation chains (`ruleA | ruleB | ruleC`). */
OrToken operator|(const Token& tok) override;
};
/**
* @brief Structurally represents an optional token condition rule layer (OptToken).
*/
class OptToken : public Token {
private:
/** @brief Read-only target node reference to test optional status against. */
const Token& target;
public:
explicit OptToken(const Token& t) : target(t) {}
/** @brief Binds the target node rule structural dependency layout wrapper. */
explicit OptToken(const Token& t);
public:
/**
* @brief Evaluates target presence. Returns success true regardless of sub-rule evaluation outcome.
* @details If the nested rule fails, the context cursor rolls back to initial state, returning empty matches.
*/
TokenResult test(TokenContext& ctx) const override;
TokenResult parse(TokenContext& ctx) const override {
size_t start_pos = ctx.cursor;
if (target.parse(ctx).success) return {
true,
std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos))
};
ctx.cursor = start_pos;
return { true, std::u32string(ctx.input.substr(start_pos, 0)) };
}
/** @brief Stub override providing standard compliance with the base Token interface signature. */
OptToken operator~() override;
};
/**
* @brief Evaluates zero-to-infinite loops of a single repetitive match signature (RepToken).
*/
class RepToken : public Token {
private:
/** @brief The base token node sequence layer evaluated in loops. */
const Token& target;
public:
explicit RepToken(const Token& t) : target(t) {}
/** @brief Binds the repeated structural blueprint node wrapper configuration. */
explicit RepToken(const Token& t);
public:
/**
* @brief Greedily attempts to match target repeatedly until an execution failure boundary is hit.
* @details Includes internal safety loops checking cursor delta advancement to guarantee infinite
* empty-matching sub-loops do not cause thread lockups.
*/
TokenResult test(TokenContext& ctx) const override;
TokenResult parse(TokenContext& ctx) const override {
size_t start_pos = ctx.cursor;
while (ctx.has_more()) {
size_t loop_start = ctx.cursor;
if (!target.parse(ctx).success || ctx.cursor == loop_start) {
ctx.cursor = loop_start;
break;
}
}
return { true, std::u32string(ctx.input.substr(start_pos, ctx.cursor - start_pos)) };
}
/** @brief Stub override providing standard compliance with the base Token interface signature. */
RepToken operator*() override;
};
}
+147 -56
View File
@@ -26,24 +26,13 @@ namespace spider {
return (c & 0xC0) == 0x80;
}
constexpr isize isValidSeq(const char* src, isize len) {
if (len == 0) return 0;
isize m = seqlen(u8(src[0]));
if (m == 0 || m > len) return 0;
for (isize i = 1; i < m; i++) {
if (!isCont(u8(src[i]))) return 0;
}
return m;
}
// ----------------- //
// UTF-8 into UTF-32 //
// ----------------- //
inline isize decode(const char* src, isize len, u32& out) {
// check input is valid
isize charlen = isValidSeq(src, len);
if (charlen == 0) return 0;
inline bool decodeArr(const char* src, isize chlen, u32& out) {
// Check character length
if (chlen < 1 || chlen > 4) return false;
// map of masks, starts at 1
static constexpr u8 firstMask[5] = {
@@ -54,63 +43,165 @@ namespace spider {
0x07 // 11110xxx
};
// assemble the char
out = u8(src[0]) & firstMask[charlen];
for (isize i = 1; i < charlen; ++i) {
out <<= 6;
out |= u8(src[i]) & 0x3F;
}
return charlen;
}
/**
* A simpler version, which consider it already
* having a validated input array
*/
inline u32 decodeArr(const char* src, isize chlen) {
// map of masks, starts at 1
static constexpr u8 firstMask[5] = {
0x00, // unused
0x7F, // 0xxxxxxx
0x1F, // 110xxxxx
0x0F, // 1110xxxx
0x07 // 11110xxx
};
// assemble the char
u32 out = u8(src[0]) & firstMask[chlen];
u32 result = u8(src[0]) & firstMask[chlen];
for (isize i = 1; i < chlen; ++i) {
out <<= 6;
out |= u8(src[i]) & 0x3F;
if (!isCont(u8(src[i]))) return false;
result = (result << 6) | (u8(src[i]) & 0x3F);
}
return out;
}
inline bool charAt(std::string_view str, isize& index, u32& out) {
isize chlen = isValidSeq(str.begin(), str.size());
if(chlen == 0) return false;
out = decodeArr(str.begin(), chlen);
index += chlen;
// Security & Constraints Validation
// So, as it turns out, you can define UTF-8
// characters to be longer than they should.
// This protects against the (albeit niche) hack
// of sending sensitive chars like "\", "<", etc.
// and have a parser not target them.
// So take this: ALWAYS use UTF-32 to make comparisons!
// Byte-sequences are NOT to be trusted NEVER.
// Still, this ensures it's DOUBLE correct.
if (chlen == 2 && result < 0x80) return false; // Overlong 2-byte
if (chlen == 3 && result < 0x0800) return false; // Overlong 3-byte
if (chlen == 4 && result < 0x10000) return false; // Overlong 4-byte
if (result >= 0xD800 && result <= 0xDFFF) return false; // UTF-16 Surrogates
if (result > 0x10FFFF) return false; // Out of Unicode bounds
out = result;
return true;
}
inline isize decode(const char* src, isize len, u32& out) {
if (len <= 0) return 0;
isize m = seqlen(u8(src[0]));
if (m == 0 || m > len) return 0;
if (decodeArr(src, m, out)) return m;
return 0;
}
inline bool toUTF32(std::string_view str, std::u32string& out) {
isize _i = 0, _size;
isize _i = 0;
isize csize = str.size();
const char* data = str.data();
auto cptr = str.cbegin();
auto csize = str.size();
while (_i < csize) {
u32 codepoint;
isize _size = decode(data + _i, csize - _i, codepoint);
if (_size == 0) return false;
while(_i < csize) {
_size = isValidSeq(cptr + _i, csize - _i);
if(_size == 0) return false;
out += decodeArr(cptr + _i, _size);
out += char32_t(codepoint);
_i += _size;
}
return _i == csize;
}
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {}
inline const char* getControlCharName(u8 c) {
static const char* names[32] = {
"NUL", "SOH", "STX", "ETX", "EOT", "ENQ", "ACK", "BEL",
"BS", "HT", "LF", "VT", "FF", "CR", "SO", "SI",
"DLE", "DC1", "DC2", "DC3", "DC4", "NAK", "SYN", "ETB",
"CAN", "EM", "SUB", "ESC", "FS", "GS", "RS", "US"
};
if (c < 32) return names[c];
if (c == 127) return "DEL";
return nullptr;
}
inline void hexdump(const char* data, isize length, pos at, std::ostream& ostr) {
auto old_flags = ostr.flags();
auto old_fill = ostr.fill();
isize i = 0;
while (i < length) {
u8 lead = u8(data[i]);
isize m = seqlen(lead);
// Byte, Line, Col
ostr << std::setfill('0') << std::hex << std::uppercase;
ostr << "0x" << std::setw(8) << (at.byteoff + i);
ostr << std::setfill(' ') << std::dec;
ostr << " (" << std::setw(3) << at.line << ", " << std::setw(3) << at.col << ") : ";
// 1. Invalid Lead Byte handling
if (m == 0) {
ostr << std::setfill('0') << std::hex << std::uppercase;
ostr << "0x" << std::setw(2) << u32(lead);
ostr << std::setw(15) << std::setfill(' ') << " "; // Pad to match normal spacing
ostr << " : INVALID LEAD\n";
if (lead == '\n') { at.line++; at.col = 1; } else { at.col++; }
i++;
continue;
}
// 2. Truncated Sequence handling (Not enough bytes left in buffer)
if ((i + m) > length) {
isize available = length - i;
// Print what we can see
for (isize j = 0; j < available; ++j) {
ostr << "0x" << std::setw(2) << std::setfill('0') << std::hex << int(u8(data[i + j])) << " ";
}
// Fill missing bytes with ??
for (isize j = available; j < m; ++j) {
ostr << "0x?? ";
}
// Pad the rest of the column width
isize spaces_to_print = 20 - (m * 5);
if(spaces_to_print) ostr << std::setw(i32(spaces_to_print)) << std::setfill(' ') << " ";
ostr << ": TRUNC SEQ\n";
// Update row tracking based on what we actually processed
for (isize j = 0; j < available; ++j) {
if (u8(data[i + j]) == '\n') { at.line++; at.col = 1; } else { at.col++; }
}
i += available;
continue;
}
// 3. Fully Available Sequence Processing
u32 codepoint = 0;
bool valid = decodeArr(data + i, m, codepoint);
for (isize j = 0; j < m; ++j) {
ostr << "0x" << std::setw(2) << std::setfill('0') << std::hex << int(u8(data[i + j])) << " ";
}
isize spaces_to_print = 20 - (m * 5);
if(spaces_to_print) ostr << std::setw(i32(spaces_to_print)) << std::setfill(' ') << " ";
if (valid) {
ostr << ": U+" << std::setw(5) << std::setfill('0') << std::uppercase << std::hex << codepoint << " ";
// Check for control characters
const char* ctrlName = getControlCharName(lead); // Multi-byte UTF-8 can't be ASCII control chars
if (m == 1 && ctrlName != nullptr) {
ostr << "(" << ctrlName << ")\n";
} else {
ostr.write(data + i, i64(m));
ostr << "\n";
}
// Track position updates
if (m == 1 && lead == '\n') {
at.line++;
at.col = 1;
} else {
at.col++;
}
} else {
ostr << ": INVALID SEQ\n";
at.col++;
}
i += m;
}
ostr.flags(old_flags);
ostr.fill(old_fill);
at.byteoff += length;
}
}