okay, changed structure a lot

This commit is contained in:
2026-08-04 08:05:52 -06:00
parent 3779848355
commit fa511520c6
5 changed files with 139 additions and 129 deletions
-66
View File
@@ -7,74 +7,8 @@ namespace spider {
}
// Test runner helper
void run_test(const std::string& name, const std::string& input) {
std::cout << "========================================\n";
std::cout << " TEST: " << name << "\n";
std::cout << "========================================\n";
spider::pos tracking_pos;
spider::utf8::hexdump(input.data(), input.size(), tracking_pos, std::cout);
std::cout << "\n";
}
void utf8sequences() {
// Permutation 1: Standard, valid mixed UTF-8 sequences (1, 2, 3, and 4 bytes)
// - 'A' -> 1 byte (U+0041)
// - '¢' (cents) -> 2 bytes (U+00A2)
// - '€' (euro) -> 3 bytes (U+20AC)
// - '𐍈' (gothic) -> 4 bytes (U+10348)
run_test("Valid Mixed Length Sequences", "A\xC2\xA2\xE2\x82\xAC\xF0\x90\x8D\x88");
// Permutation 2: Embedded Control Characters
// Should display mnemonics like (HT), (LF), (CR) without breaking formatting
run_test("ASCII Control Characters", "Text\tWith\r\nNewlines");
// Permutation 3: Invalid Lead Byte
// The byte 0xFF is structurally illegal under any UTF-8 definition.
// Expected behavior: Display single byte as INVALID LEAD, shift 1 byte over.
run_test("Invalid Lead Byte (0xFF)", "ABC\xFFXYZ");
// Permutation 4: Invalid Continuation Sequence
// A 3-byte header (\xE2) where the second byte (\x00) is a bad continuation.
// Expected behavior: Show the entire sequence up to 'm' bytes, flag as INVALID SEQUENCE.
run_test("Invalid Continuation Structure", std::string("Before \xE2\x00\xAC After", 16));
// Permutation 5: Truncated Sequence at End-of-Buffer
// A 4-byte emoji header (\xF0\x9F) but the string completely cuts off.
// Expected behavior: Display remaining space placeholders as '??' -> TRUNCATED SEQUENCE.
run_test("Truncated Sequence (Missing trailing bytes)", "Hello \xF0\x9F");
// Permutation 6: Overlong Encoding Security Vulnerability
// Attempting to write ASCII 'I' (normally 0x49) using 2 bytes: \xC1\x89
// Expected behavior: Caught by constraints checks, flagged as INVALID SEQUENCE.
run_test("Security Hack: Overlong Encoding", "Safe\xC1\x89Hack");
// Permutation 7: Out-of-bounds / Restricted Ranges
// - \xED\xA0\x80 is a UTF-16 Surrogate (U+D800)
// - \xF4\x90\x80\x80 is outside valid Unicode space (> U+10FFFF)
// Expected behavior: Flagged securely as INVALID SEQUENCE.
run_test("Security Hack: Restricted Ranges (Surrogates & Out-of-bounds)", "Surrogate: \xED\xA0\x80 MaxBounds: \xF4\x90\x80\x80");
} }
int main() { int main() {
std::string test = "UTF-STR WITH EMOJIS 😀🚀";
// Extracted UTF-8 chars:
std::u32string out;
bool r = spider::utf8::toUTF32(test, out);
std::cout << "INPUT: " << test << std::endl;
std::cout << "RESULT: " << int(r) << std::endl;
for(spider::u32 ch : out) {
std::cout << ch << " ";
}
std::cout << std::endl;
std::cout << "Happy Day!" << std::endl;
spider::utf8::hexdump(test.data(), test.size(), spider::pos(), std::cout);
std::cout << std::endl;
utf8sequences();
return 0; return 0;
} }
+61 -61
View File
@@ -4,7 +4,7 @@ namespace spider::asm_ebnf {
// Token Factory // Token Factory
static TokenFactory tf; TokenFactory tf;
// Char Functions // Char Functions
@@ -29,47 +29,47 @@ namespace spider::asm_ebnf {
} }
// (* Characters & Basic Predicates *) // (* Characters & Basic Predicates *)
static const Token* letter = tf.fn(isUTF8Alpha); const Token* letter = tf.fn(isUTF8Alpha);
static const Token* digit = tf.choice("0123456789"); const Token* digit = tf.choice("0123456789");
static const Token* alpha_num_char = tf.choice({ letter, digit }); const Token* alpha_num_char = tf.choice({ letter, digit });
static const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef"); const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef");
static const Token* octal_digit = tf.choice("01234567"); const Token* octal_digit = tf.choice("01234567");
static const Token* binary_digit = tf.choice("01"); const Token* binary_digit = tf.choice("01");
static const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf); const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf);
static const Token* ws_optional = tf.rep(ws_char); const Token* ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true);
static const Token* whitespace = tf.seq({ ws_char, tf.rep(ws_char) }); const Token* whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true);
static const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] }); const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
static const Token* utf8_char = tf.fn(isUTF8CharNotCrLf); const Token* utf8_char = tf.fn(isUTF8CharNotCrLf);
static const Token* char_escape = tf.seq({ tf["\\"], utf8_char }); const Token* char_escape = tf.seq({ tf["\\"], utf8_char });
static const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) }); const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
static const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] }); const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
static const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) }); const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
static const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] }); const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
// (* Literals *) // (* Literals *)
static const Token* identifier = tf.seq({ const Token* identifier = tf.tag(tf.seq({
tf.choice({ letter, tf["_"] }), tf.choice({ letter, tf["_"] }),
tf.rep(tf.choice({ alpha_num_char, tf["_"] })) tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
}); }), "identifier", true);
static const Token* comment = tf.seq({ tf[";"], tf.rep(utf8_char) }); const Token* comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true);
static const Token* sign = tf.choice("+-"); const Token* sign = tf.choice("+-");
static const Token* exponent_marker = tf.choice("eE"); const Token* exponent_marker = tf.choice("eE");
static const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) }); const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
static const Token* decimal_lit = tf.seq({ const Token* decimal_lit = tf.tag(tf.seq({
tf.opt(sign), tf.opt(sign),
digit, digit,
tf.rep(digit), tf.rep(digit),
tf.opt(tf.choice("BSIL")) tf.opt(tf.choice("BSIL"))
}); }), "decimal_lit", true);
static const Token* float_lit = tf.seq({ const Token* float_lit = tf.tag(tf.seq({
tf.opt(sign), tf.opt(sign),
tf.choice({ tf.choice({
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }), tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
@@ -77,68 +77,68 @@ namespace spider::asm_ebnf {
tf.seq({ digit, tf.rep(digit), exponent }) tf.seq({ digit, tf.rep(digit), exponent })
}), }),
tf.opt(tf.choice("FD")) tf.opt(tf.choice("FD"))
}); }), "float_lit", true);
static const Token* hex_lit = tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }); const Token* hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true);
static const Token* octal_lit = tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }); const Token* octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true);
static const Token* binary_lit = tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }); const Token* binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true);
static const Token* literal = tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }); const Token* literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal");
static const Token* literal_cast = tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }); const Token* literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast");
static const Token* literal_decl = tf.choice({ literal, literal_cast }); const Token* literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl");
// (* Operands *) // (* Operands *)
static const Token* register_tok = tf.seq({ tf["R"], alpha_num_char }); const Token* register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true);
static const Token* addrm_ind = tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }); const Token* addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true);
static const Token* addrm_ptr = tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }); const Token* addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true);
static const Token* addrm_idx = tf.seq({ const Token* addrm_idx = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional, tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
}); }), "addrm_idx", true);
static const Token* addrm_sca = tf.seq({ const Token* addrm_sca = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional, tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional, tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"] tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
}); }), "addrm_sca", true);
static const Token* addrm_dis = tf.seq({ const Token* addrm_dis = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional, tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional, tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional, tf["*"], ws_optional, literal_decl, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"] tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
}); }), "addrm_dis", true);
static const Token* addr_modes = tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }); const Token* addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm");
static const Token* operand = tf.choice({ register_tok, identifier, literal_decl, addr_modes }); const Token* operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand");
// (* Generalized Instructions *) // (* Generalized Instructions *)
static const Token* opcode = tf.seq({ letter, tf.rep(alpha_num_char) }); const Token* opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
static const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) }); const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
static const Token* instruction = tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) }); const Token* instruction = tf.tag(tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction");
// (* Added Preprocessor, Annotation *) // (* Added Preprocessor, Annotation *)
static const Token* annotation_named = tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }); const Token* annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named");
static const Token* annotation_arg = tf.choice({ annotation_named, literal_decl }); const Token* annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg");
static const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) }); const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
static const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] }); const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
static const Token* annotation = tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }); const Token* annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation");
static const Token* preprocessor_val = tf.choice({ identifier, string_lit }); const Token* preprocessor_val = tf.choice({ identifier, literal_decl });
static const Token* preprocessor = tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }); const Token* preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor");
// (* Line Structure & Program *) // (* Line Structure & Program *)
static const Token* label = tf.seq({ identifier, tf[":"] }); const Token* label = tf.tag(tf.seq({ identifier, tf[":"] }), "label");
static const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) }); const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
static const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) }); const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
static const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction }); const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
static const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline }); const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
static const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) }); const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
static const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) }); const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) });
} }
+69 -2
View File
@@ -4,8 +4,75 @@
namespace spider::asm_ebnf { namespace spider::asm_ebnf {
extern LitToken letter; extern const Token* letter;
extern const Token* digit;
extern const Token* alpha_num_char;
void createTokens(); extern const Token* hex_digit;
extern const Token* octal_digit;
extern const Token* binary_digit;
extern const Token* ws_char;
extern const Token* ws_optional;
extern const Token* whitespace;
extern const Token* newline;
extern const Token* utf8_char;
extern const Token* char_escape;
extern const Token* char_content;
extern const Token* char_lit;
extern const Token* string_char;
extern const Token* string_lit;
extern const Token* identifier;
extern const Token* comment;
extern const Token* sign;
extern const Token* exponent_marker;
extern const Token* exponent;
extern const Token* decimal_lit;
extern const Token* float_lit;
extern const Token* hex_lit;
extern const Token* octal_lit;
extern const Token* binary_lit;
extern const Token* literal;
extern const Token* literal_cast;
extern const Token* literal_decl;
extern const Token* register_tok;
extern const Token* addrm_ind;
extern const Token* addrm_ptr;
extern const Token* addrm_idx;
extern const Token* addrm_sca;
extern const Token* addrm_dis;
extern const Token* addr_modes;
extern const Token* operand;
extern const Token* opcode;
extern const Token* operand_list;
extern const Token* instruction;
extern const Token* annotation_named;
extern const Token* annotation_arg;
extern const Token* annotation_args;
extern const Token* annotation_pars;
extern const Token* annotation;
extern const Token* preprocessor_val;
extern const Token* preprocessor;
extern const Token* label;
extern const Token* line_label;
extern const Token* line_annotation;
extern const Token* line_content;
extern const Token* line;
extern const Token* line_last;
extern const Token* program;
} }
+7
View File
@@ -63,6 +63,13 @@ namespace spider {
return t; return t;
} }
Token* TokenFactory::tag(const Token* target, std::string_view tagname, bool flatten = false) {
uptr<Token> p = std::make_unique<TagToken>(target, tagname, flatten);
auto t = p.get();
arena.emplace_back(std::move(p));
return t;
}
// ============================================================================ // ============================================================================
// LitToken Implementation // LitToken Implementation
// ============================================================================ // ============================================================================
+2
View File
@@ -94,6 +94,8 @@ namespace spider {
Token* rep(const Token* target); Token* rep(const Token* target);
Token* tag(const Token* target, std::string_view tagname, bool flatten = false);
}; };
/** /**