utf-8 hexdump implmented
This commit is contained in:
@@ -7,6 +7,56 @@ namespace spider {
|
||||
|
||||
|
||||
|
||||
}
|
||||
|
||||
// Test runner helper
|
||||
void run_test(const std::string& name, const std::string& input) {
|
||||
std::cout << "========================================\n";
|
||||
std::cout << " TEST: " << name << "\n";
|
||||
std::cout << "========================================\n";
|
||||
|
||||
spider::pos tracking_pos;
|
||||
spider::utf8::hexdump(input.data(), input.size(), tracking_pos, std::cout);
|
||||
std::cout << "\n";
|
||||
}
|
||||
|
||||
void utf8sequences() {
|
||||
// Permutation 1: Standard, valid mixed UTF-8 sequences (1, 2, 3, and 4 bytes)
|
||||
// - 'A' -> 1 byte (U+0041)
|
||||
// - '¢' (cents) -> 2 bytes (U+00A2)
|
||||
// - '€' (euro) -> 3 bytes (U+20AC)
|
||||
// - '𐍈' (gothic) -> 4 bytes (U+10348)
|
||||
run_test("Valid Mixed Length Sequences", "A\xC2\xA2\xE2\x82\xAC\xF0\x90\x8D\x88");
|
||||
|
||||
// Permutation 2: Embedded Control Characters
|
||||
// Should display mnemonics like (HT), (LF), (CR) without breaking formatting
|
||||
run_test("ASCII Control Characters", "Text\tWith\r\nNewlines");
|
||||
|
||||
// Permutation 3: Invalid Lead Byte
|
||||
// The byte 0xFF is structurally illegal under any UTF-8 definition.
|
||||
// Expected behavior: Display single byte as INVALID LEAD, shift 1 byte over.
|
||||
run_test("Invalid Lead Byte (0xFF)", "ABC\xFFXYZ");
|
||||
|
||||
// Permutation 4: Invalid Continuation Sequence
|
||||
// A 3-byte header (\xE2) where the second byte (\x00) is a bad continuation.
|
||||
// Expected behavior: Show the entire sequence up to 'm' bytes, flag as INVALID SEQUENCE.
|
||||
run_test("Invalid Continuation Structure", std::string("Before \xE2\x00\xAC After", 16));
|
||||
|
||||
// Permutation 5: Truncated Sequence at End-of-Buffer
|
||||
// A 4-byte emoji header (\xF0\x9F) but the string completely cuts off.
|
||||
// Expected behavior: Display remaining space placeholders as '??' -> TRUNCATED SEQUENCE.
|
||||
run_test("Truncated Sequence (Missing trailing bytes)", "Hello \xF0\x9F");
|
||||
|
||||
// Permutation 6: Overlong Encoding Security Vulnerability
|
||||
// Attempting to write ASCII 'I' (normally 0x49) using 2 bytes: \xC1\x89
|
||||
// Expected behavior: Caught by constraints checks, flagged as INVALID SEQUENCE.
|
||||
run_test("Security Hack: Overlong Encoding", "Safe\xC1\x89Hack");
|
||||
|
||||
// Permutation 7: Out-of-bounds / Restricted Ranges
|
||||
// - \xED\xA0\x80 is a UTF-16 Surrogate (U+D800)
|
||||
// - \xF4\x90\x80\x80 is outside valid Unicode space (> U+10FFFF)
|
||||
// Expected behavior: Flagged securely as INVALID SEQUENCE.
|
||||
run_test("Security Hack: Restricted Ranges (Surrogates & Out-of-bounds)", "Surrogate: \xED\xA0\x80 MaxBounds: \xF4\x90\x80\x80");
|
||||
}
|
||||
|
||||
int main() {
|
||||
@@ -23,5 +73,8 @@ int main() {
|
||||
std::cout << std::endl;
|
||||
|
||||
std::cout << "Happy Day!" << std::endl;
|
||||
spider::utf8::hexdump(test.data(), test.size(), spider::pos(), std::cout);
|
||||
std::cout << std::endl;
|
||||
utf8sequences();
|
||||
return 0;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user