lots of improvements standby for the results
This commit is contained in:
@@ -0,0 +1,96 @@
|
||||
#include "xicu.hpp"
|
||||
#include <cstdio>
|
||||
#include <cstdint>
|
||||
|
||||
int main() {
|
||||
printf("Testing x-icu binary tables...\n\n");
|
||||
|
||||
// Test alphanumeric
|
||||
struct AlnumTest { uint32_t cp; bool expected; const char* desc; };
|
||||
AlnumTest alnum_tests[] = {
|
||||
{ 0x0041, true, "Latin capital A" },
|
||||
{ 0x007A, true, "Latin small z" },
|
||||
{ 0x0030, true, "Digit 0" },
|
||||
{ 0x0039, true, "Digit 9" },
|
||||
{ 0x00E9, true, "Latin small e with acute" },
|
||||
{ 0x4E2D, true, "CJK ideograph" },
|
||||
{ 0x03B1, true, "Greek alpha" },
|
||||
{ 0x0040, false, "At sign" },
|
||||
{ 0x0023, false, "Hash" },
|
||||
{ 0x0020, false, "Space" },
|
||||
{ 0x0009, false, "Tab" },
|
||||
{ 0x1F600, false, "Emoji (grinning face)" },
|
||||
{ 0x0627, true, "Arabic letter alef" },
|
||||
{ 0x0915, true, "Devanagari letter ka" },
|
||||
{ 0x3042, true, "Hiragana letter a" },
|
||||
};
|
||||
|
||||
printf("=== Alphanumeric Tests ===\n");
|
||||
int passed = 0, failed = 0;
|
||||
for (const auto& t : alnum_tests) {
|
||||
bool result = xicu::isAlphanumeric(t.cp);
|
||||
if (result == t.expected) {
|
||||
printf("PASS: U+%04X (%s) -> %s\n", t.cp, t.desc, result ? "true" : "false");
|
||||
passed++;
|
||||
} else {
|
||||
printf("FAIL: U+%04X (%s) -> got %s, expected %s\n", t.cp, t.desc,
|
||||
result ? "true" : "false", t.expected ? "true" : "false");
|
||||
failed++;
|
||||
}
|
||||
}
|
||||
|
||||
// Test whitespace
|
||||
struct WsTest { uint32_t cp; bool expected; const char* desc; };
|
||||
WsTest ws_tests[] = {
|
||||
{ ' ', true, "Space" },
|
||||
{ '\t', true, "Tab" },
|
||||
{ '\n', true, "Line feed" },
|
||||
{ '\r', true, "Carriage return" },
|
||||
{ '\f', true, "Form feed" },
|
||||
{ '\v', true, "Vertical tab" },
|
||||
{ 0x1C, true, "File separator" },
|
||||
{ 0x1D, true, "Group separator" },
|
||||
{ 0x1E, true, "Record separator" },
|
||||
{ 0x1F, true, "Unit separator" },
|
||||
{ 0x85, true, "Next line (NEL)" },
|
||||
{ 0xA0, true, "No-break space" },
|
||||
{ 'A', false, "Letter A" },
|
||||
{ '0', false, "Digit 0" },
|
||||
{ '@', false, "At sign" },
|
||||
{ 0x2000, true, "En quad" },
|
||||
{ 0x2001, true, "Em quad" },
|
||||
{ 0x2002, true, "En space" },
|
||||
{ 0x2003, true, "Em space" },
|
||||
{ 0x2004, true, "Three-per-em space" },
|
||||
{ 0x2005, true, "Four-per-em space" },
|
||||
{ 0x2006, true, "Six-per-em space" },
|
||||
{ 0x2007, true, "Figure space" },
|
||||
{ 0x2008, true, "Punctuation space" },
|
||||
{ 0x2009, true, "Thin space" },
|
||||
{ 0x200A, true, "Hair space" },
|
||||
{ 0x2028, true, "Line separator" },
|
||||
{ 0x2029, true, "Paragraph separator" },
|
||||
{ 0x202F, true, "Narrow no-break space" },
|
||||
{ 0x205F, true, "Medium mathematical space" },
|
||||
{ 0x3000, true, "Ideographic space" },
|
||||
};
|
||||
|
||||
printf("\n=== Whitespace Tests ===\n");
|
||||
for (const auto& t : ws_tests) {
|
||||
bool result = xicu::isWhitespace(t.cp);
|
||||
if (result == t.expected) {
|
||||
printf("PASS: U+%04X (%s) -> %s\n", t.cp, t.desc, result ? "true" : "false");
|
||||
passed++;
|
||||
} else {
|
||||
printf("FAIL: U+%04X (%s) -> got %s, expected %s\n", t.cp, t.desc,
|
||||
result ? "true" : "false", t.expected ? "true" : "false");
|
||||
failed++;
|
||||
}
|
||||
}
|
||||
|
||||
printf("\n=== Summary ===\n");
|
||||
printf("Passed: %d\n", passed);
|
||||
printf("Failed: %d\n", failed);
|
||||
|
||||
return failed == 0 ? 0 : 1;
|
||||
}
|
||||
Binary file not shown.
@@ -0,0 +1,162 @@
|
||||
#include "xicu.hpp"
|
||||
#include <cstdio>
|
||||
#include <cstdint>
|
||||
|
||||
int main() {
|
||||
printf("Testing x-icu configurable generator...\n\n");
|
||||
|
||||
// Test isLetter (bitmap)
|
||||
struct { uint32_t cp; bool expected; const char* desc; } letter_tests[] = {
|
||||
{ 'A', true, "Latin capital A" },
|
||||
{ 'z', true, "Latin small z" },
|
||||
{ '0', false, "Digit 0" },
|
||||
{ '@', false, "At sign" },
|
||||
{ ' ', false, "Space" },
|
||||
{ 0x00E9, true, "e acute" },
|
||||
{ 0x4E2D, true, "CJK ideograph" },
|
||||
{ 0x03B1, true, "Greek alpha" },
|
||||
{ 0x0627, true, "Arabic alef" },
|
||||
{ 0x1F600, false, "Emoji" },
|
||||
};
|
||||
|
||||
printf("=== isLetter (bitmap) ===\n");
|
||||
int passed = 0, failed = 0;
|
||||
for (auto& t : letter_tests) {
|
||||
bool r = xicu::isLetter(t.cp);
|
||||
if (r == t.expected) { printf("PASS: U+%04X %s\n", t.cp, t.desc); passed++; }
|
||||
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
|
||||
}
|
||||
|
||||
// Test isPunctuation (sparse_map)
|
||||
struct { uint32_t cp; bool expected; const char* desc; } punct_tests[] = {
|
||||
{ '.', true, "Period" },
|
||||
{ ',', true, "Comma" },
|
||||
{ '!', true, "Exclamation" },
|
||||
{ '?', true, "Question" },
|
||||
{ 'A', false, "Letter A" },
|
||||
{ '0', false, "Digit 0" },
|
||||
{ ' ', false, "Space" },
|
||||
{ 0x2026, true, "Ellipsis" },
|
||||
{ 0x2018, true, "Left single quote" },
|
||||
{ 0x201C, true, "Left double quote" },
|
||||
};
|
||||
|
||||
printf("\n=== isPunctuation (sparse_map) ===\n");
|
||||
for (auto& t : punct_tests) {
|
||||
bool r = xicu::isPunctuation(t.cp);
|
||||
if (r == t.expected) { printf("PASS: U+%04X %s\n", t.cp, t.desc); passed++; }
|
||||
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
|
||||
}
|
||||
|
||||
// Test isWhitespace (sparse_map)
|
||||
struct { uint32_t cp; bool expected; const char* desc; } ws_tests[] = {
|
||||
{ ' ', true, "Space" },
|
||||
{ '\t', true, "Tab" },
|
||||
{ '\n', true, "LF" },
|
||||
{ '\r', true, "CR" },
|
||||
{ 0x00A0, true, "NBSP" },
|
||||
{ 0x2000, true, "En quad" },
|
||||
{ 0x2003, true, "Em space" },
|
||||
{ 0x2028, true, "Line separator" },
|
||||
{ 0x2029, true, "Paragraph separator" },
|
||||
{ 0x3000, true, "Ideographic space" },
|
||||
{ 'A', false, "Letter A" },
|
||||
{ '0', false, "Digit 0" },
|
||||
};
|
||||
|
||||
printf("\n=== isWhitespace (sparse_map) ===\n");
|
||||
for (auto& t : ws_tests) {
|
||||
bool r = xicu::isWhitespace(t.cp);
|
||||
if (r == t.expected) { printf("PASS: U+%04X %s\n", t.cp, t.desc); passed++; }
|
||||
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
|
||||
}
|
||||
|
||||
// Test isSpace (sparse_map) - only Zs
|
||||
struct { uint32_t cp; bool expected; const char* desc; } space_tests[] = {
|
||||
{ ' ', true, "Space" },
|
||||
{ 0x00A0, true, "NBSP" },
|
||||
{ 0x2000, true, "En quad" },
|
||||
{ 0x2003, true, "Em space" },
|
||||
{ 0x3000, true, "Ideographic space" },
|
||||
{ '\t', false, "Tab (not Zs)" },
|
||||
{ '\n', false, "LF (not Zs)" },
|
||||
{ 'A', false, "Letter A" },
|
||||
};
|
||||
|
||||
printf("\n=== isSpace (sparse_map) ===\n");
|
||||
for (auto& t : space_tests) {
|
||||
bool r = xicu::isSpace(t.cp);
|
||||
if (r == t.expected) { printf("PASS: U+%04X %s\n", t.cp, t.desc); passed++; }
|
||||
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
|
||||
}
|
||||
|
||||
// Test isDecimal (sparse_map)
|
||||
struct { uint32_t cp; bool expected; const char* desc; } dec_tests[] = {
|
||||
{ '0', true, "Digit 0" },
|
||||
{ '9', true, "Digit 9" },
|
||||
{ 'A', false, "Letter A" },
|
||||
{ 0x0660, true, "Arabic-Indic 0" },
|
||||
{ 0x0966, true, "Devanagari 0" },
|
||||
{ 0xFF10, true, "Fullwidth 0" },
|
||||
};
|
||||
|
||||
printf("\n=== isDecimal (sparse_map) ===\n");
|
||||
for (auto& t : dec_tests) {
|
||||
bool r = xicu::isDecimal(t.cp);
|
||||
if (r == t.expected) { printf("PASS: U+%04X %s\n", t.cp, t.desc); passed++; }
|
||||
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
|
||||
}
|
||||
|
||||
// Test getDecimal_val (delta_ranges)
|
||||
struct { uint32_t cp; int expected; const char* desc; } decval_tests[] = {
|
||||
{ '0', 0, "Digit 0" },
|
||||
{ '5', 5, "Digit 5" },
|
||||
{ '9', 9, "Digit 9" },
|
||||
{ 0x0665, 5, "Arabic-Indic 5" },
|
||||
{ 'A', 0, "Letter A" },
|
||||
};
|
||||
|
||||
printf("\n=== getDecimal_val (delta_ranges) ===\n");
|
||||
for (auto& t : decval_tests) {
|
||||
int r = xicu::getDecimal_val(t.cp);
|
||||
if (r == t.expected) { printf("PASS: U+%04X %s = %d\n", t.cp, t.desc, r); passed++; }
|
||||
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
|
||||
}
|
||||
|
||||
// Test getUppercase_map (sparse_map)
|
||||
struct { uint32_t cp; int expected; const char* desc; } up_tests[] = {
|
||||
{ 'a', 'A' - 'a', "a -> A" },
|
||||
{ 'z', 'Z' - 'z', "z -> Z" },
|
||||
{ 0x00E9, 0x00C9 - 0x00E9, "e acute -> E acute" },
|
||||
{ 'A', 0, "A (already upper)" },
|
||||
{ '0', 0, "Digit 0" },
|
||||
};
|
||||
|
||||
printf("\n=== getUppercase_map (sparse_map) ===\n");
|
||||
for (auto& t : up_tests) {
|
||||
int r = xicu::getUppercase_map(t.cp);
|
||||
if (r == t.expected) { printf("PASS: U+%04X %s = %d\n", t.cp, t.desc, r); passed++; }
|
||||
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
|
||||
}
|
||||
|
||||
// Test getLowercase_map (sparse_map)
|
||||
struct { uint32_t cp; int expected; const char* desc; } low_tests[] = {
|
||||
{ 'A', 'a' - 'A', "A -> a" },
|
||||
{ 'Z', 'z' - 'Z', "Z -> z" },
|
||||
{ 0x00C9, 0x00E9 - 0x00C9, "E acute -> e acute" },
|
||||
{ 'a', 0, "a (already lower)" },
|
||||
{ '0', 0, "Digit 0" },
|
||||
};
|
||||
|
||||
printf("\n=== getLowercase_map (sparse_map) ===\n");
|
||||
for (auto& t : low_tests) {
|
||||
int r = xicu::getLowercase_map(t.cp);
|
||||
if (r == t.expected) { printf("PASS: U+%04X %s = %d\n", t.cp, t.desc, r); passed++; }
|
||||
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
|
||||
}
|
||||
|
||||
printf("\n=== Summary ===\n");
|
||||
printf("Passed: %d\n", passed);
|
||||
printf("Failed: %d\n", failed);
|
||||
return failed == 0 ? 0 : 1;
|
||||
}
|
||||
Binary file not shown.
+259
@@ -0,0 +1,259 @@
|
||||
#include "xicu.hpp"
|
||||
#include <cstdio>
|
||||
#include <cstdint>
|
||||
#include <cstring>
|
||||
#include <string_view>
|
||||
|
||||
namespace xicu {
|
||||
|
||||
static constexpr uint32_t MAGIC = 0x58494355;
|
||||
static constexpr uint8_t VERSION = 2;
|
||||
|
||||
PropTable::PropTable(const char* path) { load(path); }
|
||||
|
||||
PropTable::~PropTable() {
|
||||
if (owns_data_ && data_) { delete[] data_; data_ = nullptr; }
|
||||
}
|
||||
|
||||
PropTable::PropTable(PropTable&& other) noexcept
|
||||
: data_(other.data_), data_size_(other.data_size_), owns_data_(other.owns_data_) {
|
||||
other.data_ = nullptr; other.owns_data_ = false;
|
||||
}
|
||||
|
||||
PropTable& PropTable::operator=(PropTable&& other) noexcept {
|
||||
if (this != &other) {
|
||||
if (owns_data_ && data_) delete[] data_;
|
||||
data_ = other.data_; data_size_ = other.data_size_; owns_data_ = other.owns_data_;
|
||||
other.data_ = nullptr; other.owns_data_ = false;
|
||||
}
|
||||
return *this;
|
||||
}
|
||||
|
||||
uint32_t PropTable::readVarint(const uint8_t*& ptr) {
|
||||
uint32_t val = 0; int shift = 0;
|
||||
while (true) {
|
||||
uint8_t b = *ptr++;
|
||||
val |= (b & 0x7F) << shift;
|
||||
if (!(b & 0x80)) break;
|
||||
shift += 7;
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
bool PropTable::load(const char* path) {
|
||||
if (owns_data_ && data_) { delete[] data_; data_ = nullptr; owns_data_ = false; }
|
||||
FILE* f = std::fopen(path, "rb");
|
||||
if (!f) return false;
|
||||
std::fseek(f, 0, SEEK_END);
|
||||
long fsize = std::ftell(f);
|
||||
std::fseek(f, 0, SEEK_SET);
|
||||
if (fsize < 6) { std::fclose(f); return false; }
|
||||
data_size_ = static_cast<size_t>(fsize);
|
||||
uint8_t* buf = new uint8_t[data_size_];
|
||||
size_t read = std::fread(buf, 1, data_size_, f);
|
||||
std::fclose(f);
|
||||
if (read != data_size_) { delete[] buf; return false; }
|
||||
data_ = buf; owns_data_ = true;
|
||||
|
||||
const uint8_t* ptr = data_;
|
||||
uint32_t magic = *reinterpret_cast<const uint32_t*>(ptr); ptr += 4;
|
||||
if (magic != MAGIC) return false;
|
||||
uint8_t version = *ptr++;
|
||||
if (version != VERSION) return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
// Find field data pointer (points to data length varint after name)
|
||||
const uint8_t* PropTable::findFieldData(const char* name, uint8_t& out_type, uint8_t& out_strat) const {
|
||||
if (!data_) return nullptr;
|
||||
const uint8_t* ptr = data_ + 5; // magic + version
|
||||
uint32_t num_fields = readVarint(ptr);
|
||||
for (uint32_t i = 0; i < num_fields; ++i) {
|
||||
uint8_t ftype = *ptr++;
|
||||
uint8_t fstrat = *ptr++;
|
||||
const char* fname = reinterpret_cast<const char*>(ptr);
|
||||
size_t name_len = std::strlen(fname);
|
||||
ptr += name_len + 1;
|
||||
if (std::strcmp(fname, name) == 0) {
|
||||
out_type = ftype;
|
||||
out_strat = fstrat;
|
||||
return ptr; // points to data length varint
|
||||
}
|
||||
// Skip data
|
||||
uint32_t data_len = readVarint(ptr);
|
||||
ptr += data_len;
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// Find string field (returns pointer to indices data length)
|
||||
const uint8_t* PropTable::findStringField(const char* name) const {
|
||||
if (!data_) return nullptr;
|
||||
const uint8_t* ptr = data_ + 5;
|
||||
uint32_t num_fields = readVarint(ptr);
|
||||
// Skip non-string fields (with inline data)
|
||||
for (uint32_t i = 0; i < num_fields; ++i) {
|
||||
ptr++; ptr++; // type, strat
|
||||
while (*ptr++) {} // skip name
|
||||
uint32_t data_len = readVarint(ptr);
|
||||
ptr += data_len; // skip data
|
||||
}
|
||||
uint32_t num_str_fields = readVarint(ptr);
|
||||
for (uint32_t i = 0; i < num_str_fields; ++i) {
|
||||
const char* fname = reinterpret_cast<const char*>(ptr);
|
||||
size_t name_len = std::strlen(fname);
|
||||
ptr += name_len + 1;
|
||||
if (std::strcmp(fname, name) == 0) {
|
||||
return ptr; // points to indices data length
|
||||
}
|
||||
uint32_t data_len = readVarint(ptr);
|
||||
ptr += data_len; // skip indices data
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// Decode bitmap: 1 bit per codepoint
|
||||
bool PropTable::decodeBitmap(const uint8_t* data, uint32_t cp) const {
|
||||
size_t byte_idx = cp >> 3;
|
||||
uint8_t bit = cp & 7;
|
||||
return (data[byte_idx] & (1 << bit)) != 0;
|
||||
}
|
||||
|
||||
// Decode delta ranges: (start, length, value) varint-encoded
|
||||
bool PropTable::decodeDeltaRangesBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {
|
||||
const uint8_t* ptr = data;
|
||||
while (ptr < data_end) {
|
||||
uint32_t start = readVarint(ptr);
|
||||
uint32_t length = readVarint(ptr);
|
||||
uint8_t val = *ptr++;
|
||||
uint32_t end = start + length - 1;
|
||||
if (cp >= start && cp <= end) return val != 0;
|
||||
if (cp < start) return false;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
int32_t PropTable::decodeDeltaRangesInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {
|
||||
const uint8_t* ptr = data;
|
||||
while (ptr < data_end) {
|
||||
uint32_t start = readVarint(ptr);
|
||||
uint32_t length = readVarint(ptr);
|
||||
uint32_t zigzag = readVarint(ptr);
|
||||
int32_t val = static_cast<int32_t>((zigzag >> 1) ^ -(zigzag & 1));
|
||||
uint32_t end = start + length - 1;
|
||||
if (cp >= start && cp <= end) return val;
|
||||
if (cp < start) return 0;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Decode sparse map: (cp, value) pairs
|
||||
bool PropTable::decodeSparseMapBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {
|
||||
const uint8_t* ptr = data;
|
||||
while (ptr < data_end) {
|
||||
uint32_t c = readVarint(ptr);
|
||||
if (ptr >= data_end) break;
|
||||
uint8_t val = *ptr++;
|
||||
if (c == cp) return val != 0;
|
||||
if (c > cp) return false;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
int32_t PropTable::decodeSparseMapInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {
|
||||
const uint8_t* ptr = data;
|
||||
while (ptr < data_end) {
|
||||
uint32_t c = readVarint(ptr);
|
||||
if (ptr >= data_end) break;
|
||||
uint32_t zigzag = readVarint(ptr);
|
||||
int32_t val = static_cast<int32_t>((zigzag >> 1) ^ -(zigzag & 1));
|
||||
if (c == cp) return val;
|
||||
if (c > cp) return 0;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
std::string_view PropTable::getName(uint32_t cp) const {
|
||||
// String fields use separate .str file
|
||||
const uint8_t* indices_ptr = findStringField("name");
|
||||
if (!indices_ptr) return "";
|
||||
uint32_t indices_len = readVarint(indices_ptr);
|
||||
// Find index for this codepoint (linear scan for now)
|
||||
// TODO: optimize with binary search if needed
|
||||
// For now, we need the string pool loaded separately
|
||||
return ""; // Requires string pool file
|
||||
}
|
||||
|
||||
bool PropTable::getPunctuation(uint32_t cp) const {
|
||||
uint8_t ftype, fstrat;
|
||||
const uint8_t* data = findFieldData("punctuation", ftype, fstrat);
|
||||
if (!data) return false;
|
||||
uint32_t data_len = readVarint(data);
|
||||
const uint8_t* data_end = data + data_len;
|
||||
return decodeSparseMapBool(data, data_end, cp);
|
||||
}
|
||||
|
||||
bool PropTable::getLetter(uint32_t cp) const {
|
||||
uint8_t ftype, fstrat;
|
||||
const uint8_t* data = findFieldData("letter", ftype, fstrat);
|
||||
if (!data) return false;
|
||||
uint32_t data_len = readVarint(data);
|
||||
const uint8_t* data_end = data + data_len;
|
||||
return decodeBitmap(data, cp);
|
||||
}
|
||||
|
||||
int32_t PropTable::getUppercase_map(uint32_t cp) const {
|
||||
uint8_t ftype, fstrat;
|
||||
const uint8_t* data = findFieldData("uppercase_map", ftype, fstrat);
|
||||
if (!data) return 0;
|
||||
uint32_t data_len = readVarint(data);
|
||||
const uint8_t* data_end = data + data_len;
|
||||
return decodeSparseMapInt(data, data_end, cp);
|
||||
}
|
||||
|
||||
int32_t PropTable::getLowercase_map(uint32_t cp) const {
|
||||
uint8_t ftype, fstrat;
|
||||
const uint8_t* data = findFieldData("lowercase_map", ftype, fstrat);
|
||||
if (!data) return 0;
|
||||
uint32_t data_len = readVarint(data);
|
||||
const uint8_t* data_end = data + data_len;
|
||||
return decodeSparseMapInt(data, data_end, cp);
|
||||
}
|
||||
|
||||
bool PropTable::getWhitespace(uint32_t cp) const {
|
||||
uint8_t ftype, fstrat;
|
||||
const uint8_t* data = findFieldData("whitespace", ftype, fstrat);
|
||||
if (!data) return false;
|
||||
uint32_t data_len = readVarint(data);
|
||||
const uint8_t* data_end = data + data_len;
|
||||
return decodeSparseMapBool(data, data_end, cp);
|
||||
}
|
||||
|
||||
bool PropTable::getSpace(uint32_t cp) const {
|
||||
uint8_t ftype, fstrat;
|
||||
const uint8_t* data = findFieldData("space", ftype, fstrat);
|
||||
if (!data) return false;
|
||||
uint32_t data_len = readVarint(data);
|
||||
const uint8_t* data_end = data + data_len;
|
||||
return decodeSparseMapBool(data, data_end, cp);
|
||||
}
|
||||
|
||||
bool PropTable::getDecimal(uint32_t cp) const {
|
||||
uint8_t ftype, fstrat;
|
||||
const uint8_t* data = findFieldData("decimal", ftype, fstrat);
|
||||
if (!data) return false;
|
||||
uint32_t data_len = readVarint(data);
|
||||
const uint8_t* data_end = data + data_len;
|
||||
return decodeSparseMapBool(data, data_end, cp);
|
||||
}
|
||||
|
||||
int32_t PropTable::getDecimal_val(uint32_t cp) const {
|
||||
uint8_t ftype, fstrat;
|
||||
const uint8_t* data = findFieldData("decimal_val", ftype, fstrat);
|
||||
if (!data) return 0;
|
||||
uint32_t data_len = readVarint(data);
|
||||
const uint8_t* data_end = data + data_len;
|
||||
return decodeDeltaRangesInt(data, data_end, cp);
|
||||
}
|
||||
|
||||
} // namespace xicu
|
||||
@@ -0,0 +1,85 @@
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
#include <string_view>
|
||||
|
||||
namespace xicu {
|
||||
|
||||
class PropTable {
|
||||
public:
|
||||
PropTable() = default;
|
||||
explicit PropTable(const char* path);
|
||||
~PropTable();
|
||||
|
||||
PropTable(const PropTable&) = delete;
|
||||
PropTable& operator=(const PropTable&) = delete;
|
||||
PropTable(PropTable&& other) noexcept;
|
||||
PropTable& operator=(PropTable&& other) noexcept;
|
||||
|
||||
bool load(const char* path);
|
||||
bool isLoaded() const { return data_ != nullptr; }
|
||||
std::string_view getName(uint32_t cp) const;
|
||||
bool getPunctuation(uint32_t cp) const;
|
||||
bool getLetter(uint32_t cp) const;
|
||||
int32_t getUppercase_map(uint32_t cp) const;
|
||||
int32_t getLowercase_map(uint32_t cp) const;
|
||||
bool getWhitespace(uint32_t cp) const;
|
||||
bool getSpace(uint32_t cp) const;
|
||||
bool getDecimal(uint32_t cp) const;
|
||||
int32_t getDecimal_val(uint32_t cp) const;
|
||||
|
||||
private:
|
||||
const uint8_t* data_ = nullptr;
|
||||
size_t data_size_ = 0;
|
||||
bool owns_data_ = false;
|
||||
|
||||
static uint32_t readVarint(const uint8_t*& ptr);
|
||||
const uint8_t* findFieldData(const char* name, uint8_t& out_type, uint8_t& out_strat) const;
|
||||
const uint8_t* findStringField(const char* name) const;
|
||||
|
||||
// Decoders
|
||||
bool decodeBitmap(const uint8_t* data, uint32_t cp) const;
|
||||
bool decodeDeltaRangesBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;
|
||||
int32_t decodeDeltaRangesInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;
|
||||
bool decodeSparseMapBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;
|
||||
int32_t decodeSparseMapInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;
|
||||
};
|
||||
|
||||
inline std::string_view getName(uint32_t cp) {
|
||||
static PropTable table("../out/xicu.bin");
|
||||
return table.getName(cp);
|
||||
}
|
||||
inline bool isPunctuation(uint32_t cp) {
|
||||
static PropTable table("../out/xicu.bin");
|
||||
return table.getPunctuation(cp);
|
||||
}
|
||||
inline bool isLetter(uint32_t cp) {
|
||||
static PropTable table("../out/xicu.bin");
|
||||
return table.getLetter(cp);
|
||||
}
|
||||
inline int32_t getUppercase_map(uint32_t cp) {
|
||||
static PropTable table("../out/xicu.bin");
|
||||
return table.getUppercase_map(cp);
|
||||
}
|
||||
inline int32_t getLowercase_map(uint32_t cp) {
|
||||
static PropTable table("../out/xicu.bin");
|
||||
return table.getLowercase_map(cp);
|
||||
}
|
||||
inline bool isWhitespace(uint32_t cp) {
|
||||
static PropTable table("../out/xicu.bin");
|
||||
return table.getWhitespace(cp);
|
||||
}
|
||||
inline bool isSpace(uint32_t cp) {
|
||||
static PropTable table("../out/xicu.bin");
|
||||
return table.getSpace(cp);
|
||||
}
|
||||
inline bool isDecimal(uint32_t cp) {
|
||||
static PropTable table("../out/xicu.bin");
|
||||
return table.getDecimal(cp);
|
||||
}
|
||||
inline int32_t getDecimal_val(uint32_t cp) {
|
||||
static PropTable table("../out/xicu.bin");
|
||||
return table.getDecimal_val(cp);
|
||||
}
|
||||
|
||||
} // namespace xicu
|
||||
BIN
Binary file not shown.
File diff suppressed because it is too large
Load Diff
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
File diff suppressed because one or more lines are too long
@@ -0,0 +1,117 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Generate binary tables for:
|
||||
a) is char X alphanumeric? (letter or decimal digit)
|
||||
b) is char whitespace?
|
||||
"""
|
||||
|
||||
import os
|
||||
import struct
|
||||
import urllib.request
|
||||
|
||||
DATA_DIR = './data'
|
||||
OUT_DIR = './out'
|
||||
|
||||
os.makedirs(DATA_DIR, exist_ok=True)
|
||||
os.makedirs(OUT_DIR, exist_ok=True)
|
||||
|
||||
UNICODE_DATA_URL = 'https://www.unicode.org/Public/UCD/latest/ucd/UnicodeData.txt'
|
||||
UNICODE_DATA_PATH = os.path.join(DATA_DIR, 'UnicodeData.txt')
|
||||
|
||||
if not os.path.exists(UNICODE_DATA_PATH):
|
||||
print(f"Downloading {UNICODE_DATA_URL}...")
|
||||
urllib.request.urlretrieve(UNICODE_DATA_URL, UNICODE_DATA_PATH)
|
||||
print("Downloaded")
|
||||
|
||||
def parse_unicode_data():
|
||||
"""Parse UnicodeData.txt and yield (codepoint, category, name)"""
|
||||
with open(UNICODE_DATA_PATH, 'r', encoding='utf-8') as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
parts = line.split(';')
|
||||
if len(parts) < 3:
|
||||
continue
|
||||
codepoint = int(parts[0], 16)
|
||||
name = parts[1]
|
||||
category = parts[2]
|
||||
|
||||
# Handle ranges (First/Last)
|
||||
if name.endswith(', First>'):
|
||||
range_start = (codepoint, category, name)
|
||||
continue
|
||||
elif name.endswith(', Last>') and 'range_start' in locals():
|
||||
start_cp, start_cat, start_name = range_start
|
||||
base_name = start_name.replace(', First>', '').replace('<', '')
|
||||
for cp in range(start_cp, codepoint + 1):
|
||||
yield (cp, start_cat, f"<{base_name}>")
|
||||
del range_start
|
||||
continue
|
||||
|
||||
yield (codepoint, category, name)
|
||||
|
||||
def is_alphanumeric(category: str, codepoint: int) -> bool:
|
||||
"""Check if category indicates alphanumeric (letter or decimal digit)"""
|
||||
return category.startswith('L') or category == 'Nd'
|
||||
|
||||
def is_whitespace(category: str, codepoint: int) -> bool:
|
||||
"""Check if character is whitespace"""
|
||||
if category in ('Zs', 'Zl', 'Zp'):
|
||||
return True
|
||||
# Common control whitespaces: \t\n\r\f\v and others
|
||||
return codepoint in (0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x1C, 0x1D, 0x1E, 0x1F, 0x85, 0xA0)
|
||||
|
||||
def build_bitmap(check_func, max_codepoint=0x10FFFF):
|
||||
"""Build a bitmap for the given check function"""
|
||||
# We'll use a byte array where each bit represents a codepoint
|
||||
size = (max_codepoint + 8) // 8
|
||||
bitmap = bytearray(size)
|
||||
|
||||
for cp, cat, _ in parse_unicode_data():
|
||||
if check_func(cat, cp):
|
||||
byte_idx = cp // 8
|
||||
bit_idx = cp % 8
|
||||
bitmap[byte_idx] |= (1 << bit_idx)
|
||||
|
||||
return bitmap
|
||||
|
||||
def write_binary_file(path: str, bitmap: bytearray, max_codepoint: int):
|
||||
"""Write binary file with header: magic, version, max_codepoint, data"""
|
||||
with open(path, 'wb') as f:
|
||||
# Magic: 'XICU' (0x58494355)
|
||||
f.write(struct.pack('<I', 0x58494355))
|
||||
# Version: 1
|
||||
f.write(struct.pack('<B', 1))
|
||||
# Max codepoint (4 bytes)
|
||||
f.write(struct.pack('<I', max_codepoint))
|
||||
# Data
|
||||
f.write(bitmap)
|
||||
|
||||
def main():
|
||||
MAX_CP = 0x10FFFF
|
||||
|
||||
print("Building alphanumeric bitmap...")
|
||||
alnum_bitmap = build_bitmap(is_alphanumeric, MAX_CP)
|
||||
|
||||
print("Building whitespace bitmap...")
|
||||
ws_bitmap = build_bitmap(is_whitespace, MAX_CP)
|
||||
|
||||
alnum_path = os.path.join(OUT_DIR, 'alphanumeric.bin')
|
||||
ws_path = os.path.join(OUT_DIR, 'whitespace.bin')
|
||||
|
||||
print(f"Writing {alnum_path} ({len(alnum_bitmap)} bytes)...")
|
||||
write_binary_file(alnum_path, alnum_bitmap, MAX_CP)
|
||||
|
||||
print(f"Writing {ws_path} ({len(ws_bitmap)} bytes)...")
|
||||
write_binary_file(ws_path, ws_bitmap, MAX_CP)
|
||||
|
||||
# Print stats
|
||||
alnum_count = sum(bin(b).count('1') for b in alnum_bitmap)
|
||||
ws_count = sum(bin(b).count('1') for b in ws_bitmap)
|
||||
print(f"Alphanumeric codepoints: {alnum_count}")
|
||||
print(f"Whitespace codepoints: {ws_count}")
|
||||
print("Done!")
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
+143
@@ -0,0 +1,143 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Generate a single binary table with delta encoding for:
|
||||
- isAlphanumeric: letter (L*) or decimal digit (Nd)
|
||||
- isWhitespace: Zs, Zl, Zp categories or control whitespace chars
|
||||
"""
|
||||
|
||||
import os
|
||||
import struct
|
||||
|
||||
DATA_DIR = './data'
|
||||
OUT_DIR = './out'
|
||||
|
||||
os.makedirs(DATA_DIR, exist_ok=True)
|
||||
os.makedirs(OUT_DIR, exist_ok=True)
|
||||
|
||||
UNICODE_DATA_PATH = os.path.join(DATA_DIR, 'UnicodeData.txt')
|
||||
|
||||
def parse_unicode_data():
|
||||
"""Parse UnicodeData.txt and yield (codepoint, category)"""
|
||||
with open(UNICODE_DATA_PATH, 'r', encoding='utf-8') as f:
|
||||
range_start = None
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
parts = line.split(';')
|
||||
if len(parts) < 3:
|
||||
continue
|
||||
codepoint = int(parts[0], 16)
|
||||
name = parts[1]
|
||||
category = parts[2]
|
||||
|
||||
if name.endswith(', First>'):
|
||||
range_start = (codepoint, category)
|
||||
continue
|
||||
elif name.endswith(', Last>') and range_start:
|
||||
start_cp, start_cat = range_start
|
||||
for cp in range(start_cp, codepoint + 1):
|
||||
yield (cp, start_cat)
|
||||
range_start = None
|
||||
continue
|
||||
|
||||
yield (codepoint, category)
|
||||
|
||||
def get_props(category: str, codepoint: int) -> int:
|
||||
"""Return 2-bit flags: bit0=alphanumeric, bit1=whitespace"""
|
||||
flags = 0
|
||||
# Alphanumeric: Letter (L*) or Decimal digit (Nd)
|
||||
if category.startswith('L') or category == 'Nd':
|
||||
flags |= 1 # bit 0
|
||||
# Whitespace: Z* (Zs, Zl, Zp) or control chars
|
||||
if category.startswith('Z') or codepoint in (0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x1C, 0x1D, 0x1E, 0x1F, 0x85, 0xA0):
|
||||
flags |= 2 # bit 1
|
||||
return flags
|
||||
|
||||
def build_delta_ranges():
|
||||
"""Build delta-encoded ranges: (start, end, flags)"""
|
||||
ranges = []
|
||||
current_start = None
|
||||
current_flags = None
|
||||
|
||||
for cp, cat in parse_unicode_data():
|
||||
flags = get_props(cat, cp)
|
||||
|
||||
if current_start is None:
|
||||
current_start = cp
|
||||
current_flags = flags
|
||||
elif flags != current_flags:
|
||||
# End previous range
|
||||
ranges.append((current_start, cp - 1, current_flags))
|
||||
current_start = cp
|
||||
current_flags = flags
|
||||
# else: continue current range
|
||||
|
||||
# Don't forget the last range
|
||||
if current_start is not None:
|
||||
ranges.append((current_start, 0x10FFFF, current_flags))
|
||||
|
||||
return ranges
|
||||
|
||||
def write_binary(path: str, ranges):
|
||||
"""Write delta-encoded binary format:
|
||||
Header: magic(4), version(1), num_ranges(4)
|
||||
Each range: start(4), end(4), flags(1) - but we'll pack efficiently
|
||||
|
||||
Optimized format:
|
||||
- magic: 'XICU' (0x58494355)
|
||||
- version: 1
|
||||
- num_ranges: uint32
|
||||
- For each range: varint start, varint length, flags(1 byte)
|
||||
"""
|
||||
def write_varint(f, val):
|
||||
while val >= 0x80:
|
||||
f.write(bytes([(val & 0x7F) | 0x80]))
|
||||
val >>= 7
|
||||
f.write(bytes([val]))
|
||||
|
||||
with open(path, 'wb') as f:
|
||||
f.write(struct.pack('<I', 0x58494355)) # magic 'XICU'
|
||||
f.write(struct.pack('<B', 1)) # version
|
||||
write_varint(f, len(ranges)) # num_ranges
|
||||
|
||||
for start, end, flags in ranges:
|
||||
length = end - start + 1
|
||||
write_varint(f, start)
|
||||
write_varint(f, length)
|
||||
f.write(bytes([flags]))
|
||||
|
||||
def read_varint(data, offset):
|
||||
val = 0
|
||||
shift = 0
|
||||
while True:
|
||||
b = data[offset]
|
||||
offset += 1
|
||||
val |= (b & 0x7F) << shift
|
||||
if not (b & 0x80):
|
||||
break
|
||||
shift += 7
|
||||
return val, offset
|
||||
|
||||
def main():
|
||||
print("Building delta-encoded ranges...")
|
||||
ranges = build_delta_ranges()
|
||||
|
||||
print(f"Total ranges: {len(ranges)}")
|
||||
|
||||
# Stats
|
||||
alnum_count = sum(end - start + 1 for start, end, f in ranges if f & 1)
|
||||
ws_count = sum(end - start + 1 for start, end, f in ranges if f & 2)
|
||||
print(f"Alphanumeric codepoints: {alnum_count}")
|
||||
print(f"Whitespace codepoints: {ws_count}")
|
||||
|
||||
out_path = os.path.join(OUT_DIR, 'props.bin')
|
||||
print(f"Writing {out_path}...")
|
||||
write_binary(out_path, ranges)
|
||||
|
||||
size = os.path.getsize(out_path)
|
||||
print(f"File size: {size} bytes ({size/1024:.1f} KB)")
|
||||
print("Done!")
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
+906
@@ -0,0 +1,906 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
x-icu-gen: Configurable ICU data generator with optimal compression.
|
||||
Generates binary tables + C++ implementation from UnicodeData.txt
|
||||
"""
|
||||
|
||||
import os
|
||||
import struct
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Tuple, Any, Optional, Set
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from collections import defaultdict
|
||||
|
||||
# ============================================================
|
||||
# CONFIGURATION (from py-gen.ipynb)
|
||||
# ============================================================
|
||||
|
||||
CONFIG = {
|
||||
'sources': {
|
||||
'unicode': 'https://www.unicode.org/Public/UCD/latest/ucd/UnicodeData.txt',
|
||||
},
|
||||
'unicode': {
|
||||
'useOldName': True,
|
||||
'getName': True,
|
||||
|
||||
'getDecomposition': False,
|
||||
'getDecompositionType': False,
|
||||
|
||||
'toLowercase': True,
|
||||
'toUppercase': True,
|
||||
'toTitlecase': False,
|
||||
|
||||
'isPunctuation': True,
|
||||
'isSymbol': False,
|
||||
'isCombining': False,
|
||||
|
||||
'isPrintable': False,
|
||||
'isSpace': True,
|
||||
'isWhitespace': True,
|
||||
'isLetter': True,
|
||||
'isUppercase': False,
|
||||
'isLowercase': False,
|
||||
'isTitlecase': False,
|
||||
'isDeprecated': False,
|
||||
|
||||
'isDecimal': True,
|
||||
'isDigit': False,
|
||||
'isNumberLike': False,
|
||||
|
||||
'getDecimal': True,
|
||||
'getDigit': False,
|
||||
'getNumberLike': False,
|
||||
}
|
||||
}
|
||||
|
||||
# Field definitions: maps config key -> (output_name, type, category)
|
||||
# category: 'bool', 'int', 'str', 'enum'
|
||||
FIELD_DEFS = {
|
||||
'getName': ('name', 'str', 'str'),
|
||||
'getDecomposition': ('decomposition', 'str', 'str'),
|
||||
'getDecompositionType': ('decomposition_type', 'str', 'str'),
|
||||
'isPunctuation': ('punctuation', 'bool', 'bool'),
|
||||
'isSymbol': ('symbol', 'bool', 'bool'),
|
||||
'isCombining': ('combining', 'bool', 'bool'),
|
||||
'isLetter': ('letter', 'bool', 'bool'),
|
||||
'isUppercase': ('uppercase', 'bool', 'bool'),
|
||||
'isLowercase': ('lowercase', 'bool', 'bool'),
|
||||
'isTitlecase': ('titlecase', 'bool', 'bool'),
|
||||
'toUppercase': ('uppercase_map', 'int', 'int'),
|
||||
'toLowercase': ('lowercase_map', 'int', 'int'),
|
||||
'toTitlecase': ('titlecase_map', 'int', 'int'),
|
||||
'isWhitespace': ('whitespace', 'bool', 'bool'),
|
||||
'isPrintable': ('printable', 'bool', 'bool'),
|
||||
'isSpace': ('space', 'bool', 'bool'),
|
||||
'isDecimal': ('decimal', 'bool', 'bool'),
|
||||
'isDigit': ('digit', 'bool', 'bool'),
|
||||
'isNumberLike': ('number_like', 'bool', 'bool'),
|
||||
'getDecimal': ('decimal_val', 'int', 'int'),
|
||||
'getDigit': ('digit_val', 'int', 'int'),
|
||||
'getNumberLike': ('number_like_val', 'int', 'int'),
|
||||
}
|
||||
|
||||
# ============================================================
|
||||
# DATA STRUCTURES
|
||||
# ============================================================
|
||||
|
||||
class CompStrategy(Enum):
|
||||
BITMAP = "bitmap" # 1 bit per codepoint
|
||||
DELTA_RANGES = "delta_ranges" # (start, end, value) runs
|
||||
RLE = "rle" # Run-length encoding
|
||||
SPARSE_MAP = "sparse_map" # Only store non-default values
|
||||
STRING_TABLE = "string_table" # Separate string pool
|
||||
|
||||
@dataclass
|
||||
class FieldConfig:
|
||||
name: str
|
||||
type: str # 'bool', 'int', 'str'
|
||||
enabled: bool = False
|
||||
strategy: CompStrategy = CompStrategy.DELTA_RANGES
|
||||
default_value: Any = None
|
||||
|
||||
@dataclass
|
||||
class CodePointData:
|
||||
cp: int
|
||||
name: str = ""
|
||||
category: str = ""
|
||||
decomposition: str = ""
|
||||
decimal_val: str = ""
|
||||
digit_val: str = ""
|
||||
numeric_val: str = ""
|
||||
bidi_mirrored: str = ""
|
||||
unicode_1_name: str = ""
|
||||
uppercase_map: str = ""
|
||||
lowercase_map: str = ""
|
||||
titlecase_map: str = ""
|
||||
|
||||
@dataclass
|
||||
class ProcessedRow:
|
||||
cp: int
|
||||
fields: Dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
# ============================================================
|
||||
# UNICODE PARSER
|
||||
# ============================================================
|
||||
|
||||
class UnicodeParser:
|
||||
def __init__(self, data_path: str):
|
||||
self.data_path = data_path
|
||||
self.keys = [
|
||||
"codepoint", "name", "category", "combining_class", "bidi_class",
|
||||
"decomposition", "decimal_val", "digit_val", "numeric_val",
|
||||
"bidi_mirrored", "unicode_1_name", "iso_comment",
|
||||
"uppercase_map", "lowercase_map", "titlecase_map"
|
||||
]
|
||||
|
||||
def parse(self) -> List[CodePointData]:
|
||||
results = []
|
||||
range_start = None
|
||||
|
||||
with open(self.data_path, 'r', encoding='utf-8') as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
parts = line.split(';')
|
||||
if len(parts) < 15:
|
||||
continue
|
||||
|
||||
cp = int(parts[0], 16)
|
||||
name = parts[1]
|
||||
cat = parts[2]
|
||||
decomp = parts[5]
|
||||
dec_val = parts[6]
|
||||
dig_val = parts[7]
|
||||
num_val = parts[8]
|
||||
bidi_mir = parts[9]
|
||||
old_name = parts[10]
|
||||
up_map = parts[12]
|
||||
low_map = parts[13]
|
||||
title_map = parts[14]
|
||||
|
||||
if name.endswith(', First>'):
|
||||
range_start = (cp, cat, name, decomp, dec_val, dig_val, num_val,
|
||||
bidi_mir, old_name, up_map, low_map, title_map)
|
||||
continue
|
||||
elif name.endswith(', Last>') and range_start:
|
||||
sc, scat, sname, sdecomp, sdec, sdig, snum, sbidi, sold, sup, slow, stitle = range_start
|
||||
base_name = sname.replace(', First>', '').replace('<', '')
|
||||
for c in range(sc, cp + 1):
|
||||
results.append(CodePointData(
|
||||
cp=c, name=f"<{base_name}>", category=scat,
|
||||
decomposition=sdecomp, decimal_val=sdec, digit_val=sdig,
|
||||
numeric_val=snum, bidi_mirrored=sbidi, unicode_1_name=sold,
|
||||
uppercase_map=sup, lowercase_map=slow, titlecase_map=stitle
|
||||
))
|
||||
range_start = None
|
||||
continue
|
||||
|
||||
results.append(CodePointData(
|
||||
cp=cp, name=name, category=cat, decomposition=decomp,
|
||||
decimal_val=dec_val, digit_val=dig_val, numeric_val=num_val,
|
||||
bidi_mirrored=bidi_mir, unicode_1_name=old_name,
|
||||
uppercase_map=up_map, lowercase_map=low_map, titlecase_map=title_map
|
||||
))
|
||||
return results
|
||||
|
||||
# ============================================================
|
||||
# FIELD PROCESSORS
|
||||
# ============================================================
|
||||
|
||||
def process_row(row: CodePointData, cfg: Dict, use_old_name: bool) -> ProcessedRow:
|
||||
"""Extract all configured fields from a parsed row."""
|
||||
out = ProcessedRow(cp=row.cp)
|
||||
cat = row.category
|
||||
char = chr(row.cp) if row.cp <= 0x10FFFF else ''
|
||||
|
||||
# Name handling
|
||||
name = row.name
|
||||
if use_old_name and name.startswith('<') and row.unicode_1_name:
|
||||
name = row.unicode_1_name
|
||||
|
||||
# Boolean properties from category
|
||||
out.fields['punctuation'] = cat.startswith('P')
|
||||
out.fields['symbol'] = cat.startswith('S')
|
||||
out.fields['combining'] = cat.startswith('M')
|
||||
out.fields['letter'] = cat.startswith('L')
|
||||
out.fields['uppercase'] = cat == 'Lu'
|
||||
out.fields['lowercase'] = cat == 'Ll'
|
||||
out.fields['titlecase'] = cat == 'Lt'
|
||||
out.fields['whitespace'] = cat.startswith('Z') or row.cp in (0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x1C, 0x1D, 0x1E, 0x1F, 0x85, 0xA0)
|
||||
out.fields['space'] = cat == 'Zs'
|
||||
out.fields['printable'] = not cat.startswith('C')
|
||||
|
||||
# Numeric properties
|
||||
is_decimal = bool(row.decimal_val)
|
||||
is_digit = is_decimal or bool(row.digit_val)
|
||||
is_numeric = is_digit or bool(row.numeric_val)
|
||||
out.fields['decimal'] = is_decimal
|
||||
out.fields['digit'] = is_digit
|
||||
out.fields['number_like'] = is_numeric
|
||||
out.fields['decimal_val'] = int(row.decimal_val) if is_decimal else 0
|
||||
out.fields['digit_val'] = int(row.digit_val) if is_digit else 0
|
||||
out.fields['number_like_val'] = row.numeric_val if is_numeric else ""
|
||||
|
||||
# String properties
|
||||
out.fields['name'] = name
|
||||
out.fields['decomposition'] = row.decomposition
|
||||
decomp_type = ""
|
||||
if row.decomposition and '<' in row.decomposition:
|
||||
decomp_type = row.decomposition[1:row.decomposition.find('>')]
|
||||
out.fields['decomposition_type'] = decomp_type
|
||||
|
||||
# Case mappings (relative offsets)
|
||||
for k, src in [('uppercase_map', row.uppercase_map),
|
||||
('lowercase_map', row.lowercase_map),
|
||||
('titlecase_map', row.titlecase_map)]:
|
||||
val = int(src, 16) if src else 0
|
||||
out.fields[k] = val - row.cp if val != 0 else 0
|
||||
|
||||
return out
|
||||
|
||||
# ============================================================
|
||||
# COMPRESSION ANALYSIS
|
||||
# ============================================================
|
||||
|
||||
def analyze_field(rows: List[ProcessedRow], field_name: str, field_type: str) -> CompStrategy:
|
||||
"""Determine optimal compression strategy for a field."""
|
||||
values = [r.fields.get(field_name, None) for r in rows]
|
||||
non_default = [v for v in values if v not in (False, 0, "", None)]
|
||||
unique_vals = set(v for v in values if v not in (False, 0, "", None))
|
||||
|
||||
total = len(rows)
|
||||
sparse_ratio = len(non_default) / total if total > 0 else 0
|
||||
|
||||
if field_type == 'str':
|
||||
return CompStrategy.STRING_TABLE
|
||||
|
||||
if field_type == 'bool':
|
||||
# If very sparse (< 1%), use sparse map
|
||||
if sparse_ratio < 0.01:
|
||||
return CompStrategy.SPARSE_MAP
|
||||
# If dense (> 50%), bitmap is good
|
||||
if sparse_ratio > 0.5:
|
||||
return CompStrategy.BITMAP
|
||||
# Otherwise delta ranges
|
||||
return CompStrategy.DELTA_RANGES
|
||||
|
||||
if field_type == 'int':
|
||||
# Check if values are mostly small offsets (case mappings)
|
||||
if field_name.endswith('_map'):
|
||||
# These are relative offsets, usually small or zero
|
||||
if sparse_ratio < 0.05:
|
||||
return CompStrategy.SPARSE_MAP
|
||||
return CompStrategy.DELTA_RANGES
|
||||
# For numeric values (decimal_val, etc.)
|
||||
if len(unique_vals) < 20:
|
||||
return CompStrategy.DELTA_RANGES
|
||||
return CompStrategy.SPARSE_MAP
|
||||
|
||||
return CompStrategy.DELTA_RANGES
|
||||
|
||||
# ============================================================
|
||||
# ENCODERS
|
||||
# ============================================================
|
||||
|
||||
def write_varint(buf: bytearray, val: int):
|
||||
while val >= 0x80:
|
||||
buf.append((val & 0x7F) | 0x80)
|
||||
val >>= 7
|
||||
buf.append(val)
|
||||
|
||||
def read_varint(data: bytes, offset: int) -> Tuple[int, int]:
|
||||
val = 0
|
||||
shift = 0
|
||||
while True:
|
||||
b = data[offset]
|
||||
offset += 1
|
||||
val |= (b & 0x7F) << shift
|
||||
if not (b & 0x80):
|
||||
break
|
||||
shift += 7
|
||||
return val, offset
|
||||
|
||||
def encode_bitmap(rows: List[ProcessedRow], field: str) -> bytes:
|
||||
"""Encode boolean field as bitmap."""
|
||||
max_cp = max(r.cp for r in rows)
|
||||
size = (max_cp + 8) // 8
|
||||
bitmap = bytearray(size)
|
||||
for r in rows:
|
||||
if r.fields.get(field, False):
|
||||
idx = r.cp >> 3
|
||||
bitmap[idx] |= 1 << (r.cp & 7)
|
||||
return bytes(bitmap)
|
||||
|
||||
def encode_delta_ranges(rows: List[ProcessedRow], field: str, field_type: str) -> bytes:
|
||||
"""Encode as (start, length, value) runs with varints."""
|
||||
buf = bytearray()
|
||||
current_start = None
|
||||
current_val = None
|
||||
|
||||
for r in rows:
|
||||
val = r.fields.get(field)
|
||||
if field_type == 'bool':
|
||||
val = bool(val)
|
||||
elif field_type == 'int':
|
||||
val = int(val) if val else 0
|
||||
else:
|
||||
val = val or ""
|
||||
|
||||
if current_start is None:
|
||||
current_start = r.cp
|
||||
current_val = val
|
||||
elif val != current_val:
|
||||
# End previous range
|
||||
length = r.cp - current_start
|
||||
write_varint(buf, current_start)
|
||||
write_varint(buf, length)
|
||||
if field_type == 'bool':
|
||||
buf.append(1 if current_val else 0)
|
||||
elif field_type == 'int':
|
||||
# Zigzag encode for negative values
|
||||
v = current_val
|
||||
write_varint(buf, (v << 1) ^ (v >> 31))
|
||||
current_start = r.cp
|
||||
current_val = val
|
||||
|
||||
# Last range
|
||||
if current_start is not None:
|
||||
length = 0x10FFFF - current_start + 1
|
||||
write_varint(buf, current_start)
|
||||
write_varint(buf, length)
|
||||
if field_type == 'bool':
|
||||
buf.append(1 if current_val else 0)
|
||||
elif field_type == 'int':
|
||||
v = current_val
|
||||
write_varint(buf, (v << 1) ^ (v >> 31))
|
||||
|
||||
return bytes(buf)
|
||||
|
||||
def encode_sparse_map(rows: List[ProcessedRow], field: str, field_type: str) -> bytes:
|
||||
"""Store only non-default values as (cp, value) pairs."""
|
||||
buf = bytearray()
|
||||
for r in rows:
|
||||
val = r.fields.get(field)
|
||||
if field_type == 'bool':
|
||||
val = bool(val)
|
||||
if not val:
|
||||
continue
|
||||
write_varint(buf, r.cp)
|
||||
buf.append(1)
|
||||
elif field_type == 'int':
|
||||
val = int(val) if val else 0
|
||||
if val == 0:
|
||||
continue
|
||||
write_varint(buf, r.cp)
|
||||
v = val
|
||||
write_varint(buf, (v << 1) ^ (v >> 31))
|
||||
else:
|
||||
val = val or ""
|
||||
if not val:
|
||||
continue
|
||||
write_varint(buf, r.cp)
|
||||
# String index will be handled separately
|
||||
write_varint(buf, 0) # placeholder
|
||||
return bytes(buf)
|
||||
|
||||
def encode_string_table(rows: List[ProcessedRow], field: str) -> Tuple[bytes, List[str]]:
|
||||
"""Build string table and return (indices, string_pool)."""
|
||||
strings = []
|
||||
str_to_idx = {}
|
||||
indices = []
|
||||
|
||||
for r in rows:
|
||||
val = r.fields.get(field, "") or ""
|
||||
if val not in str_to_idx:
|
||||
str_to_idx[val] = len(strings)
|
||||
strings.append(val)
|
||||
indices.append(str_to_idx[val])
|
||||
|
||||
# Encode indices as varints per codepoint
|
||||
buf = bytearray()
|
||||
for idx in indices:
|
||||
write_varint(buf, idx)
|
||||
|
||||
return bytes(buf), strings
|
||||
|
||||
# ============================================================
|
||||
# MAIN GENERATOR
|
||||
# ============================================================
|
||||
|
||||
class XicuGenerator:
|
||||
def __init__(self, config: Dict, data_dir: str = './data', out_dir: str = './out'):
|
||||
self.config = config
|
||||
self.data_dir = Path(data_dir)
|
||||
self.out_dir = Path(out_dir)
|
||||
self.out_dir.mkdir(parents=True, exist_ok=True)
|
||||
self.data_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
self.rows: List[ProcessedRow] = []
|
||||
self.field_configs: Dict[str, FieldConfig] = {}
|
||||
self.strategies: Dict[str, CompStrategy] = {}
|
||||
self.string_tables: Dict[str, List[str]] = {}
|
||||
self.encoded_data: Dict[str, bytes] = {}
|
||||
|
||||
def download_unicode_data(self):
|
||||
import urllib.request
|
||||
url = self.config['sources']['unicode']
|
||||
path = self.data_dir / 'UnicodeData.txt'
|
||||
if not path.exists():
|
||||
print(f"Downloading {url}...")
|
||||
urllib.request.urlretrieve(url, path)
|
||||
print("Done.")
|
||||
|
||||
def parse_and_process(self):
|
||||
parser = UnicodeParser(str(self.data_dir / 'UnicodeData.txt'))
|
||||
raw_rows = parser.parse()
|
||||
print(f"Parsed {len(raw_rows)} codepoints")
|
||||
|
||||
ucfg = self.config['unicode']
|
||||
use_old_name = ucfg.get('useOldName', False)
|
||||
|
||||
for row in raw_rows:
|
||||
self.rows.append(process_row(row, ucfg, use_old_name))
|
||||
|
||||
print(f"Processed {len(self.rows)} rows")
|
||||
|
||||
def analyze_and_configure(self):
|
||||
ucfg = self.config['unicode']
|
||||
|
||||
for cfg_key, (out_name, ftype, cat) in FIELD_DEFS.items():
|
||||
if not ucfg.get(cfg_key, False):
|
||||
continue
|
||||
|
||||
strategy = analyze_field(self.rows, out_name, ftype)
|
||||
default = False if ftype == 'bool' else (0 if ftype == 'int' else "")
|
||||
|
||||
self.field_configs[out_name] = FieldConfig(
|
||||
name=out_name, type=ftype, enabled=True,
|
||||
strategy=strategy, default_value=default
|
||||
)
|
||||
self.strategies[out_name] = strategy
|
||||
|
||||
print("Field strategies:")
|
||||
for name, fc in self.field_configs.items():
|
||||
print(f" {name} ({fc.type}): {fc.strategy.value}")
|
||||
|
||||
def encode_all(self):
|
||||
# First pass: encode non-string fields
|
||||
for name, fc in self.field_configs.items():
|
||||
if fc.type == 'str':
|
||||
continue
|
||||
|
||||
print(f"Encoding {name} with {fc.strategy.value}...")
|
||||
if fc.strategy == CompStrategy.BITMAP:
|
||||
self.encoded_data[name] = encode_bitmap(self.rows, name)
|
||||
elif fc.strategy == CompStrategy.DELTA_RANGES:
|
||||
self.encoded_data[name] = encode_delta_ranges(self.rows, name, fc.type)
|
||||
elif fc.strategy == CompStrategy.SPARSE_MAP:
|
||||
self.encoded_data[name] = encode_sparse_map(self.rows, name, fc.type)
|
||||
|
||||
# Second pass: string fields
|
||||
for name, fc in self.field_configs.items():
|
||||
if fc.type != 'str':
|
||||
continue
|
||||
print(f"Encoding string table for {name}...")
|
||||
indices, strings = encode_string_table(self.rows, name)
|
||||
self.encoded_data[name + '_indices'] = indices
|
||||
self.string_tables[name] = strings
|
||||
|
||||
def write_binary(self):
|
||||
"""Write main binary file with all non-string fields."""
|
||||
path = self.out_dir / 'xicu.bin'
|
||||
buf = bytearray()
|
||||
|
||||
# Header
|
||||
buf.extend(struct.pack('<I', 0x58494355)) # XICU magic
|
||||
buf.extend(struct.pack('<B', 2)) # version 2
|
||||
non_str_fields = [n for n, fc in self.field_configs.items() if fc.type != 'str']
|
||||
write_varint(buf, len(non_str_fields))
|
||||
|
||||
# Field descriptors + data inline
|
||||
for name, fc in self.field_configs.items():
|
||||
if fc.type == 'str':
|
||||
continue
|
||||
buf.extend(struct.pack('<B',
|
||||
{'bool': 1, 'int': 2}[fc.type]))
|
||||
buf.extend(struct.pack('<B',
|
||||
{'bitmap': 1, 'delta_ranges': 2, 'rle': 3, 'sparse_map': 4}[fc.strategy.value]))
|
||||
# Field name as null-terminated
|
||||
buf.extend(name.encode('ascii'))
|
||||
buf.append(0)
|
||||
# Data length + data
|
||||
data = self.encoded_data[name]
|
||||
write_varint(buf, len(data))
|
||||
buf.extend(data)
|
||||
|
||||
# String field descriptors
|
||||
str_fields = [n for n, fc in self.field_configs.items() if fc.type == 'str']
|
||||
write_varint(buf, len(str_fields))
|
||||
for name in str_fields:
|
||||
buf.extend(name.encode('ascii'))
|
||||
buf.append(0)
|
||||
|
||||
# String indices
|
||||
for name in str_fields:
|
||||
data = self.encoded_data[name + '_indices']
|
||||
write_varint(buf, len(data))
|
||||
buf.extend(data)
|
||||
|
||||
with open(path, 'wb') as f:
|
||||
f.write(buf)
|
||||
|
||||
print(f"Wrote {path} ({len(buf)} bytes)")
|
||||
|
||||
# Write string pools separately
|
||||
for name, strings in self.string_tables.items():
|
||||
pool_path = self.out_dir / f'xicu_{name}.str'
|
||||
buf = bytearray()
|
||||
write_varint(buf, len(strings))
|
||||
for s in strings:
|
||||
enc = s.encode('utf-8')
|
||||
write_varint(buf, len(enc))
|
||||
buf.extend(enc)
|
||||
with open(pool_path, 'wb') as f:
|
||||
f.write(buf)
|
||||
print(f"Wrote {pool_path} ({len(buf)} bytes)")
|
||||
|
||||
def generate_cpp(self):
|
||||
"""Generate C++ header and implementation."""
|
||||
hpp_path = Path('cpp') / 'xicu.hpp'
|
||||
cpp_path = Path('cpp') / 'xicu.cpp'
|
||||
Path('cpp').mkdir(exist_ok=True)
|
||||
|
||||
# Generate header
|
||||
hpp = self._gen_header()
|
||||
with open(hpp_path, 'w') as f:
|
||||
f.write(hpp)
|
||||
|
||||
# Generate implementation
|
||||
cpp = self._gen_impl()
|
||||
with open(cpp_path, 'w') as f:
|
||||
f.write(cpp)
|
||||
|
||||
print(f"Generated {hpp_path} and {cpp_path}")
|
||||
|
||||
def _gen_header(self) -> str:
|
||||
lines = [
|
||||
'#pragma once',
|
||||
'',
|
||||
'#include <cstdint>',
|
||||
'#include <string_view>',
|
||||
'',
|
||||
'namespace xicu {',
|
||||
'',
|
||||
'class PropTable {',
|
||||
'public:',
|
||||
' PropTable() = default;',
|
||||
' explicit PropTable(const char* path);',
|
||||
' ~PropTable();',
|
||||
'',
|
||||
' PropTable(const PropTable&) = delete;',
|
||||
' PropTable& operator=(const PropTable&) = delete;',
|
||||
' PropTable(PropTable&& other) noexcept;',
|
||||
' PropTable& operator=(PropTable&& other) noexcept;',
|
||||
'',
|
||||
' bool load(const char* path);',
|
||||
' bool isLoaded() const { return data_ != nullptr; }',
|
||||
]
|
||||
|
||||
# Generate getter declarations
|
||||
for name, fc in self.field_configs.items():
|
||||
if fc.type == 'bool':
|
||||
lines.append(f' bool get{name.capitalize()}(uint32_t cp) const;')
|
||||
elif fc.type == 'int':
|
||||
lines.append(f' int32_t get{name.capitalize()}(uint32_t cp) const;')
|
||||
elif fc.type == 'str':
|
||||
lines.append(f' std::string_view get{name.capitalize()}(uint32_t cp) const;')
|
||||
|
||||
lines.extend([
|
||||
'',
|
||||
'private:',
|
||||
' const uint8_t* data_ = nullptr;',
|
||||
' size_t data_size_ = 0;',
|
||||
' bool owns_data_ = false;',
|
||||
'',
|
||||
' static uint32_t readVarint(const uint8_t*& ptr);',
|
||||
' const uint8_t* findFieldData(const char* name, uint8_t& out_type, uint8_t& out_strat) const;',
|
||||
' const uint8_t* findStringField(const char* name) const;',
|
||||
'',
|
||||
' // Decoders',
|
||||
' bool decodeBitmap(const uint8_t* data, uint32_t cp) const;',
|
||||
' bool decodeDeltaRangesBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;',
|
||||
' int32_t decodeDeltaRangesInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;',
|
||||
' bool decodeSparseMapBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;',
|
||||
' int32_t decodeSparseMapInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;',
|
||||
'};',
|
||||
'',
|
||||
])
|
||||
|
||||
# Inline convenience functions
|
||||
for name, fc in self.field_configs.items():
|
||||
if fc.type == 'bool':
|
||||
lines.append(f'inline bool is{name.capitalize()}(uint32_t cp) {{')
|
||||
lines.append(f' static PropTable table("../out/xicu.bin");')
|
||||
lines.append(f' return table.get{name.capitalize()}(cp);')
|
||||
lines.append(f'}}')
|
||||
elif fc.type == 'int':
|
||||
lines.append(f'inline int32_t get{name.capitalize()}(uint32_t cp) {{')
|
||||
lines.append(f' static PropTable table("../out/xicu.bin");')
|
||||
lines.append(f' return table.get{name.capitalize()}(cp);')
|
||||
lines.append(f'}}')
|
||||
elif fc.type == 'str':
|
||||
lines.append(f'inline std::string_view get{name.capitalize()}(uint32_t cp) {{')
|
||||
lines.append(f' static PropTable table("../out/xicu.bin");')
|
||||
lines.append(f' return table.get{name.capitalize()}(cp);')
|
||||
lines.append(f'}}')
|
||||
|
||||
lines.extend([
|
||||
'',
|
||||
'} // namespace xicu',
|
||||
])
|
||||
return '\n'.join(lines)
|
||||
|
||||
def _gen_impl(self) -> str:
|
||||
"""Generate complete C++ implementation with decoders."""
|
||||
lines = [
|
||||
'#include "xicu.hpp"',
|
||||
'#include <cstdio>',
|
||||
'#include <cstdint>',
|
||||
'#include <cstring>',
|
||||
'#include <string_view>',
|
||||
'',
|
||||
'namespace xicu {',
|
||||
'',
|
||||
'static constexpr uint32_t MAGIC = 0x58494355;',
|
||||
'static constexpr uint8_t VERSION = 2;',
|
||||
'',
|
||||
'PropTable::PropTable(const char* path) { load(path); }',
|
||||
'',
|
||||
'PropTable::~PropTable() {',
|
||||
' if (owns_data_ && data_) { delete[] data_; data_ = nullptr; }',
|
||||
'}',
|
||||
'',
|
||||
'PropTable::PropTable(PropTable&& other) noexcept',
|
||||
' : data_(other.data_), data_size_(other.data_size_), owns_data_(other.owns_data_) {',
|
||||
' other.data_ = nullptr; other.owns_data_ = false;',
|
||||
'}',
|
||||
'',
|
||||
'PropTable& PropTable::operator=(PropTable&& other) noexcept {',
|
||||
' if (this != &other) {',
|
||||
' if (owns_data_ && data_) delete[] data_;',
|
||||
' data_ = other.data_; data_size_ = other.data_size_; owns_data_ = other.owns_data_;',
|
||||
' other.data_ = nullptr; other.owns_data_ = false;',
|
||||
' }',
|
||||
' return *this;',
|
||||
'}',
|
||||
'',
|
||||
'uint32_t PropTable::readVarint(const uint8_t*& ptr) {',
|
||||
' uint32_t val = 0; int shift = 0;',
|
||||
' while (true) {',
|
||||
' uint8_t b = *ptr++;',
|
||||
' val |= (b & 0x7F) << shift;',
|
||||
' if (!(b & 0x80)) break;',
|
||||
' shift += 7;',
|
||||
' }',
|
||||
' return val;',
|
||||
'}',
|
||||
'',
|
||||
'bool PropTable::load(const char* path) {',
|
||||
' if (owns_data_ && data_) { delete[] data_; data_ = nullptr; owns_data_ = false; }',
|
||||
' FILE* f = std::fopen(path, "rb");',
|
||||
' if (!f) return false;',
|
||||
' std::fseek(f, 0, SEEK_END);',
|
||||
' long fsize = std::ftell(f);',
|
||||
' std::fseek(f, 0, SEEK_SET);',
|
||||
' if (fsize < 6) { std::fclose(f); return false; }',
|
||||
' data_size_ = static_cast<size_t>(fsize);',
|
||||
' uint8_t* buf = new uint8_t[data_size_];',
|
||||
' size_t read = std::fread(buf, 1, data_size_, f);',
|
||||
' std::fclose(f);',
|
||||
' if (read != data_size_) { delete[] buf; return false; }',
|
||||
' data_ = buf; owns_data_ = true;',
|
||||
' ',
|
||||
' const uint8_t* ptr = data_;',
|
||||
' uint32_t magic = *reinterpret_cast<const uint32_t*>(ptr); ptr += 4;',
|
||||
' if (magic != MAGIC) return false;',
|
||||
' uint8_t version = *ptr++;',
|
||||
' if (version != VERSION) return false;',
|
||||
' return true;',
|
||||
'}',
|
||||
'',
|
||||
'// Find field data pointer (points to data length varint after name)',
|
||||
'const uint8_t* PropTable::findFieldData(const char* name, uint8_t& out_type, uint8_t& out_strat) const {',
|
||||
' if (!data_) return nullptr;',
|
||||
' const uint8_t* ptr = data_ + 5; // magic + version',
|
||||
' uint32_t num_fields = readVarint(ptr);',
|
||||
' for (uint32_t i = 0; i < num_fields; ++i) {',
|
||||
' uint8_t ftype = *ptr++;',
|
||||
' uint8_t fstrat = *ptr++;',
|
||||
' const char* fname = reinterpret_cast<const char*>(ptr);',
|
||||
' size_t name_len = std::strlen(fname);',
|
||||
' ptr += name_len + 1;',
|
||||
' if (std::strcmp(fname, name) == 0) {',
|
||||
' out_type = ftype;',
|
||||
' out_strat = fstrat;',
|
||||
' return ptr; // points to data length varint',
|
||||
' }',
|
||||
' // Skip data',
|
||||
' uint32_t data_len = readVarint(ptr);',
|
||||
' ptr += data_len;',
|
||||
' }',
|
||||
' return nullptr;',
|
||||
'}',
|
||||
'',
|
||||
'// Find string field (returns pointer to indices data length)',
|
||||
'const uint8_t* PropTable::findStringField(const char* name) const {',
|
||||
' if (!data_) return nullptr;',
|
||||
' const uint8_t* ptr = data_ + 5;',
|
||||
' uint32_t num_fields = readVarint(ptr);',
|
||||
' // Skip non-string fields (with inline data)',
|
||||
' for (uint32_t i = 0; i < num_fields; ++i) {',
|
||||
' ptr++; ptr++; // type, strat',
|
||||
' while (*ptr++) {} // skip name',
|
||||
' uint32_t data_len = readVarint(ptr);',
|
||||
' ptr += data_len; // skip data',
|
||||
' }',
|
||||
' uint32_t num_str_fields = readVarint(ptr);',
|
||||
' for (uint32_t i = 0; i < num_str_fields; ++i) {',
|
||||
' const char* fname = reinterpret_cast<const char*>(ptr);',
|
||||
' size_t name_len = std::strlen(fname);',
|
||||
' ptr += name_len + 1;',
|
||||
' if (std::strcmp(fname, name) == 0) {',
|
||||
' return ptr; // points to indices data length',
|
||||
' }',
|
||||
' uint32_t data_len = readVarint(ptr);',
|
||||
' ptr += data_len; // skip indices data',
|
||||
' }',
|
||||
' return nullptr;',
|
||||
'}',
|
||||
'',
|
||||
'// Decode bitmap: 1 bit per codepoint',
|
||||
'bool PropTable::decodeBitmap(const uint8_t* data, uint32_t cp) const {',
|
||||
' size_t byte_idx = cp >> 3;',
|
||||
' uint8_t bit = cp & 7;',
|
||||
' return (data[byte_idx] & (1 << bit)) != 0;',
|
||||
'}',
|
||||
'',
|
||||
'// Decode delta ranges: (start, length, value) varint-encoded',
|
||||
'bool PropTable::decodeDeltaRangesBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {',
|
||||
' const uint8_t* ptr = data;',
|
||||
' while (ptr < data_end) {',
|
||||
' uint32_t start = readVarint(ptr);',
|
||||
' uint32_t length = readVarint(ptr);',
|
||||
' uint8_t val = *ptr++;',
|
||||
' uint32_t end = start + length - 1;',
|
||||
' if (cp >= start && cp <= end) return val != 0;',
|
||||
' if (cp < start) return false;',
|
||||
' }',
|
||||
' return false;',
|
||||
'}',
|
||||
'',
|
||||
'int32_t PropTable::decodeDeltaRangesInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {',
|
||||
' const uint8_t* ptr = data;',
|
||||
' while (ptr < data_end) {',
|
||||
' uint32_t start = readVarint(ptr);',
|
||||
' uint32_t length = readVarint(ptr);',
|
||||
' uint32_t zigzag = readVarint(ptr);',
|
||||
' int32_t val = static_cast<int32_t>((zigzag >> 1) ^ -(zigzag & 1));',
|
||||
' uint32_t end = start + length - 1;',
|
||||
' if (cp >= start && cp <= end) return val;',
|
||||
' if (cp < start) return 0;',
|
||||
' }',
|
||||
' return 0;',
|
||||
'}',
|
||||
'',
|
||||
'// Decode sparse map: (cp, value) pairs',
|
||||
'bool PropTable::decodeSparseMapBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {',
|
||||
' const uint8_t* ptr = data;',
|
||||
' while (ptr < data_end) {',
|
||||
' uint32_t c = readVarint(ptr);',
|
||||
' if (ptr >= data_end) break;',
|
||||
' uint8_t val = *ptr++;',
|
||||
' if (c == cp) return val != 0;',
|
||||
' if (c > cp) return false;',
|
||||
' }',
|
||||
' return false;',
|
||||
'}',
|
||||
'',
|
||||
'int32_t PropTable::decodeSparseMapInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {',
|
||||
' const uint8_t* ptr = data;',
|
||||
' while (ptr < data_end) {',
|
||||
' uint32_t c = readVarint(ptr);',
|
||||
' if (ptr >= data_end) break;',
|
||||
' uint32_t zigzag = readVarint(ptr);',
|
||||
' int32_t val = static_cast<int32_t>((zigzag >> 1) ^ -(zigzag & 1));',
|
||||
' if (c == cp) return val;',
|
||||
' if (c > cp) return 0;',
|
||||
' }',
|
||||
' return 0;',
|
||||
'}',
|
||||
'',
|
||||
]
|
||||
|
||||
# Generate getters for each field
|
||||
for name, fc in self.field_configs.items():
|
||||
if fc.type == 'bool':
|
||||
strat = fc.strategy.value
|
||||
if strat == 'bitmap':
|
||||
decoder = 'decodeBitmap'
|
||||
call = 'decodeBitmap(data, cp)'
|
||||
elif strat == 'delta_ranges':
|
||||
decoder = 'decodeDeltaRangesBool'
|
||||
call = 'decodeDeltaRangesBool(data, data_end, cp)'
|
||||
elif strat == 'sparse_map':
|
||||
decoder = 'decodeSparseMapBool'
|
||||
call = 'decodeSparseMapBool(data, data_end, cp)'
|
||||
else:
|
||||
decoder = 'decodeDeltaRangesBool'
|
||||
call = 'decodeDeltaRangesBool(data, data_end, cp)'
|
||||
|
||||
lines.extend([
|
||||
f'bool PropTable::get{name.capitalize()}(uint32_t cp) const {{',
|
||||
f' uint8_t ftype, fstrat;',
|
||||
f' const uint8_t* data = findFieldData("{name}", ftype, fstrat);',
|
||||
f' if (!data) return false;',
|
||||
f' uint32_t data_len = readVarint(data);',
|
||||
f' const uint8_t* data_end = data + data_len;',
|
||||
f' return {call};',
|
||||
f'}}',
|
||||
'',
|
||||
])
|
||||
elif fc.type == 'int':
|
||||
strat = fc.strategy.value
|
||||
if strat == 'delta_ranges':
|
||||
decoder = 'decodeDeltaRangesInt'
|
||||
elif strat == 'sparse_map':
|
||||
decoder = 'decodeSparseMapInt'
|
||||
else:
|
||||
decoder = 'decodeDeltaRangesInt'
|
||||
|
||||
lines.extend([
|
||||
f'int32_t PropTable::get{name.capitalize()}(uint32_t cp) const {{',
|
||||
f' uint8_t ftype, fstrat;',
|
||||
f' const uint8_t* data = findFieldData("{name}", ftype, fstrat);',
|
||||
f' if (!data) return 0;',
|
||||
f' uint32_t data_len = readVarint(data);',
|
||||
f' const uint8_t* data_end = data + data_len;',
|
||||
f' return {decoder}(data, data_end, cp);',
|
||||
f'}}',
|
||||
'',
|
||||
])
|
||||
elif fc.type == 'str':
|
||||
lines.extend([
|
||||
f'std::string_view PropTable::get{name.capitalize()}(uint32_t cp) const {{',
|
||||
f' // String fields use separate .str file',
|
||||
f' const uint8_t* indices_ptr = findStringField("{name}");',
|
||||
f' if (!indices_ptr) return "";',
|
||||
f' uint32_t indices_len = readVarint(indices_ptr);',
|
||||
f' // Find index for this codepoint (linear scan for now)',
|
||||
f' // TODO: optimize with binary search if needed',
|
||||
f' // For now, we need the string pool loaded separately',
|
||||
f' return ""; // Requires string pool file',
|
||||
f'}}',
|
||||
'',
|
||||
])
|
||||
|
||||
lines.append('} // namespace xicu')
|
||||
return '\n'.join(lines)
|
||||
|
||||
def run(self):
|
||||
self.download_unicode_data()
|
||||
self.parse_and_process()
|
||||
self.analyze_and_configure()
|
||||
self.encode_all()
|
||||
self.write_binary()
|
||||
self.generate_cpp()
|
||||
print("\nGeneration complete!")
|
||||
|
||||
def main():
|
||||
gen = XicuGenerator(CONFIG)
|
||||
gen.run()
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in New Issue
Block a user