lots of improvements standby for the results

This commit is contained in:
2026-09-27 01:09:57 -06:00
parent 812ac62393
commit 150e5e8636
16 changed files with 2726 additions and 9 deletions
+96
View File
@@ -0,0 +1,96 @@
#include "xicu.hpp"
#include <cstdio>
#include <cstdint>
int main() {
printf("Testing x-icu binary tables...\n\n");
// Test alphanumeric
struct AlnumTest { uint32_t cp; bool expected; const char* desc; };
AlnumTest alnum_tests[] = {
{ 0x0041, true, "Latin capital A" },
{ 0x007A, true, "Latin small z" },
{ 0x0030, true, "Digit 0" },
{ 0x0039, true, "Digit 9" },
{ 0x00E9, true, "Latin small e with acute" },
{ 0x4E2D, true, "CJK ideograph" },
{ 0x03B1, true, "Greek alpha" },
{ 0x0040, false, "At sign" },
{ 0x0023, false, "Hash" },
{ 0x0020, false, "Space" },
{ 0x0009, false, "Tab" },
{ 0x1F600, false, "Emoji (grinning face)" },
{ 0x0627, true, "Arabic letter alef" },
{ 0x0915, true, "Devanagari letter ka" },
{ 0x3042, true, "Hiragana letter a" },
};
printf("=== Alphanumeric Tests ===\n");
int passed = 0, failed = 0;
for (const auto& t : alnum_tests) {
bool result = xicu::isAlphanumeric(t.cp);
if (result == t.expected) {
printf("PASS: U+%04X (%s) -> %s\n", t.cp, t.desc, result ? "true" : "false");
passed++;
} else {
printf("FAIL: U+%04X (%s) -> got %s, expected %s\n", t.cp, t.desc,
result ? "true" : "false", t.expected ? "true" : "false");
failed++;
}
}
// Test whitespace
struct WsTest { uint32_t cp; bool expected; const char* desc; };
WsTest ws_tests[] = {
{ ' ', true, "Space" },
{ '\t', true, "Tab" },
{ '\n', true, "Line feed" },
{ '\r', true, "Carriage return" },
{ '\f', true, "Form feed" },
{ '\v', true, "Vertical tab" },
{ 0x1C, true, "File separator" },
{ 0x1D, true, "Group separator" },
{ 0x1E, true, "Record separator" },
{ 0x1F, true, "Unit separator" },
{ 0x85, true, "Next line (NEL)" },
{ 0xA0, true, "No-break space" },
{ 'A', false, "Letter A" },
{ '0', false, "Digit 0" },
{ '@', false, "At sign" },
{ 0x2000, true, "En quad" },
{ 0x2001, true, "Em quad" },
{ 0x2002, true, "En space" },
{ 0x2003, true, "Em space" },
{ 0x2004, true, "Three-per-em space" },
{ 0x2005, true, "Four-per-em space" },
{ 0x2006, true, "Six-per-em space" },
{ 0x2007, true, "Figure space" },
{ 0x2008, true, "Punctuation space" },
{ 0x2009, true, "Thin space" },
{ 0x200A, true, "Hair space" },
{ 0x2028, true, "Line separator" },
{ 0x2029, true, "Paragraph separator" },
{ 0x202F, true, "Narrow no-break space" },
{ 0x205F, true, "Medium mathematical space" },
{ 0x3000, true, "Ideographic space" },
};
printf("\n=== Whitespace Tests ===\n");
for (const auto& t : ws_tests) {
bool result = xicu::isWhitespace(t.cp);
if (result == t.expected) {
printf("PASS: U+%04X (%s) -> %s\n", t.cp, t.desc, result ? "true" : "false");
passed++;
} else {
printf("FAIL: U+%04X (%s) -> got %s, expected %s\n", t.cp, t.desc,
result ? "true" : "false", t.expected ? "true" : "false");
failed++;
}
}
printf("\n=== Summary ===\n");
printf("Passed: %d\n", passed);
printf("Failed: %d\n", failed);
return failed == 0 ? 0 : 1;
}
BIN
View File
Binary file not shown.
+162
View File
@@ -0,0 +1,162 @@
#include "xicu.hpp"
#include <cstdio>
#include <cstdint>
int main() {
printf("Testing x-icu configurable generator...\n\n");
// Test isLetter (bitmap)
struct { uint32_t cp; bool expected; const char* desc; } letter_tests[] = {
{ 'A', true, "Latin capital A" },
{ 'z', true, "Latin small z" },
{ '0', false, "Digit 0" },
{ '@', false, "At sign" },
{ ' ', false, "Space" },
{ 0x00E9, true, "e acute" },
{ 0x4E2D, true, "CJK ideograph" },
{ 0x03B1, true, "Greek alpha" },
{ 0x0627, true, "Arabic alef" },
{ 0x1F600, false, "Emoji" },
};
printf("=== isLetter (bitmap) ===\n");
int passed = 0, failed = 0;
for (auto& t : letter_tests) {
bool r = xicu::isLetter(t.cp);
if (r == t.expected) { printf("PASS: U+%04X %s\n", t.cp, t.desc); passed++; }
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
}
// Test isPunctuation (sparse_map)
struct { uint32_t cp; bool expected; const char* desc; } punct_tests[] = {
{ '.', true, "Period" },
{ ',', true, "Comma" },
{ '!', true, "Exclamation" },
{ '?', true, "Question" },
{ 'A', false, "Letter A" },
{ '0', false, "Digit 0" },
{ ' ', false, "Space" },
{ 0x2026, true, "Ellipsis" },
{ 0x2018, true, "Left single quote" },
{ 0x201C, true, "Left double quote" },
};
printf("\n=== isPunctuation (sparse_map) ===\n");
for (auto& t : punct_tests) {
bool r = xicu::isPunctuation(t.cp);
if (r == t.expected) { printf("PASS: U+%04X %s\n", t.cp, t.desc); passed++; }
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
}
// Test isWhitespace (sparse_map)
struct { uint32_t cp; bool expected; const char* desc; } ws_tests[] = {
{ ' ', true, "Space" },
{ '\t', true, "Tab" },
{ '\n', true, "LF" },
{ '\r', true, "CR" },
{ 0x00A0, true, "NBSP" },
{ 0x2000, true, "En quad" },
{ 0x2003, true, "Em space" },
{ 0x2028, true, "Line separator" },
{ 0x2029, true, "Paragraph separator" },
{ 0x3000, true, "Ideographic space" },
{ 'A', false, "Letter A" },
{ '0', false, "Digit 0" },
};
printf("\n=== isWhitespace (sparse_map) ===\n");
for (auto& t : ws_tests) {
bool r = xicu::isWhitespace(t.cp);
if (r == t.expected) { printf("PASS: U+%04X %s\n", t.cp, t.desc); passed++; }
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
}
// Test isSpace (sparse_map) - only Zs
struct { uint32_t cp; bool expected; const char* desc; } space_tests[] = {
{ ' ', true, "Space" },
{ 0x00A0, true, "NBSP" },
{ 0x2000, true, "En quad" },
{ 0x2003, true, "Em space" },
{ 0x3000, true, "Ideographic space" },
{ '\t', false, "Tab (not Zs)" },
{ '\n', false, "LF (not Zs)" },
{ 'A', false, "Letter A" },
};
printf("\n=== isSpace (sparse_map) ===\n");
for (auto& t : space_tests) {
bool r = xicu::isSpace(t.cp);
if (r == t.expected) { printf("PASS: U+%04X %s\n", t.cp, t.desc); passed++; }
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
}
// Test isDecimal (sparse_map)
struct { uint32_t cp; bool expected; const char* desc; } dec_tests[] = {
{ '0', true, "Digit 0" },
{ '9', true, "Digit 9" },
{ 'A', false, "Letter A" },
{ 0x0660, true, "Arabic-Indic 0" },
{ 0x0966, true, "Devanagari 0" },
{ 0xFF10, true, "Fullwidth 0" },
};
printf("\n=== isDecimal (sparse_map) ===\n");
for (auto& t : dec_tests) {
bool r = xicu::isDecimal(t.cp);
if (r == t.expected) { printf("PASS: U+%04X %s\n", t.cp, t.desc); passed++; }
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
}
// Test getDecimal_val (delta_ranges)
struct { uint32_t cp; int expected; const char* desc; } decval_tests[] = {
{ '0', 0, "Digit 0" },
{ '5', 5, "Digit 5" },
{ '9', 9, "Digit 9" },
{ 0x0665, 5, "Arabic-Indic 5" },
{ 'A', 0, "Letter A" },
};
printf("\n=== getDecimal_val (delta_ranges) ===\n");
for (auto& t : decval_tests) {
int r = xicu::getDecimal_val(t.cp);
if (r == t.expected) { printf("PASS: U+%04X %s = %d\n", t.cp, t.desc, r); passed++; }
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
}
// Test getUppercase_map (sparse_map)
struct { uint32_t cp; int expected; const char* desc; } up_tests[] = {
{ 'a', 'A' - 'a', "a -> A" },
{ 'z', 'Z' - 'z', "z -> Z" },
{ 0x00E9, 0x00C9 - 0x00E9, "e acute -> E acute" },
{ 'A', 0, "A (already upper)" },
{ '0', 0, "Digit 0" },
};
printf("\n=== getUppercase_map (sparse_map) ===\n");
for (auto& t : up_tests) {
int r = xicu::getUppercase_map(t.cp);
if (r == t.expected) { printf("PASS: U+%04X %s = %d\n", t.cp, t.desc, r); passed++; }
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
}
// Test getLowercase_map (sparse_map)
struct { uint32_t cp; int expected; const char* desc; } low_tests[] = {
{ 'A', 'a' - 'A', "A -> a" },
{ 'Z', 'z' - 'Z', "Z -> z" },
{ 0x00C9, 0x00E9 - 0x00C9, "E acute -> e acute" },
{ 'a', 0, "a (already lower)" },
{ '0', 0, "Digit 0" },
};
printf("\n=== getLowercase_map (sparse_map) ===\n");
for (auto& t : low_tests) {
int r = xicu::getLowercase_map(t.cp);
if (r == t.expected) { printf("PASS: U+%04X %s = %d\n", t.cp, t.desc, r); passed++; }
else { printf("FAIL: U+%04X %s got %d exp %d\n", t.cp, t.desc, r, t.expected); failed++; }
}
printf("\n=== Summary ===\n");
printf("Passed: %d\n", passed);
printf("Failed: %d\n", failed);
return failed == 0 ? 0 : 1;
}
Binary file not shown.
+259
View File
@@ -0,0 +1,259 @@
#include "xicu.hpp"
#include <cstdio>
#include <cstdint>
#include <cstring>
#include <string_view>
namespace xicu {
static constexpr uint32_t MAGIC = 0x58494355;
static constexpr uint8_t VERSION = 2;
PropTable::PropTable(const char* path) { load(path); }
PropTable::~PropTable() {
if (owns_data_ && data_) { delete[] data_; data_ = nullptr; }
}
PropTable::PropTable(PropTable&& other) noexcept
: data_(other.data_), data_size_(other.data_size_), owns_data_(other.owns_data_) {
other.data_ = nullptr; other.owns_data_ = false;
}
PropTable& PropTable::operator=(PropTable&& other) noexcept {
if (this != &other) {
if (owns_data_ && data_) delete[] data_;
data_ = other.data_; data_size_ = other.data_size_; owns_data_ = other.owns_data_;
other.data_ = nullptr; other.owns_data_ = false;
}
return *this;
}
uint32_t PropTable::readVarint(const uint8_t*& ptr) {
uint32_t val = 0; int shift = 0;
while (true) {
uint8_t b = *ptr++;
val |= (b & 0x7F) << shift;
if (!(b & 0x80)) break;
shift += 7;
}
return val;
}
bool PropTable::load(const char* path) {
if (owns_data_ && data_) { delete[] data_; data_ = nullptr; owns_data_ = false; }
FILE* f = std::fopen(path, "rb");
if (!f) return false;
std::fseek(f, 0, SEEK_END);
long fsize = std::ftell(f);
std::fseek(f, 0, SEEK_SET);
if (fsize < 6) { std::fclose(f); return false; }
data_size_ = static_cast<size_t>(fsize);
uint8_t* buf = new uint8_t[data_size_];
size_t read = std::fread(buf, 1, data_size_, f);
std::fclose(f);
if (read != data_size_) { delete[] buf; return false; }
data_ = buf; owns_data_ = true;
const uint8_t* ptr = data_;
uint32_t magic = *reinterpret_cast<const uint32_t*>(ptr); ptr += 4;
if (magic != MAGIC) return false;
uint8_t version = *ptr++;
if (version != VERSION) return false;
return true;
}
// Find field data pointer (points to data length varint after name)
const uint8_t* PropTable::findFieldData(const char* name, uint8_t& out_type, uint8_t& out_strat) const {
if (!data_) return nullptr;
const uint8_t* ptr = data_ + 5; // magic + version
uint32_t num_fields = readVarint(ptr);
for (uint32_t i = 0; i < num_fields; ++i) {
uint8_t ftype = *ptr++;
uint8_t fstrat = *ptr++;
const char* fname = reinterpret_cast<const char*>(ptr);
size_t name_len = std::strlen(fname);
ptr += name_len + 1;
if (std::strcmp(fname, name) == 0) {
out_type = ftype;
out_strat = fstrat;
return ptr; // points to data length varint
}
// Skip data
uint32_t data_len = readVarint(ptr);
ptr += data_len;
}
return nullptr;
}
// Find string field (returns pointer to indices data length)
const uint8_t* PropTable::findStringField(const char* name) const {
if (!data_) return nullptr;
const uint8_t* ptr = data_ + 5;
uint32_t num_fields = readVarint(ptr);
// Skip non-string fields (with inline data)
for (uint32_t i = 0; i < num_fields; ++i) {
ptr++; ptr++; // type, strat
while (*ptr++) {} // skip name
uint32_t data_len = readVarint(ptr);
ptr += data_len; // skip data
}
uint32_t num_str_fields = readVarint(ptr);
for (uint32_t i = 0; i < num_str_fields; ++i) {
const char* fname = reinterpret_cast<const char*>(ptr);
size_t name_len = std::strlen(fname);
ptr += name_len + 1;
if (std::strcmp(fname, name) == 0) {
return ptr; // points to indices data length
}
uint32_t data_len = readVarint(ptr);
ptr += data_len; // skip indices data
}
return nullptr;
}
// Decode bitmap: 1 bit per codepoint
bool PropTable::decodeBitmap(const uint8_t* data, uint32_t cp) const {
size_t byte_idx = cp >> 3;
uint8_t bit = cp & 7;
return (data[byte_idx] & (1 << bit)) != 0;
}
// Decode delta ranges: (start, length, value) varint-encoded
bool PropTable::decodeDeltaRangesBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {
const uint8_t* ptr = data;
while (ptr < data_end) {
uint32_t start = readVarint(ptr);
uint32_t length = readVarint(ptr);
uint8_t val = *ptr++;
uint32_t end = start + length - 1;
if (cp >= start && cp <= end) return val != 0;
if (cp < start) return false;
}
return false;
}
int32_t PropTable::decodeDeltaRangesInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {
const uint8_t* ptr = data;
while (ptr < data_end) {
uint32_t start = readVarint(ptr);
uint32_t length = readVarint(ptr);
uint32_t zigzag = readVarint(ptr);
int32_t val = static_cast<int32_t>((zigzag >> 1) ^ -(zigzag & 1));
uint32_t end = start + length - 1;
if (cp >= start && cp <= end) return val;
if (cp < start) return 0;
}
return 0;
}
// Decode sparse map: (cp, value) pairs
bool PropTable::decodeSparseMapBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {
const uint8_t* ptr = data;
while (ptr < data_end) {
uint32_t c = readVarint(ptr);
if (ptr >= data_end) break;
uint8_t val = *ptr++;
if (c == cp) return val != 0;
if (c > cp) return false;
}
return false;
}
int32_t PropTable::decodeSparseMapInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {
const uint8_t* ptr = data;
while (ptr < data_end) {
uint32_t c = readVarint(ptr);
if (ptr >= data_end) break;
uint32_t zigzag = readVarint(ptr);
int32_t val = static_cast<int32_t>((zigzag >> 1) ^ -(zigzag & 1));
if (c == cp) return val;
if (c > cp) return 0;
}
return 0;
}
std::string_view PropTable::getName(uint32_t cp) const {
// String fields use separate .str file
const uint8_t* indices_ptr = findStringField("name");
if (!indices_ptr) return "";
uint32_t indices_len = readVarint(indices_ptr);
// Find index for this codepoint (linear scan for now)
// TODO: optimize with binary search if needed
// For now, we need the string pool loaded separately
return ""; // Requires string pool file
}
bool PropTable::getPunctuation(uint32_t cp) const {
uint8_t ftype, fstrat;
const uint8_t* data = findFieldData("punctuation", ftype, fstrat);
if (!data) return false;
uint32_t data_len = readVarint(data);
const uint8_t* data_end = data + data_len;
return decodeSparseMapBool(data, data_end, cp);
}
bool PropTable::getLetter(uint32_t cp) const {
uint8_t ftype, fstrat;
const uint8_t* data = findFieldData("letter", ftype, fstrat);
if (!data) return false;
uint32_t data_len = readVarint(data);
const uint8_t* data_end = data + data_len;
return decodeBitmap(data, cp);
}
int32_t PropTable::getUppercase_map(uint32_t cp) const {
uint8_t ftype, fstrat;
const uint8_t* data = findFieldData("uppercase_map", ftype, fstrat);
if (!data) return 0;
uint32_t data_len = readVarint(data);
const uint8_t* data_end = data + data_len;
return decodeSparseMapInt(data, data_end, cp);
}
int32_t PropTable::getLowercase_map(uint32_t cp) const {
uint8_t ftype, fstrat;
const uint8_t* data = findFieldData("lowercase_map", ftype, fstrat);
if (!data) return 0;
uint32_t data_len = readVarint(data);
const uint8_t* data_end = data + data_len;
return decodeSparseMapInt(data, data_end, cp);
}
bool PropTable::getWhitespace(uint32_t cp) const {
uint8_t ftype, fstrat;
const uint8_t* data = findFieldData("whitespace", ftype, fstrat);
if (!data) return false;
uint32_t data_len = readVarint(data);
const uint8_t* data_end = data + data_len;
return decodeSparseMapBool(data, data_end, cp);
}
bool PropTable::getSpace(uint32_t cp) const {
uint8_t ftype, fstrat;
const uint8_t* data = findFieldData("space", ftype, fstrat);
if (!data) return false;
uint32_t data_len = readVarint(data);
const uint8_t* data_end = data + data_len;
return decodeSparseMapBool(data, data_end, cp);
}
bool PropTable::getDecimal(uint32_t cp) const {
uint8_t ftype, fstrat;
const uint8_t* data = findFieldData("decimal", ftype, fstrat);
if (!data) return false;
uint32_t data_len = readVarint(data);
const uint8_t* data_end = data + data_len;
return decodeSparseMapBool(data, data_end, cp);
}
int32_t PropTable::getDecimal_val(uint32_t cp) const {
uint8_t ftype, fstrat;
const uint8_t* data = findFieldData("decimal_val", ftype, fstrat);
if (!data) return 0;
uint32_t data_len = readVarint(data);
const uint8_t* data_end = data + data_len;
return decodeDeltaRangesInt(data, data_end, cp);
}
} // namespace xicu
+85
View File
@@ -0,0 +1,85 @@
#pragma once
#include <cstdint>
#include <string_view>
namespace xicu {
class PropTable {
public:
PropTable() = default;
explicit PropTable(const char* path);
~PropTable();
PropTable(const PropTable&) = delete;
PropTable& operator=(const PropTable&) = delete;
PropTable(PropTable&& other) noexcept;
PropTable& operator=(PropTable&& other) noexcept;
bool load(const char* path);
bool isLoaded() const { return data_ != nullptr; }
std::string_view getName(uint32_t cp) const;
bool getPunctuation(uint32_t cp) const;
bool getLetter(uint32_t cp) const;
int32_t getUppercase_map(uint32_t cp) const;
int32_t getLowercase_map(uint32_t cp) const;
bool getWhitespace(uint32_t cp) const;
bool getSpace(uint32_t cp) const;
bool getDecimal(uint32_t cp) const;
int32_t getDecimal_val(uint32_t cp) const;
private:
const uint8_t* data_ = nullptr;
size_t data_size_ = 0;
bool owns_data_ = false;
static uint32_t readVarint(const uint8_t*& ptr);
const uint8_t* findFieldData(const char* name, uint8_t& out_type, uint8_t& out_strat) const;
const uint8_t* findStringField(const char* name) const;
// Decoders
bool decodeBitmap(const uint8_t* data, uint32_t cp) const;
bool decodeDeltaRangesBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;
int32_t decodeDeltaRangesInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;
bool decodeSparseMapBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;
int32_t decodeSparseMapInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;
};
inline std::string_view getName(uint32_t cp) {
static PropTable table("../out/xicu.bin");
return table.getName(cp);
}
inline bool isPunctuation(uint32_t cp) {
static PropTable table("../out/xicu.bin");
return table.getPunctuation(cp);
}
inline bool isLetter(uint32_t cp) {
static PropTable table("../out/xicu.bin");
return table.getLetter(cp);
}
inline int32_t getUppercase_map(uint32_t cp) {
static PropTable table("../out/xicu.bin");
return table.getUppercase_map(cp);
}
inline int32_t getLowercase_map(uint32_t cp) {
static PropTable table("../out/xicu.bin");
return table.getLowercase_map(cp);
}
inline bool isWhitespace(uint32_t cp) {
static PropTable table("../out/xicu.bin");
return table.getWhitespace(cp);
}
inline bool isSpace(uint32_t cp) {
static PropTable table("../out/xicu.bin");
return table.getSpace(cp);
}
inline bool isDecimal(uint32_t cp) {
static PropTable table("../out/xicu.bin");
return table.getDecimal(cp);
}
inline int32_t getDecimal_val(uint32_t cp) {
static PropTable table("../out/xicu.bin");
return table.getDecimal_val(cp);
}
} // namespace xicu
BIN
View File
Binary file not shown.
File diff suppressed because it is too large Load Diff
Binary file not shown.
BIN
View File
Binary file not shown.
Binary file not shown.
BIN
View File
Binary file not shown.
+183
View File
File diff suppressed because one or more lines are too long
+117
View File
@@ -0,0 +1,117 @@
#!/usr/bin/env python3
"""
Generate binary tables for:
a) is char X alphanumeric? (letter or decimal digit)
b) is char whitespace?
"""
import os
import struct
import urllib.request
DATA_DIR = './data'
OUT_DIR = './out'
os.makedirs(DATA_DIR, exist_ok=True)
os.makedirs(OUT_DIR, exist_ok=True)
UNICODE_DATA_URL = 'https://www.unicode.org/Public/UCD/latest/ucd/UnicodeData.txt'
UNICODE_DATA_PATH = os.path.join(DATA_DIR, 'UnicodeData.txt')
if not os.path.exists(UNICODE_DATA_PATH):
print(f"Downloading {UNICODE_DATA_URL}...")
urllib.request.urlretrieve(UNICODE_DATA_URL, UNICODE_DATA_PATH)
print("Downloaded")
def parse_unicode_data():
"""Parse UnicodeData.txt and yield (codepoint, category, name)"""
with open(UNICODE_DATA_PATH, 'r', encoding='utf-8') as f:
for line in f:
line = line.strip()
if not line:
continue
parts = line.split(';')
if len(parts) < 3:
continue
codepoint = int(parts[0], 16)
name = parts[1]
category = parts[2]
# Handle ranges (First/Last)
if name.endswith(', First>'):
range_start = (codepoint, category, name)
continue
elif name.endswith(', Last>') and 'range_start' in locals():
start_cp, start_cat, start_name = range_start
base_name = start_name.replace(', First>', '').replace('<', '')
for cp in range(start_cp, codepoint + 1):
yield (cp, start_cat, f"<{base_name}>")
del range_start
continue
yield (codepoint, category, name)
def is_alphanumeric(category: str, codepoint: int) -> bool:
"""Check if category indicates alphanumeric (letter or decimal digit)"""
return category.startswith('L') or category == 'Nd'
def is_whitespace(category: str, codepoint: int) -> bool:
"""Check if character is whitespace"""
if category in ('Zs', 'Zl', 'Zp'):
return True
# Common control whitespaces: \t\n\r\f\v and others
return codepoint in (0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x1C, 0x1D, 0x1E, 0x1F, 0x85, 0xA0)
def build_bitmap(check_func, max_codepoint=0x10FFFF):
"""Build a bitmap for the given check function"""
# We'll use a byte array where each bit represents a codepoint
size = (max_codepoint + 8) // 8
bitmap = bytearray(size)
for cp, cat, _ in parse_unicode_data():
if check_func(cat, cp):
byte_idx = cp // 8
bit_idx = cp % 8
bitmap[byte_idx] |= (1 << bit_idx)
return bitmap
def write_binary_file(path: str, bitmap: bytearray, max_codepoint: int):
"""Write binary file with header: magic, version, max_codepoint, data"""
with open(path, 'wb') as f:
# Magic: 'XICU' (0x58494355)
f.write(struct.pack('<I', 0x58494355))
# Version: 1
f.write(struct.pack('<B', 1))
# Max codepoint (4 bytes)
f.write(struct.pack('<I', max_codepoint))
# Data
f.write(bitmap)
def main():
MAX_CP = 0x10FFFF
print("Building alphanumeric bitmap...")
alnum_bitmap = build_bitmap(is_alphanumeric, MAX_CP)
print("Building whitespace bitmap...")
ws_bitmap = build_bitmap(is_whitespace, MAX_CP)
alnum_path = os.path.join(OUT_DIR, 'alphanumeric.bin')
ws_path = os.path.join(OUT_DIR, 'whitespace.bin')
print(f"Writing {alnum_path} ({len(alnum_bitmap)} bytes)...")
write_binary_file(alnum_path, alnum_bitmap, MAX_CP)
print(f"Writing {ws_path} ({len(ws_bitmap)} bytes)...")
write_binary_file(ws_path, ws_bitmap, MAX_CP)
# Print stats
alnum_count = sum(bin(b).count('1') for b in alnum_bitmap)
ws_count = sum(bin(b).count('1') for b in ws_bitmap)
print(f"Alphanumeric codepoints: {alnum_count}")
print(f"Whitespace codepoints: {ws_count}")
print("Done!")
if __name__ == '__main__':
main()
+143
View File
@@ -0,0 +1,143 @@
#!/usr/bin/env python3
"""
Generate a single binary table with delta encoding for:
- isAlphanumeric: letter (L*) or decimal digit (Nd)
- isWhitespace: Zs, Zl, Zp categories or control whitespace chars
"""
import os
import struct
DATA_DIR = './data'
OUT_DIR = './out'
os.makedirs(DATA_DIR, exist_ok=True)
os.makedirs(OUT_DIR, exist_ok=True)
UNICODE_DATA_PATH = os.path.join(DATA_DIR, 'UnicodeData.txt')
def parse_unicode_data():
"""Parse UnicodeData.txt and yield (codepoint, category)"""
with open(UNICODE_DATA_PATH, 'r', encoding='utf-8') as f:
range_start = None
for line in f:
line = line.strip()
if not line:
continue
parts = line.split(';')
if len(parts) < 3:
continue
codepoint = int(parts[0], 16)
name = parts[1]
category = parts[2]
if name.endswith(', First>'):
range_start = (codepoint, category)
continue
elif name.endswith(', Last>') and range_start:
start_cp, start_cat = range_start
for cp in range(start_cp, codepoint + 1):
yield (cp, start_cat)
range_start = None
continue
yield (codepoint, category)
def get_props(category: str, codepoint: int) -> int:
"""Return 2-bit flags: bit0=alphanumeric, bit1=whitespace"""
flags = 0
# Alphanumeric: Letter (L*) or Decimal digit (Nd)
if category.startswith('L') or category == 'Nd':
flags |= 1 # bit 0
# Whitespace: Z* (Zs, Zl, Zp) or control chars
if category.startswith('Z') or codepoint in (0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x1C, 0x1D, 0x1E, 0x1F, 0x85, 0xA0):
flags |= 2 # bit 1
return flags
def build_delta_ranges():
"""Build delta-encoded ranges: (start, end, flags)"""
ranges = []
current_start = None
current_flags = None
for cp, cat in parse_unicode_data():
flags = get_props(cat, cp)
if current_start is None:
current_start = cp
current_flags = flags
elif flags != current_flags:
# End previous range
ranges.append((current_start, cp - 1, current_flags))
current_start = cp
current_flags = flags
# else: continue current range
# Don't forget the last range
if current_start is not None:
ranges.append((current_start, 0x10FFFF, current_flags))
return ranges
def write_binary(path: str, ranges):
"""Write delta-encoded binary format:
Header: magic(4), version(1), num_ranges(4)
Each range: start(4), end(4), flags(1) - but we'll pack efficiently
Optimized format:
- magic: 'XICU' (0x58494355)
- version: 1
- num_ranges: uint32
- For each range: varint start, varint length, flags(1 byte)
"""
def write_varint(f, val):
while val >= 0x80:
f.write(bytes([(val & 0x7F) | 0x80]))
val >>= 7
f.write(bytes([val]))
with open(path, 'wb') as f:
f.write(struct.pack('<I', 0x58494355)) # magic 'XICU'
f.write(struct.pack('<B', 1)) # version
write_varint(f, len(ranges)) # num_ranges
for start, end, flags in ranges:
length = end - start + 1
write_varint(f, start)
write_varint(f, length)
f.write(bytes([flags]))
def read_varint(data, offset):
val = 0
shift = 0
while True:
b = data[offset]
offset += 1
val |= (b & 0x7F) << shift
if not (b & 0x80):
break
shift += 7
return val, offset
def main():
print("Building delta-encoded ranges...")
ranges = build_delta_ranges()
print(f"Total ranges: {len(ranges)}")
# Stats
alnum_count = sum(end - start + 1 for start, end, f in ranges if f & 1)
ws_count = sum(end - start + 1 for start, end, f in ranges if f & 2)
print(f"Alphanumeric codepoints: {alnum_count}")
print(f"Whitespace codepoints: {ws_count}")
out_path = os.path.join(OUT_DIR, 'props.bin')
print(f"Writing {out_path}...")
write_binary(out_path, ranges)
size = os.path.getsize(out_path)
print(f"File size: {size} bytes ({size/1024:.1f} KB)")
print("Done!")
if __name__ == '__main__':
main()
+906
View File
@@ -0,0 +1,906 @@
#!/usr/bin/env python3
"""
x-icu-gen: Configurable ICU data generator with optimal compression.
Generates binary tables + C++ implementation from UnicodeData.txt
"""
import os
import struct
import json
from pathlib import Path
from typing import Dict, List, Tuple, Any, Optional, Set
from dataclasses import dataclass, field
from enum import Enum
from collections import defaultdict
# ============================================================
# CONFIGURATION (from py-gen.ipynb)
# ============================================================
CONFIG = {
'sources': {
'unicode': 'https://www.unicode.org/Public/UCD/latest/ucd/UnicodeData.txt',
},
'unicode': {
'useOldName': True,
'getName': True,
'getDecomposition': False,
'getDecompositionType': False,
'toLowercase': True,
'toUppercase': True,
'toTitlecase': False,
'isPunctuation': True,
'isSymbol': False,
'isCombining': False,
'isPrintable': False,
'isSpace': True,
'isWhitespace': True,
'isLetter': True,
'isUppercase': False,
'isLowercase': False,
'isTitlecase': False,
'isDeprecated': False,
'isDecimal': True,
'isDigit': False,
'isNumberLike': False,
'getDecimal': True,
'getDigit': False,
'getNumberLike': False,
}
}
# Field definitions: maps config key -> (output_name, type, category)
# category: 'bool', 'int', 'str', 'enum'
FIELD_DEFS = {
'getName': ('name', 'str', 'str'),
'getDecomposition': ('decomposition', 'str', 'str'),
'getDecompositionType': ('decomposition_type', 'str', 'str'),
'isPunctuation': ('punctuation', 'bool', 'bool'),
'isSymbol': ('symbol', 'bool', 'bool'),
'isCombining': ('combining', 'bool', 'bool'),
'isLetter': ('letter', 'bool', 'bool'),
'isUppercase': ('uppercase', 'bool', 'bool'),
'isLowercase': ('lowercase', 'bool', 'bool'),
'isTitlecase': ('titlecase', 'bool', 'bool'),
'toUppercase': ('uppercase_map', 'int', 'int'),
'toLowercase': ('lowercase_map', 'int', 'int'),
'toTitlecase': ('titlecase_map', 'int', 'int'),
'isWhitespace': ('whitespace', 'bool', 'bool'),
'isPrintable': ('printable', 'bool', 'bool'),
'isSpace': ('space', 'bool', 'bool'),
'isDecimal': ('decimal', 'bool', 'bool'),
'isDigit': ('digit', 'bool', 'bool'),
'isNumberLike': ('number_like', 'bool', 'bool'),
'getDecimal': ('decimal_val', 'int', 'int'),
'getDigit': ('digit_val', 'int', 'int'),
'getNumberLike': ('number_like_val', 'int', 'int'),
}
# ============================================================
# DATA STRUCTURES
# ============================================================
class CompStrategy(Enum):
BITMAP = "bitmap" # 1 bit per codepoint
DELTA_RANGES = "delta_ranges" # (start, end, value) runs
RLE = "rle" # Run-length encoding
SPARSE_MAP = "sparse_map" # Only store non-default values
STRING_TABLE = "string_table" # Separate string pool
@dataclass
class FieldConfig:
name: str
type: str # 'bool', 'int', 'str'
enabled: bool = False
strategy: CompStrategy = CompStrategy.DELTA_RANGES
default_value: Any = None
@dataclass
class CodePointData:
cp: int
name: str = ""
category: str = ""
decomposition: str = ""
decimal_val: str = ""
digit_val: str = ""
numeric_val: str = ""
bidi_mirrored: str = ""
unicode_1_name: str = ""
uppercase_map: str = ""
lowercase_map: str = ""
titlecase_map: str = ""
@dataclass
class ProcessedRow:
cp: int
fields: Dict[str, Any] = field(default_factory=dict)
# ============================================================
# UNICODE PARSER
# ============================================================
class UnicodeParser:
def __init__(self, data_path: str):
self.data_path = data_path
self.keys = [
"codepoint", "name", "category", "combining_class", "bidi_class",
"decomposition", "decimal_val", "digit_val", "numeric_val",
"bidi_mirrored", "unicode_1_name", "iso_comment",
"uppercase_map", "lowercase_map", "titlecase_map"
]
def parse(self) -> List[CodePointData]:
results = []
range_start = None
with open(self.data_path, 'r', encoding='utf-8') as f:
for line in f:
line = line.strip()
if not line:
continue
parts = line.split(';')
if len(parts) < 15:
continue
cp = int(parts[0], 16)
name = parts[1]
cat = parts[2]
decomp = parts[5]
dec_val = parts[6]
dig_val = parts[7]
num_val = parts[8]
bidi_mir = parts[9]
old_name = parts[10]
up_map = parts[12]
low_map = parts[13]
title_map = parts[14]
if name.endswith(', First>'):
range_start = (cp, cat, name, decomp, dec_val, dig_val, num_val,
bidi_mir, old_name, up_map, low_map, title_map)
continue
elif name.endswith(', Last>') and range_start:
sc, scat, sname, sdecomp, sdec, sdig, snum, sbidi, sold, sup, slow, stitle = range_start
base_name = sname.replace(', First>', '').replace('<', '')
for c in range(sc, cp + 1):
results.append(CodePointData(
cp=c, name=f"<{base_name}>", category=scat,
decomposition=sdecomp, decimal_val=sdec, digit_val=sdig,
numeric_val=snum, bidi_mirrored=sbidi, unicode_1_name=sold,
uppercase_map=sup, lowercase_map=slow, titlecase_map=stitle
))
range_start = None
continue
results.append(CodePointData(
cp=cp, name=name, category=cat, decomposition=decomp,
decimal_val=dec_val, digit_val=dig_val, numeric_val=num_val,
bidi_mirrored=bidi_mir, unicode_1_name=old_name,
uppercase_map=up_map, lowercase_map=low_map, titlecase_map=title_map
))
return results
# ============================================================
# FIELD PROCESSORS
# ============================================================
def process_row(row: CodePointData, cfg: Dict, use_old_name: bool) -> ProcessedRow:
"""Extract all configured fields from a parsed row."""
out = ProcessedRow(cp=row.cp)
cat = row.category
char = chr(row.cp) if row.cp <= 0x10FFFF else ''
# Name handling
name = row.name
if use_old_name and name.startswith('<') and row.unicode_1_name:
name = row.unicode_1_name
# Boolean properties from category
out.fields['punctuation'] = cat.startswith('P')
out.fields['symbol'] = cat.startswith('S')
out.fields['combining'] = cat.startswith('M')
out.fields['letter'] = cat.startswith('L')
out.fields['uppercase'] = cat == 'Lu'
out.fields['lowercase'] = cat == 'Ll'
out.fields['titlecase'] = cat == 'Lt'
out.fields['whitespace'] = cat.startswith('Z') or row.cp in (0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x1C, 0x1D, 0x1E, 0x1F, 0x85, 0xA0)
out.fields['space'] = cat == 'Zs'
out.fields['printable'] = not cat.startswith('C')
# Numeric properties
is_decimal = bool(row.decimal_val)
is_digit = is_decimal or bool(row.digit_val)
is_numeric = is_digit or bool(row.numeric_val)
out.fields['decimal'] = is_decimal
out.fields['digit'] = is_digit
out.fields['number_like'] = is_numeric
out.fields['decimal_val'] = int(row.decimal_val) if is_decimal else 0
out.fields['digit_val'] = int(row.digit_val) if is_digit else 0
out.fields['number_like_val'] = row.numeric_val if is_numeric else ""
# String properties
out.fields['name'] = name
out.fields['decomposition'] = row.decomposition
decomp_type = ""
if row.decomposition and '<' in row.decomposition:
decomp_type = row.decomposition[1:row.decomposition.find('>')]
out.fields['decomposition_type'] = decomp_type
# Case mappings (relative offsets)
for k, src in [('uppercase_map', row.uppercase_map),
('lowercase_map', row.lowercase_map),
('titlecase_map', row.titlecase_map)]:
val = int(src, 16) if src else 0
out.fields[k] = val - row.cp if val != 0 else 0
return out
# ============================================================
# COMPRESSION ANALYSIS
# ============================================================
def analyze_field(rows: List[ProcessedRow], field_name: str, field_type: str) -> CompStrategy:
"""Determine optimal compression strategy for a field."""
values = [r.fields.get(field_name, None) for r in rows]
non_default = [v for v in values if v not in (False, 0, "", None)]
unique_vals = set(v for v in values if v not in (False, 0, "", None))
total = len(rows)
sparse_ratio = len(non_default) / total if total > 0 else 0
if field_type == 'str':
return CompStrategy.STRING_TABLE
if field_type == 'bool':
# If very sparse (< 1%), use sparse map
if sparse_ratio < 0.01:
return CompStrategy.SPARSE_MAP
# If dense (> 50%), bitmap is good
if sparse_ratio > 0.5:
return CompStrategy.BITMAP
# Otherwise delta ranges
return CompStrategy.DELTA_RANGES
if field_type == 'int':
# Check if values are mostly small offsets (case mappings)
if field_name.endswith('_map'):
# These are relative offsets, usually small or zero
if sparse_ratio < 0.05:
return CompStrategy.SPARSE_MAP
return CompStrategy.DELTA_RANGES
# For numeric values (decimal_val, etc.)
if len(unique_vals) < 20:
return CompStrategy.DELTA_RANGES
return CompStrategy.SPARSE_MAP
return CompStrategy.DELTA_RANGES
# ============================================================
# ENCODERS
# ============================================================
def write_varint(buf: bytearray, val: int):
while val >= 0x80:
buf.append((val & 0x7F) | 0x80)
val >>= 7
buf.append(val)
def read_varint(data: bytes, offset: int) -> Tuple[int, int]:
val = 0
shift = 0
while True:
b = data[offset]
offset += 1
val |= (b & 0x7F) << shift
if not (b & 0x80):
break
shift += 7
return val, offset
def encode_bitmap(rows: List[ProcessedRow], field: str) -> bytes:
"""Encode boolean field as bitmap."""
max_cp = max(r.cp for r in rows)
size = (max_cp + 8) // 8
bitmap = bytearray(size)
for r in rows:
if r.fields.get(field, False):
idx = r.cp >> 3
bitmap[idx] |= 1 << (r.cp & 7)
return bytes(bitmap)
def encode_delta_ranges(rows: List[ProcessedRow], field: str, field_type: str) -> bytes:
"""Encode as (start, length, value) runs with varints."""
buf = bytearray()
current_start = None
current_val = None
for r in rows:
val = r.fields.get(field)
if field_type == 'bool':
val = bool(val)
elif field_type == 'int':
val = int(val) if val else 0
else:
val = val or ""
if current_start is None:
current_start = r.cp
current_val = val
elif val != current_val:
# End previous range
length = r.cp - current_start
write_varint(buf, current_start)
write_varint(buf, length)
if field_type == 'bool':
buf.append(1 if current_val else 0)
elif field_type == 'int':
# Zigzag encode for negative values
v = current_val
write_varint(buf, (v << 1) ^ (v >> 31))
current_start = r.cp
current_val = val
# Last range
if current_start is not None:
length = 0x10FFFF - current_start + 1
write_varint(buf, current_start)
write_varint(buf, length)
if field_type == 'bool':
buf.append(1 if current_val else 0)
elif field_type == 'int':
v = current_val
write_varint(buf, (v << 1) ^ (v >> 31))
return bytes(buf)
def encode_sparse_map(rows: List[ProcessedRow], field: str, field_type: str) -> bytes:
"""Store only non-default values as (cp, value) pairs."""
buf = bytearray()
for r in rows:
val = r.fields.get(field)
if field_type == 'bool':
val = bool(val)
if not val:
continue
write_varint(buf, r.cp)
buf.append(1)
elif field_type == 'int':
val = int(val) if val else 0
if val == 0:
continue
write_varint(buf, r.cp)
v = val
write_varint(buf, (v << 1) ^ (v >> 31))
else:
val = val or ""
if not val:
continue
write_varint(buf, r.cp)
# String index will be handled separately
write_varint(buf, 0) # placeholder
return bytes(buf)
def encode_string_table(rows: List[ProcessedRow], field: str) -> Tuple[bytes, List[str]]:
"""Build string table and return (indices, string_pool)."""
strings = []
str_to_idx = {}
indices = []
for r in rows:
val = r.fields.get(field, "") or ""
if val not in str_to_idx:
str_to_idx[val] = len(strings)
strings.append(val)
indices.append(str_to_idx[val])
# Encode indices as varints per codepoint
buf = bytearray()
for idx in indices:
write_varint(buf, idx)
return bytes(buf), strings
# ============================================================
# MAIN GENERATOR
# ============================================================
class XicuGenerator:
def __init__(self, config: Dict, data_dir: str = './data', out_dir: str = './out'):
self.config = config
self.data_dir = Path(data_dir)
self.out_dir = Path(out_dir)
self.out_dir.mkdir(parents=True, exist_ok=True)
self.data_dir.mkdir(parents=True, exist_ok=True)
self.rows: List[ProcessedRow] = []
self.field_configs: Dict[str, FieldConfig] = {}
self.strategies: Dict[str, CompStrategy] = {}
self.string_tables: Dict[str, List[str]] = {}
self.encoded_data: Dict[str, bytes] = {}
def download_unicode_data(self):
import urllib.request
url = self.config['sources']['unicode']
path = self.data_dir / 'UnicodeData.txt'
if not path.exists():
print(f"Downloading {url}...")
urllib.request.urlretrieve(url, path)
print("Done.")
def parse_and_process(self):
parser = UnicodeParser(str(self.data_dir / 'UnicodeData.txt'))
raw_rows = parser.parse()
print(f"Parsed {len(raw_rows)} codepoints")
ucfg = self.config['unicode']
use_old_name = ucfg.get('useOldName', False)
for row in raw_rows:
self.rows.append(process_row(row, ucfg, use_old_name))
print(f"Processed {len(self.rows)} rows")
def analyze_and_configure(self):
ucfg = self.config['unicode']
for cfg_key, (out_name, ftype, cat) in FIELD_DEFS.items():
if not ucfg.get(cfg_key, False):
continue
strategy = analyze_field(self.rows, out_name, ftype)
default = False if ftype == 'bool' else (0 if ftype == 'int' else "")
self.field_configs[out_name] = FieldConfig(
name=out_name, type=ftype, enabled=True,
strategy=strategy, default_value=default
)
self.strategies[out_name] = strategy
print("Field strategies:")
for name, fc in self.field_configs.items():
print(f" {name} ({fc.type}): {fc.strategy.value}")
def encode_all(self):
# First pass: encode non-string fields
for name, fc in self.field_configs.items():
if fc.type == 'str':
continue
print(f"Encoding {name} with {fc.strategy.value}...")
if fc.strategy == CompStrategy.BITMAP:
self.encoded_data[name] = encode_bitmap(self.rows, name)
elif fc.strategy == CompStrategy.DELTA_RANGES:
self.encoded_data[name] = encode_delta_ranges(self.rows, name, fc.type)
elif fc.strategy == CompStrategy.SPARSE_MAP:
self.encoded_data[name] = encode_sparse_map(self.rows, name, fc.type)
# Second pass: string fields
for name, fc in self.field_configs.items():
if fc.type != 'str':
continue
print(f"Encoding string table for {name}...")
indices, strings = encode_string_table(self.rows, name)
self.encoded_data[name + '_indices'] = indices
self.string_tables[name] = strings
def write_binary(self):
"""Write main binary file with all non-string fields."""
path = self.out_dir / 'xicu.bin'
buf = bytearray()
# Header
buf.extend(struct.pack('<I', 0x58494355)) # XICU magic
buf.extend(struct.pack('<B', 2)) # version 2
non_str_fields = [n for n, fc in self.field_configs.items() if fc.type != 'str']
write_varint(buf, len(non_str_fields))
# Field descriptors + data inline
for name, fc in self.field_configs.items():
if fc.type == 'str':
continue
buf.extend(struct.pack('<B',
{'bool': 1, 'int': 2}[fc.type]))
buf.extend(struct.pack('<B',
{'bitmap': 1, 'delta_ranges': 2, 'rle': 3, 'sparse_map': 4}[fc.strategy.value]))
# Field name as null-terminated
buf.extend(name.encode('ascii'))
buf.append(0)
# Data length + data
data = self.encoded_data[name]
write_varint(buf, len(data))
buf.extend(data)
# String field descriptors
str_fields = [n for n, fc in self.field_configs.items() if fc.type == 'str']
write_varint(buf, len(str_fields))
for name in str_fields:
buf.extend(name.encode('ascii'))
buf.append(0)
# String indices
for name in str_fields:
data = self.encoded_data[name + '_indices']
write_varint(buf, len(data))
buf.extend(data)
with open(path, 'wb') as f:
f.write(buf)
print(f"Wrote {path} ({len(buf)} bytes)")
# Write string pools separately
for name, strings in self.string_tables.items():
pool_path = self.out_dir / f'xicu_{name}.str'
buf = bytearray()
write_varint(buf, len(strings))
for s in strings:
enc = s.encode('utf-8')
write_varint(buf, len(enc))
buf.extend(enc)
with open(pool_path, 'wb') as f:
f.write(buf)
print(f"Wrote {pool_path} ({len(buf)} bytes)")
def generate_cpp(self):
"""Generate C++ header and implementation."""
hpp_path = Path('cpp') / 'xicu.hpp'
cpp_path = Path('cpp') / 'xicu.cpp'
Path('cpp').mkdir(exist_ok=True)
# Generate header
hpp = self._gen_header()
with open(hpp_path, 'w') as f:
f.write(hpp)
# Generate implementation
cpp = self._gen_impl()
with open(cpp_path, 'w') as f:
f.write(cpp)
print(f"Generated {hpp_path} and {cpp_path}")
def _gen_header(self) -> str:
lines = [
'#pragma once',
'',
'#include <cstdint>',
'#include <string_view>',
'',
'namespace xicu {',
'',
'class PropTable {',
'public:',
' PropTable() = default;',
' explicit PropTable(const char* path);',
' ~PropTable();',
'',
' PropTable(const PropTable&) = delete;',
' PropTable& operator=(const PropTable&) = delete;',
' PropTable(PropTable&& other) noexcept;',
' PropTable& operator=(PropTable&& other) noexcept;',
'',
' bool load(const char* path);',
' bool isLoaded() const { return data_ != nullptr; }',
]
# Generate getter declarations
for name, fc in self.field_configs.items():
if fc.type == 'bool':
lines.append(f' bool get{name.capitalize()}(uint32_t cp) const;')
elif fc.type == 'int':
lines.append(f' int32_t get{name.capitalize()}(uint32_t cp) const;')
elif fc.type == 'str':
lines.append(f' std::string_view get{name.capitalize()}(uint32_t cp) const;')
lines.extend([
'',
'private:',
' const uint8_t* data_ = nullptr;',
' size_t data_size_ = 0;',
' bool owns_data_ = false;',
'',
' static uint32_t readVarint(const uint8_t*& ptr);',
' const uint8_t* findFieldData(const char* name, uint8_t& out_type, uint8_t& out_strat) const;',
' const uint8_t* findStringField(const char* name) const;',
'',
' // Decoders',
' bool decodeBitmap(const uint8_t* data, uint32_t cp) const;',
' bool decodeDeltaRangesBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;',
' int32_t decodeDeltaRangesInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;',
' bool decodeSparseMapBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;',
' int32_t decodeSparseMapInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;',
'};',
'',
])
# Inline convenience functions
for name, fc in self.field_configs.items():
if fc.type == 'bool':
lines.append(f'inline bool is{name.capitalize()}(uint32_t cp) {{')
lines.append(f' static PropTable table("../out/xicu.bin");')
lines.append(f' return table.get{name.capitalize()}(cp);')
lines.append(f'}}')
elif fc.type == 'int':
lines.append(f'inline int32_t get{name.capitalize()}(uint32_t cp) {{')
lines.append(f' static PropTable table("../out/xicu.bin");')
lines.append(f' return table.get{name.capitalize()}(cp);')
lines.append(f'}}')
elif fc.type == 'str':
lines.append(f'inline std::string_view get{name.capitalize()}(uint32_t cp) {{')
lines.append(f' static PropTable table("../out/xicu.bin");')
lines.append(f' return table.get{name.capitalize()}(cp);')
lines.append(f'}}')
lines.extend([
'',
'} // namespace xicu',
])
return '\n'.join(lines)
def _gen_impl(self) -> str:
"""Generate complete C++ implementation with decoders."""
lines = [
'#include "xicu.hpp"',
'#include <cstdio>',
'#include <cstdint>',
'#include <cstring>',
'#include <string_view>',
'',
'namespace xicu {',
'',
'static constexpr uint32_t MAGIC = 0x58494355;',
'static constexpr uint8_t VERSION = 2;',
'',
'PropTable::PropTable(const char* path) { load(path); }',
'',
'PropTable::~PropTable() {',
' if (owns_data_ && data_) { delete[] data_; data_ = nullptr; }',
'}',
'',
'PropTable::PropTable(PropTable&& other) noexcept',
' : data_(other.data_), data_size_(other.data_size_), owns_data_(other.owns_data_) {',
' other.data_ = nullptr; other.owns_data_ = false;',
'}',
'',
'PropTable& PropTable::operator=(PropTable&& other) noexcept {',
' if (this != &other) {',
' if (owns_data_ && data_) delete[] data_;',
' data_ = other.data_; data_size_ = other.data_size_; owns_data_ = other.owns_data_;',
' other.data_ = nullptr; other.owns_data_ = false;',
' }',
' return *this;',
'}',
'',
'uint32_t PropTable::readVarint(const uint8_t*& ptr) {',
' uint32_t val = 0; int shift = 0;',
' while (true) {',
' uint8_t b = *ptr++;',
' val |= (b & 0x7F) << shift;',
' if (!(b & 0x80)) break;',
' shift += 7;',
' }',
' return val;',
'}',
'',
'bool PropTable::load(const char* path) {',
' if (owns_data_ && data_) { delete[] data_; data_ = nullptr; owns_data_ = false; }',
' FILE* f = std::fopen(path, "rb");',
' if (!f) return false;',
' std::fseek(f, 0, SEEK_END);',
' long fsize = std::ftell(f);',
' std::fseek(f, 0, SEEK_SET);',
' if (fsize < 6) { std::fclose(f); return false; }',
' data_size_ = static_cast<size_t>(fsize);',
' uint8_t* buf = new uint8_t[data_size_];',
' size_t read = std::fread(buf, 1, data_size_, f);',
' std::fclose(f);',
' if (read != data_size_) { delete[] buf; return false; }',
' data_ = buf; owns_data_ = true;',
' ',
' const uint8_t* ptr = data_;',
' uint32_t magic = *reinterpret_cast<const uint32_t*>(ptr); ptr += 4;',
' if (magic != MAGIC) return false;',
' uint8_t version = *ptr++;',
' if (version != VERSION) return false;',
' return true;',
'}',
'',
'// Find field data pointer (points to data length varint after name)',
'const uint8_t* PropTable::findFieldData(const char* name, uint8_t& out_type, uint8_t& out_strat) const {',
' if (!data_) return nullptr;',
' const uint8_t* ptr = data_ + 5; // magic + version',
' uint32_t num_fields = readVarint(ptr);',
' for (uint32_t i = 0; i < num_fields; ++i) {',
' uint8_t ftype = *ptr++;',
' uint8_t fstrat = *ptr++;',
' const char* fname = reinterpret_cast<const char*>(ptr);',
' size_t name_len = std::strlen(fname);',
' ptr += name_len + 1;',
' if (std::strcmp(fname, name) == 0) {',
' out_type = ftype;',
' out_strat = fstrat;',
' return ptr; // points to data length varint',
' }',
' // Skip data',
' uint32_t data_len = readVarint(ptr);',
' ptr += data_len;',
' }',
' return nullptr;',
'}',
'',
'// Find string field (returns pointer to indices data length)',
'const uint8_t* PropTable::findStringField(const char* name) const {',
' if (!data_) return nullptr;',
' const uint8_t* ptr = data_ + 5;',
' uint32_t num_fields = readVarint(ptr);',
' // Skip non-string fields (with inline data)',
' for (uint32_t i = 0; i < num_fields; ++i) {',
' ptr++; ptr++; // type, strat',
' while (*ptr++) {} // skip name',
' uint32_t data_len = readVarint(ptr);',
' ptr += data_len; // skip data',
' }',
' uint32_t num_str_fields = readVarint(ptr);',
' for (uint32_t i = 0; i < num_str_fields; ++i) {',
' const char* fname = reinterpret_cast<const char*>(ptr);',
' size_t name_len = std::strlen(fname);',
' ptr += name_len + 1;',
' if (std::strcmp(fname, name) == 0) {',
' return ptr; // points to indices data length',
' }',
' uint32_t data_len = readVarint(ptr);',
' ptr += data_len; // skip indices data',
' }',
' return nullptr;',
'}',
'',
'// Decode bitmap: 1 bit per codepoint',
'bool PropTable::decodeBitmap(const uint8_t* data, uint32_t cp) const {',
' size_t byte_idx = cp >> 3;',
' uint8_t bit = cp & 7;',
' return (data[byte_idx] & (1 << bit)) != 0;',
'}',
'',
'// Decode delta ranges: (start, length, value) varint-encoded',
'bool PropTable::decodeDeltaRangesBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {',
' const uint8_t* ptr = data;',
' while (ptr < data_end) {',
' uint32_t start = readVarint(ptr);',
' uint32_t length = readVarint(ptr);',
' uint8_t val = *ptr++;',
' uint32_t end = start + length - 1;',
' if (cp >= start && cp <= end) return val != 0;',
' if (cp < start) return false;',
' }',
' return false;',
'}',
'',
'int32_t PropTable::decodeDeltaRangesInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {',
' const uint8_t* ptr = data;',
' while (ptr < data_end) {',
' uint32_t start = readVarint(ptr);',
' uint32_t length = readVarint(ptr);',
' uint32_t zigzag = readVarint(ptr);',
' int32_t val = static_cast<int32_t>((zigzag >> 1) ^ -(zigzag & 1));',
' uint32_t end = start + length - 1;',
' if (cp >= start && cp <= end) return val;',
' if (cp < start) return 0;',
' }',
' return 0;',
'}',
'',
'// Decode sparse map: (cp, value) pairs',
'bool PropTable::decodeSparseMapBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {',
' const uint8_t* ptr = data;',
' while (ptr < data_end) {',
' uint32_t c = readVarint(ptr);',
' if (ptr >= data_end) break;',
' uint8_t val = *ptr++;',
' if (c == cp) return val != 0;',
' if (c > cp) return false;',
' }',
' return false;',
'}',
'',
'int32_t PropTable::decodeSparseMapInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {',
' const uint8_t* ptr = data;',
' while (ptr < data_end) {',
' uint32_t c = readVarint(ptr);',
' if (ptr >= data_end) break;',
' uint32_t zigzag = readVarint(ptr);',
' int32_t val = static_cast<int32_t>((zigzag >> 1) ^ -(zigzag & 1));',
' if (c == cp) return val;',
' if (c > cp) return 0;',
' }',
' return 0;',
'}',
'',
]
# Generate getters for each field
for name, fc in self.field_configs.items():
if fc.type == 'bool':
strat = fc.strategy.value
if strat == 'bitmap':
decoder = 'decodeBitmap'
call = 'decodeBitmap(data, cp)'
elif strat == 'delta_ranges':
decoder = 'decodeDeltaRangesBool'
call = 'decodeDeltaRangesBool(data, data_end, cp)'
elif strat == 'sparse_map':
decoder = 'decodeSparseMapBool'
call = 'decodeSparseMapBool(data, data_end, cp)'
else:
decoder = 'decodeDeltaRangesBool'
call = 'decodeDeltaRangesBool(data, data_end, cp)'
lines.extend([
f'bool PropTable::get{name.capitalize()}(uint32_t cp) const {{',
f' uint8_t ftype, fstrat;',
f' const uint8_t* data = findFieldData("{name}", ftype, fstrat);',
f' if (!data) return false;',
f' uint32_t data_len = readVarint(data);',
f' const uint8_t* data_end = data + data_len;',
f' return {call};',
f'}}',
'',
])
elif fc.type == 'int':
strat = fc.strategy.value
if strat == 'delta_ranges':
decoder = 'decodeDeltaRangesInt'
elif strat == 'sparse_map':
decoder = 'decodeSparseMapInt'
else:
decoder = 'decodeDeltaRangesInt'
lines.extend([
f'int32_t PropTable::get{name.capitalize()}(uint32_t cp) const {{',
f' uint8_t ftype, fstrat;',
f' const uint8_t* data = findFieldData("{name}", ftype, fstrat);',
f' if (!data) return 0;',
f' uint32_t data_len = readVarint(data);',
f' const uint8_t* data_end = data + data_len;',
f' return {decoder}(data, data_end, cp);',
f'}}',
'',
])
elif fc.type == 'str':
lines.extend([
f'std::string_view PropTable::get{name.capitalize()}(uint32_t cp) const {{',
f' // String fields use separate .str file',
f' const uint8_t* indices_ptr = findStringField("{name}");',
f' if (!indices_ptr) return "";',
f' uint32_t indices_len = readVarint(indices_ptr);',
f' // Find index for this codepoint (linear scan for now)',
f' // TODO: optimize with binary search if needed',
f' // For now, we need the string pool loaded separately',
f' return ""; // Requires string pool file',
f'}}',
'',
])
lines.append('} // namespace xicu')
return '\n'.join(lines)
def run(self):
self.download_unicode_data()
self.parse_and_process()
self.analyze_and_configure()
self.encode_all()
self.write_binary()
self.generate_cpp()
print("\nGeneration complete!")
def main():
gen = XicuGenerator(CONFIG)
gen.run()
if __name__ == '__main__':
main()