#!/usr/bin/env python3 """ x-icu-gen: Configurable ICU data generator with optimal compression. Generates binary tables + C++ implementation from UnicodeData.txt """ import os import struct import json from pathlib import Path from typing import Dict, List, Tuple, Any, Optional, Set from dataclasses import dataclass, field from enum import Enum from collections import defaultdict # ============================================================ # CONFIGURATION (from py-gen.ipynb) # ============================================================ CONFIG = { 'sources': { 'unicode': 'https://www.unicode.org/Public/UCD/latest/ucd/UnicodeData.txt', }, 'unicode': { 'useOldName': True, 'getName': True, 'getDecomposition': False, 'getDecompositionType': False, 'toLowercase': True, 'toUppercase': True, 'toTitlecase': False, 'isPunctuation': True, 'isSymbol': False, 'isCombining': False, 'isPrintable': False, 'isSpace': True, 'isWhitespace': True, 'isLetter': True, 'isUppercase': False, 'isLowercase': False, 'isTitlecase': False, 'isDeprecated': False, 'isDecimal': True, 'isDigit': False, 'isNumberLike': False, 'getDecimal': True, 'getDigit': False, 'getNumberLike': False, } } # Field definitions: maps config key -> (output_name, type, category) # category: 'bool', 'int', 'str', 'enum' FIELD_DEFS = { 'getName': ('name', 'str', 'str'), 'getDecomposition': ('decomposition', 'str', 'str'), 'getDecompositionType': ('decomposition_type', 'str', 'str'), 'isPunctuation': ('punctuation', 'bool', 'bool'), 'isSymbol': ('symbol', 'bool', 'bool'), 'isCombining': ('combining', 'bool', 'bool'), 'isLetter': ('letter', 'bool', 'bool'), 'isUppercase': ('uppercase', 'bool', 'bool'), 'isLowercase': ('lowercase', 'bool', 'bool'), 'isTitlecase': ('titlecase', 'bool', 'bool'), 'toUppercase': ('uppercase_map', 'int', 'int'), 'toLowercase': ('lowercase_map', 'int', 'int'), 'toTitlecase': ('titlecase_map', 'int', 'int'), 'isWhitespace': ('whitespace', 'bool', 'bool'), 'isPrintable': ('printable', 'bool', 'bool'), 'isSpace': ('space', 'bool', 'bool'), 'isDecimal': ('decimal', 'bool', 'bool'), 'isDigit': ('digit', 'bool', 'bool'), 'isNumberLike': ('number_like', 'bool', 'bool'), 'getDecimal': ('decimal_val', 'int', 'int'), 'getDigit': ('digit_val', 'int', 'int'), 'getNumberLike': ('number_like_val', 'int', 'int'), } # ============================================================ # DATA STRUCTURES # ============================================================ class CompStrategy(Enum): BITMAP = "bitmap" # 1 bit per codepoint DELTA_RANGES = "delta_ranges" # (start, end, value) runs RLE = "rle" # Run-length encoding SPARSE_MAP = "sparse_map" # Only store non-default values STRING_TABLE = "string_table" # Separate string pool @dataclass class FieldConfig: name: str type: str # 'bool', 'int', 'str' enabled: bool = False strategy: CompStrategy = CompStrategy.DELTA_RANGES default_value: Any = None @dataclass class CodePointData: cp: int name: str = "" category: str = "" decomposition: str = "" decimal_val: str = "" digit_val: str = "" numeric_val: str = "" bidi_mirrored: str = "" unicode_1_name: str = "" uppercase_map: str = "" lowercase_map: str = "" titlecase_map: str = "" @dataclass class ProcessedRow: cp: int fields: Dict[str, Any] = field(default_factory=dict) # ============================================================ # UNICODE PARSER # ============================================================ class UnicodeParser: def __init__(self, data_path: str): self.data_path = data_path self.keys = [ "codepoint", "name", "category", "combining_class", "bidi_class", "decomposition", "decimal_val", "digit_val", "numeric_val", "bidi_mirrored", "unicode_1_name", "iso_comment", "uppercase_map", "lowercase_map", "titlecase_map" ] def parse(self) -> List[CodePointData]: results = [] range_start = None with open(self.data_path, 'r', encoding='utf-8') as f: for line in f: line = line.strip() if not line: continue parts = line.split(';') if len(parts) < 15: continue cp = int(parts[0], 16) name = parts[1] cat = parts[2] decomp = parts[5] dec_val = parts[6] dig_val = parts[7] num_val = parts[8] bidi_mir = parts[9] old_name = parts[10] up_map = parts[12] low_map = parts[13] title_map = parts[14] if name.endswith(', First>'): range_start = (cp, cat, name, decomp, dec_val, dig_val, num_val, bidi_mir, old_name, up_map, low_map, title_map) continue elif name.endswith(', Last>') and range_start: sc, scat, sname, sdecomp, sdec, sdig, snum, sbidi, sold, sup, slow, stitle = range_start base_name = sname.replace(', First>', '').replace('<', '') for c in range(sc, cp + 1): results.append(CodePointData( cp=c, name=f"<{base_name}>", category=scat, decomposition=sdecomp, decimal_val=sdec, digit_val=sdig, numeric_val=snum, bidi_mirrored=sbidi, unicode_1_name=sold, uppercase_map=sup, lowercase_map=slow, titlecase_map=stitle )) range_start = None continue results.append(CodePointData( cp=cp, name=name, category=cat, decomposition=decomp, decimal_val=dec_val, digit_val=dig_val, numeric_val=num_val, bidi_mirrored=bidi_mir, unicode_1_name=old_name, uppercase_map=up_map, lowercase_map=low_map, titlecase_map=title_map )) return results # ============================================================ # FIELD PROCESSORS # ============================================================ def process_row(row: CodePointData, cfg: Dict, use_old_name: bool) -> ProcessedRow: """Extract all configured fields from a parsed row.""" out = ProcessedRow(cp=row.cp) cat = row.category char = chr(row.cp) if row.cp <= 0x10FFFF else '' # Name handling name = row.name if use_old_name and name.startswith('<') and row.unicode_1_name: name = row.unicode_1_name # Boolean properties from category out.fields['punctuation'] = cat.startswith('P') out.fields['symbol'] = cat.startswith('S') out.fields['combining'] = cat.startswith('M') out.fields['letter'] = cat.startswith('L') out.fields['uppercase'] = cat == 'Lu' out.fields['lowercase'] = cat == 'Ll' out.fields['titlecase'] = cat == 'Lt' out.fields['whitespace'] = cat.startswith('Z') or row.cp in (0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x1C, 0x1D, 0x1E, 0x1F, 0x85, 0xA0) out.fields['space'] = cat == 'Zs' out.fields['printable'] = not cat.startswith('C') # Numeric properties is_decimal = bool(row.decimal_val) is_digit = is_decimal or bool(row.digit_val) is_numeric = is_digit or bool(row.numeric_val) out.fields['decimal'] = is_decimal out.fields['digit'] = is_digit out.fields['number_like'] = is_numeric out.fields['decimal_val'] = int(row.decimal_val) if is_decimal else 0 out.fields['digit_val'] = int(row.digit_val) if is_digit else 0 out.fields['number_like_val'] = row.numeric_val if is_numeric else "" # String properties out.fields['name'] = name out.fields['decomposition'] = row.decomposition decomp_type = "" if row.decomposition and '<' in row.decomposition: decomp_type = row.decomposition[1:row.decomposition.find('>')] out.fields['decomposition_type'] = decomp_type # Case mappings (relative offsets) for k, src in [('uppercase_map', row.uppercase_map), ('lowercase_map', row.lowercase_map), ('titlecase_map', row.titlecase_map)]: val = int(src, 16) if src else 0 out.fields[k] = val - row.cp if val != 0 else 0 return out # ============================================================ # COMPRESSION ANALYSIS # ============================================================ def analyze_field(rows: List[ProcessedRow], field_name: str, field_type: str) -> CompStrategy: """Determine optimal compression strategy for a field.""" values = [r.fields.get(field_name, None) for r in rows] non_default = [v for v in values if v not in (False, 0, "", None)] unique_vals = set(v for v in values if v not in (False, 0, "", None)) total = len(rows) sparse_ratio = len(non_default) / total if total > 0 else 0 if field_type == 'str': return CompStrategy.STRING_TABLE if field_type == 'bool': # If very sparse (< 1%), use sparse map if sparse_ratio < 0.01: return CompStrategy.SPARSE_MAP # If dense (> 50%), bitmap is good if sparse_ratio > 0.5: return CompStrategy.BITMAP # Otherwise delta ranges return CompStrategy.DELTA_RANGES if field_type == 'int': # Check if values are mostly small offsets (case mappings) if field_name.endswith('_map'): # These are relative offsets, usually small or zero if sparse_ratio < 0.05: return CompStrategy.SPARSE_MAP return CompStrategy.DELTA_RANGES # For numeric values (decimal_val, etc.) if len(unique_vals) < 20: return CompStrategy.DELTA_RANGES return CompStrategy.SPARSE_MAP return CompStrategy.DELTA_RANGES # ============================================================ # ENCODERS # ============================================================ def write_varint(buf: bytearray, val: int): while val >= 0x80: buf.append((val & 0x7F) | 0x80) val >>= 7 buf.append(val) def read_varint(data: bytes, offset: int) -> Tuple[int, int]: val = 0 shift = 0 while True: b = data[offset] offset += 1 val |= (b & 0x7F) << shift if not (b & 0x80): break shift += 7 return val, offset def encode_bitmap(rows: List[ProcessedRow], field: str) -> bytes: """Encode boolean field as bitmap.""" max_cp = max(r.cp for r in rows) size = (max_cp + 8) // 8 bitmap = bytearray(size) for r in rows: if r.fields.get(field, False): idx = r.cp >> 3 bitmap[idx] |= 1 << (r.cp & 7) return bytes(bitmap) def encode_delta_ranges(rows: List[ProcessedRow], field: str, field_type: str) -> bytes: """Encode as (start, length, value) runs with varints.""" buf = bytearray() current_start = None current_val = None for r in rows: val = r.fields.get(field) if field_type == 'bool': val = bool(val) elif field_type == 'int': val = int(val) if val else 0 else: val = val or "" if current_start is None: current_start = r.cp current_val = val elif val != current_val: # End previous range length = r.cp - current_start write_varint(buf, current_start) write_varint(buf, length) if field_type == 'bool': buf.append(1 if current_val else 0) elif field_type == 'int': # Zigzag encode for negative values v = current_val write_varint(buf, (v << 1) ^ (v >> 31)) current_start = r.cp current_val = val # Last range if current_start is not None: length = 0x10FFFF - current_start + 1 write_varint(buf, current_start) write_varint(buf, length) if field_type == 'bool': buf.append(1 if current_val else 0) elif field_type == 'int': v = current_val write_varint(buf, (v << 1) ^ (v >> 31)) return bytes(buf) def encode_sparse_map(rows: List[ProcessedRow], field: str, field_type: str) -> bytes: """Store only non-default values as (cp, value) pairs.""" buf = bytearray() for r in rows: val = r.fields.get(field) if field_type == 'bool': val = bool(val) if not val: continue write_varint(buf, r.cp) buf.append(1) elif field_type == 'int': val = int(val) if val else 0 if val == 0: continue write_varint(buf, r.cp) v = val write_varint(buf, (v << 1) ^ (v >> 31)) else: val = val or "" if not val: continue write_varint(buf, r.cp) # String index will be handled separately write_varint(buf, 0) # placeholder return bytes(buf) def encode_string_table(rows: List[ProcessedRow], field: str) -> Tuple[bytes, List[str]]: """Build string table and return (indices, string_pool).""" strings = [] str_to_idx = {} indices = [] for r in rows: val = r.fields.get(field, "") or "" if val not in str_to_idx: str_to_idx[val] = len(strings) strings.append(val) indices.append(str_to_idx[val]) # Encode indices as varints per codepoint buf = bytearray() for idx in indices: write_varint(buf, idx) return bytes(buf), strings # ============================================================ # MAIN GENERATOR # ============================================================ class XicuGenerator: def __init__(self, config: Dict, data_dir: str = './data', out_dir: str = './out'): self.config = config self.data_dir = Path(data_dir) self.out_dir = Path(out_dir) self.out_dir.mkdir(parents=True, exist_ok=True) self.data_dir.mkdir(parents=True, exist_ok=True) self.rows: List[ProcessedRow] = [] self.field_configs: Dict[str, FieldConfig] = {} self.strategies: Dict[str, CompStrategy] = {} self.string_tables: Dict[str, List[str]] = {} self.encoded_data: Dict[str, bytes] = {} def download_unicode_data(self): import urllib.request url = self.config['sources']['unicode'] path = self.data_dir / 'UnicodeData.txt' if not path.exists(): print(f"Downloading {url}...") urllib.request.urlretrieve(url, path) print("Done.") def parse_and_process(self): parser = UnicodeParser(str(self.data_dir / 'UnicodeData.txt')) raw_rows = parser.parse() print(f"Parsed {len(raw_rows)} codepoints") ucfg = self.config['unicode'] use_old_name = ucfg.get('useOldName', False) for row in raw_rows: self.rows.append(process_row(row, ucfg, use_old_name)) print(f"Processed {len(self.rows)} rows") def analyze_and_configure(self): ucfg = self.config['unicode'] for cfg_key, (out_name, ftype, cat) in FIELD_DEFS.items(): if not ucfg.get(cfg_key, False): continue strategy = analyze_field(self.rows, out_name, ftype) default = False if ftype == 'bool' else (0 if ftype == 'int' else "") self.field_configs[out_name] = FieldConfig( name=out_name, type=ftype, enabled=True, strategy=strategy, default_value=default ) self.strategies[out_name] = strategy print("Field strategies:") for name, fc in self.field_configs.items(): print(f" {name} ({fc.type}): {fc.strategy.value}") def encode_all(self): # First pass: encode non-string fields for name, fc in self.field_configs.items(): if fc.type == 'str': continue print(f"Encoding {name} with {fc.strategy.value}...") if fc.strategy == CompStrategy.BITMAP: self.encoded_data[name] = encode_bitmap(self.rows, name) elif fc.strategy == CompStrategy.DELTA_RANGES: self.encoded_data[name] = encode_delta_ranges(self.rows, name, fc.type) elif fc.strategy == CompStrategy.SPARSE_MAP: self.encoded_data[name] = encode_sparse_map(self.rows, name, fc.type) # Second pass: string fields for name, fc in self.field_configs.items(): if fc.type != 'str': continue print(f"Encoding string table for {name}...") indices, strings = encode_string_table(self.rows, name) self.encoded_data[name + '_indices'] = indices self.string_tables[name] = strings def write_binary(self): """Write main binary file with all non-string fields.""" path = self.out_dir / 'xicu.bin' buf = bytearray() # Header buf.extend(struct.pack(' str: lines = [ '#pragma once', '', '#include ', '#include ', '', 'namespace xicu {', '', 'class PropTable {', 'public:', ' PropTable() = default;', ' explicit PropTable(const char* path);', ' ~PropTable();', '', ' PropTable(const PropTable&) = delete;', ' PropTable& operator=(const PropTable&) = delete;', ' PropTable(PropTable&& other) noexcept;', ' PropTable& operator=(PropTable&& other) noexcept;', '', ' bool load(const char* path);', ' bool isLoaded() const { return data_ != nullptr; }', ] # Generate getter declarations for name, fc in self.field_configs.items(): if fc.type == 'bool': lines.append(f' bool get{name.capitalize()}(uint32_t cp) const;') elif fc.type == 'int': lines.append(f' int32_t get{name.capitalize()}(uint32_t cp) const;') elif fc.type == 'str': lines.append(f' std::string_view get{name.capitalize()}(uint32_t cp) const;') lines.extend([ '', 'private:', ' const uint8_t* data_ = nullptr;', ' size_t data_size_ = 0;', ' bool owns_data_ = false;', '', ' static uint32_t readVarint(const uint8_t*& ptr);', ' const uint8_t* findFieldData(const char* name, uint8_t& out_type, uint8_t& out_strat) const;', ' const uint8_t* findStringField(const char* name) const;', '', ' // Decoders', ' bool decodeBitmap(const uint8_t* data, uint32_t cp) const;', ' bool decodeDeltaRangesBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;', ' int32_t decodeDeltaRangesInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;', ' bool decodeSparseMapBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;', ' int32_t decodeSparseMapInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const;', '};', '', ]) # Inline convenience functions for name, fc in self.field_configs.items(): if fc.type == 'bool': lines.append(f'inline bool is{name.capitalize()}(uint32_t cp) {{') lines.append(f' static PropTable table("../out/xicu.bin");') lines.append(f' return table.get{name.capitalize()}(cp);') lines.append(f'}}') elif fc.type == 'int': lines.append(f'inline int32_t get{name.capitalize()}(uint32_t cp) {{') lines.append(f' static PropTable table("../out/xicu.bin");') lines.append(f' return table.get{name.capitalize()}(cp);') lines.append(f'}}') elif fc.type == 'str': lines.append(f'inline std::string_view get{name.capitalize()}(uint32_t cp) {{') lines.append(f' static PropTable table("../out/xicu.bin");') lines.append(f' return table.get{name.capitalize()}(cp);') lines.append(f'}}') lines.extend([ '', '} // namespace xicu', ]) return '\n'.join(lines) def _gen_impl(self) -> str: """Generate complete C++ implementation with decoders.""" lines = [ '#include "xicu.hpp"', '#include ', '#include ', '#include ', '#include ', '', 'namespace xicu {', '', 'static constexpr uint32_t MAGIC = 0x58494355;', 'static constexpr uint8_t VERSION = 2;', '', 'PropTable::PropTable(const char* path) { load(path); }', '', 'PropTable::~PropTable() {', ' if (owns_data_ && data_) { delete[] data_; data_ = nullptr; }', '}', '', 'PropTable::PropTable(PropTable&& other) noexcept', ' : data_(other.data_), data_size_(other.data_size_), owns_data_(other.owns_data_) {', ' other.data_ = nullptr; other.owns_data_ = false;', '}', '', 'PropTable& PropTable::operator=(PropTable&& other) noexcept {', ' if (this != &other) {', ' if (owns_data_ && data_) delete[] data_;', ' data_ = other.data_; data_size_ = other.data_size_; owns_data_ = other.owns_data_;', ' other.data_ = nullptr; other.owns_data_ = false;', ' }', ' return *this;', '}', '', 'uint32_t PropTable::readVarint(const uint8_t*& ptr) {', ' uint32_t val = 0; int shift = 0;', ' while (true) {', ' uint8_t b = *ptr++;', ' val |= (b & 0x7F) << shift;', ' if (!(b & 0x80)) break;', ' shift += 7;', ' }', ' return val;', '}', '', 'bool PropTable::load(const char* path) {', ' if (owns_data_ && data_) { delete[] data_; data_ = nullptr; owns_data_ = false; }', ' FILE* f = std::fopen(path, "rb");', ' if (!f) return false;', ' std::fseek(f, 0, SEEK_END);', ' long fsize = std::ftell(f);', ' std::fseek(f, 0, SEEK_SET);', ' if (fsize < 6) { std::fclose(f); return false; }', ' data_size_ = static_cast(fsize);', ' uint8_t* buf = new uint8_t[data_size_];', ' size_t read = std::fread(buf, 1, data_size_, f);', ' std::fclose(f);', ' if (read != data_size_) { delete[] buf; return false; }', ' data_ = buf; owns_data_ = true;', ' ', ' const uint8_t* ptr = data_;', ' uint32_t magic = *reinterpret_cast(ptr); ptr += 4;', ' if (magic != MAGIC) return false;', ' uint8_t version = *ptr++;', ' if (version != VERSION) return false;', ' return true;', '}', '', '// Find field data pointer (points to data length varint after name)', 'const uint8_t* PropTable::findFieldData(const char* name, uint8_t& out_type, uint8_t& out_strat) const {', ' if (!data_) return nullptr;', ' const uint8_t* ptr = data_ + 5; // magic + version', ' uint32_t num_fields = readVarint(ptr);', ' for (uint32_t i = 0; i < num_fields; ++i) {', ' uint8_t ftype = *ptr++;', ' uint8_t fstrat = *ptr++;', ' const char* fname = reinterpret_cast(ptr);', ' size_t name_len = std::strlen(fname);', ' ptr += name_len + 1;', ' if (std::strcmp(fname, name) == 0) {', ' out_type = ftype;', ' out_strat = fstrat;', ' return ptr; // points to data length varint', ' }', ' // Skip data', ' uint32_t data_len = readVarint(ptr);', ' ptr += data_len;', ' }', ' return nullptr;', '}', '', '// Find string field (returns pointer to indices data length)', 'const uint8_t* PropTable::findStringField(const char* name) const {', ' if (!data_) return nullptr;', ' const uint8_t* ptr = data_ + 5;', ' uint32_t num_fields = readVarint(ptr);', ' // Skip non-string fields (with inline data)', ' for (uint32_t i = 0; i < num_fields; ++i) {', ' ptr++; ptr++; // type, strat', ' while (*ptr++) {} // skip name', ' uint32_t data_len = readVarint(ptr);', ' ptr += data_len; // skip data', ' }', ' uint32_t num_str_fields = readVarint(ptr);', ' for (uint32_t i = 0; i < num_str_fields; ++i) {', ' const char* fname = reinterpret_cast(ptr);', ' size_t name_len = std::strlen(fname);', ' ptr += name_len + 1;', ' if (std::strcmp(fname, name) == 0) {', ' return ptr; // points to indices data length', ' }', ' uint32_t data_len = readVarint(ptr);', ' ptr += data_len; // skip indices data', ' }', ' return nullptr;', '}', '', '// Decode bitmap: 1 bit per codepoint', 'bool PropTable::decodeBitmap(const uint8_t* data, uint32_t cp) const {', ' size_t byte_idx = cp >> 3;', ' uint8_t bit = cp & 7;', ' return (data[byte_idx] & (1 << bit)) != 0;', '}', '', '// Decode delta ranges: (start, length, value) varint-encoded', 'bool PropTable::decodeDeltaRangesBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {', ' const uint8_t* ptr = data;', ' while (ptr < data_end) {', ' uint32_t start = readVarint(ptr);', ' uint32_t length = readVarint(ptr);', ' uint8_t val = *ptr++;', ' uint32_t end = start + length - 1;', ' if (cp >= start && cp <= end) return val != 0;', ' if (cp < start) return false;', ' }', ' return false;', '}', '', 'int32_t PropTable::decodeDeltaRangesInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {', ' const uint8_t* ptr = data;', ' while (ptr < data_end) {', ' uint32_t start = readVarint(ptr);', ' uint32_t length = readVarint(ptr);', ' uint32_t zigzag = readVarint(ptr);', ' int32_t val = static_cast((zigzag >> 1) ^ -(zigzag & 1));', ' uint32_t end = start + length - 1;', ' if (cp >= start && cp <= end) return val;', ' if (cp < start) return 0;', ' }', ' return 0;', '}', '', '// Decode sparse map: (cp, value) pairs', 'bool PropTable::decodeSparseMapBool(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {', ' const uint8_t* ptr = data;', ' while (ptr < data_end) {', ' uint32_t c = readVarint(ptr);', ' if (ptr >= data_end) break;', ' uint8_t val = *ptr++;', ' if (c == cp) return val != 0;', ' if (c > cp) return false;', ' }', ' return false;', '}', '', 'int32_t PropTable::decodeSparseMapInt(const uint8_t* data, const uint8_t* data_end, uint32_t cp) const {', ' const uint8_t* ptr = data;', ' while (ptr < data_end) {', ' uint32_t c = readVarint(ptr);', ' if (ptr >= data_end) break;', ' uint32_t zigzag = readVarint(ptr);', ' int32_t val = static_cast((zigzag >> 1) ^ -(zigzag & 1));', ' if (c == cp) return val;', ' if (c > cp) return 0;', ' }', ' return 0;', '}', '', ] # Generate getters for each field for name, fc in self.field_configs.items(): if fc.type == 'bool': strat = fc.strategy.value if strat == 'bitmap': decoder = 'decodeBitmap' call = 'decodeBitmap(data, cp)' elif strat == 'delta_ranges': decoder = 'decodeDeltaRangesBool' call = 'decodeDeltaRangesBool(data, data_end, cp)' elif strat == 'sparse_map': decoder = 'decodeSparseMapBool' call = 'decodeSparseMapBool(data, data_end, cp)' else: decoder = 'decodeDeltaRangesBool' call = 'decodeDeltaRangesBool(data, data_end, cp)' lines.extend([ f'bool PropTable::get{name.capitalize()}(uint32_t cp) const {{', f' uint8_t ftype, fstrat;', f' const uint8_t* data = findFieldData("{name}", ftype, fstrat);', f' if (!data) return false;', f' uint32_t data_len = readVarint(data);', f' const uint8_t* data_end = data + data_len;', f' return {call};', f'}}', '', ]) elif fc.type == 'int': strat = fc.strategy.value if strat == 'delta_ranges': decoder = 'decodeDeltaRangesInt' elif strat == 'sparse_map': decoder = 'decodeSparseMapInt' else: decoder = 'decodeDeltaRangesInt' lines.extend([ f'int32_t PropTable::get{name.capitalize()}(uint32_t cp) const {{', f' uint8_t ftype, fstrat;', f' const uint8_t* data = findFieldData("{name}", ftype, fstrat);', f' if (!data) return 0;', f' uint32_t data_len = readVarint(data);', f' const uint8_t* data_end = data + data_len;', f' return {decoder}(data, data_end, cp);', f'}}', '', ]) elif fc.type == 'str': lines.extend([ f'std::string_view PropTable::get{name.capitalize()}(uint32_t cp) const {{', f' // String fields use separate .str file', f' const uint8_t* indices_ptr = findStringField("{name}");', f' if (!indices_ptr) return "";', f' uint32_t indices_len = readVarint(indices_ptr);', f' // Find index for this codepoint (linear scan for now)', f' // TODO: optimize with binary search if needed', f' // For now, we need the string pool loaded separately', f' return ""; // Requires string pool file', f'}}', '', ]) lines.append('} // namespace xicu') return '\n'.join(lines) def run(self): self.download_unicode_data() self.parse_and_process() self.analyze_and_configure() self.encode_all() self.write_binary() self.generate_cpp() print("\nGeneration complete!") def main(): gen = XicuGenerator(CONFIG) gen.run() if __name__ == '__main__': main()