117 lines
3.9 KiB
Python
117 lines
3.9 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Generate binary tables for:
|
|
a) is char X alphanumeric? (letter or decimal digit)
|
|
b) is char whitespace?
|
|
"""
|
|
|
|
import os
|
|
import struct
|
|
import urllib.request
|
|
|
|
DATA_DIR = './data'
|
|
OUT_DIR = './out'
|
|
|
|
os.makedirs(DATA_DIR, exist_ok=True)
|
|
os.makedirs(OUT_DIR, exist_ok=True)
|
|
|
|
UNICODE_DATA_URL = 'https://www.unicode.org/Public/UCD/latest/ucd/UnicodeData.txt'
|
|
UNICODE_DATA_PATH = os.path.join(DATA_DIR, 'UnicodeData.txt')
|
|
|
|
if not os.path.exists(UNICODE_DATA_PATH):
|
|
print(f"Downloading {UNICODE_DATA_URL}...")
|
|
urllib.request.urlretrieve(UNICODE_DATA_URL, UNICODE_DATA_PATH)
|
|
print("Downloaded")
|
|
|
|
def parse_unicode_data():
|
|
"""Parse UnicodeData.txt and yield (codepoint, category, name)"""
|
|
with open(UNICODE_DATA_PATH, 'r', encoding='utf-8') as f:
|
|
for line in f:
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
parts = line.split(';')
|
|
if len(parts) < 3:
|
|
continue
|
|
codepoint = int(parts[0], 16)
|
|
name = parts[1]
|
|
category = parts[2]
|
|
|
|
# Handle ranges (First/Last)
|
|
if name.endswith(', First>'):
|
|
range_start = (codepoint, category, name)
|
|
continue
|
|
elif name.endswith(', Last>') and 'range_start' in locals():
|
|
start_cp, start_cat, start_name = range_start
|
|
base_name = start_name.replace(', First>', '').replace('<', '')
|
|
for cp in range(start_cp, codepoint + 1):
|
|
yield (cp, start_cat, f"<{base_name}>")
|
|
del range_start
|
|
continue
|
|
|
|
yield (codepoint, category, name)
|
|
|
|
def is_alphanumeric(category: str, codepoint: int) -> bool:
|
|
"""Check if category indicates alphanumeric (letter or decimal digit)"""
|
|
return category.startswith('L') or category == 'Nd'
|
|
|
|
def is_whitespace(category: str, codepoint: int) -> bool:
|
|
"""Check if character is whitespace"""
|
|
if category in ('Zs', 'Zl', 'Zp'):
|
|
return True
|
|
# Common control whitespaces: \t\n\r\f\v and others
|
|
return codepoint in (0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x1C, 0x1D, 0x1E, 0x1F, 0x85, 0xA0)
|
|
|
|
def build_bitmap(check_func, max_codepoint=0x10FFFF):
|
|
"""Build a bitmap for the given check function"""
|
|
# We'll use a byte array where each bit represents a codepoint
|
|
size = (max_codepoint + 8) // 8
|
|
bitmap = bytearray(size)
|
|
|
|
for cp, cat, _ in parse_unicode_data():
|
|
if check_func(cat, cp):
|
|
byte_idx = cp // 8
|
|
bit_idx = cp % 8
|
|
bitmap[byte_idx] |= (1 << bit_idx)
|
|
|
|
return bitmap
|
|
|
|
def write_binary_file(path: str, bitmap: bytearray, max_codepoint: int):
|
|
"""Write binary file with header: magic, version, max_codepoint, data"""
|
|
with open(path, 'wb') as f:
|
|
# Magic: 'XICU' (0x58494355)
|
|
f.write(struct.pack('<I', 0x58494355))
|
|
# Version: 1
|
|
f.write(struct.pack('<B', 1))
|
|
# Max codepoint (4 bytes)
|
|
f.write(struct.pack('<I', max_codepoint))
|
|
# Data
|
|
f.write(bitmap)
|
|
|
|
def main():
|
|
MAX_CP = 0x10FFFF
|
|
|
|
print("Building alphanumeric bitmap...")
|
|
alnum_bitmap = build_bitmap(is_alphanumeric, MAX_CP)
|
|
|
|
print("Building whitespace bitmap...")
|
|
ws_bitmap = build_bitmap(is_whitespace, MAX_CP)
|
|
|
|
alnum_path = os.path.join(OUT_DIR, 'alphanumeric.bin')
|
|
ws_path = os.path.join(OUT_DIR, 'whitespace.bin')
|
|
|
|
print(f"Writing {alnum_path} ({len(alnum_bitmap)} bytes)...")
|
|
write_binary_file(alnum_path, alnum_bitmap, MAX_CP)
|
|
|
|
print(f"Writing {ws_path} ({len(ws_bitmap)} bytes)...")
|
|
write_binary_file(ws_path, ws_bitmap, MAX_CP)
|
|
|
|
# Print stats
|
|
alnum_count = sum(bin(b).count('1') for b in alnum_bitmap)
|
|
ws_count = sum(bin(b).count('1') for b in ws_bitmap)
|
|
print(f"Alphanumeric codepoints: {alnum_count}")
|
|
print(f"Whitespace codepoints: {ws_count}")
|
|
print("Done!")
|
|
|
|
if __name__ == '__main__':
|
|
main() |