lots of improvements standby for the results
This commit is contained in:
@@ -0,0 +1,117 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Generate binary tables for:
|
||||
a) is char X alphanumeric? (letter or decimal digit)
|
||||
b) is char whitespace?
|
||||
"""
|
||||
|
||||
import os
|
||||
import struct
|
||||
import urllib.request
|
||||
|
||||
DATA_DIR = './data'
|
||||
OUT_DIR = './out'
|
||||
|
||||
os.makedirs(DATA_DIR, exist_ok=True)
|
||||
os.makedirs(OUT_DIR, exist_ok=True)
|
||||
|
||||
UNICODE_DATA_URL = 'https://www.unicode.org/Public/UCD/latest/ucd/UnicodeData.txt'
|
||||
UNICODE_DATA_PATH = os.path.join(DATA_DIR, 'UnicodeData.txt')
|
||||
|
||||
if not os.path.exists(UNICODE_DATA_PATH):
|
||||
print(f"Downloading {UNICODE_DATA_URL}...")
|
||||
urllib.request.urlretrieve(UNICODE_DATA_URL, UNICODE_DATA_PATH)
|
||||
print("Downloaded")
|
||||
|
||||
def parse_unicode_data():
|
||||
"""Parse UnicodeData.txt and yield (codepoint, category, name)"""
|
||||
with open(UNICODE_DATA_PATH, 'r', encoding='utf-8') as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
parts = line.split(';')
|
||||
if len(parts) < 3:
|
||||
continue
|
||||
codepoint = int(parts[0], 16)
|
||||
name = parts[1]
|
||||
category = parts[2]
|
||||
|
||||
# Handle ranges (First/Last)
|
||||
if name.endswith(', First>'):
|
||||
range_start = (codepoint, category, name)
|
||||
continue
|
||||
elif name.endswith(', Last>') and 'range_start' in locals():
|
||||
start_cp, start_cat, start_name = range_start
|
||||
base_name = start_name.replace(', First>', '').replace('<', '')
|
||||
for cp in range(start_cp, codepoint + 1):
|
||||
yield (cp, start_cat, f"<{base_name}>")
|
||||
del range_start
|
||||
continue
|
||||
|
||||
yield (codepoint, category, name)
|
||||
|
||||
def is_alphanumeric(category: str, codepoint: int) -> bool:
|
||||
"""Check if category indicates alphanumeric (letter or decimal digit)"""
|
||||
return category.startswith('L') or category == 'Nd'
|
||||
|
||||
def is_whitespace(category: str, codepoint: int) -> bool:
|
||||
"""Check if character is whitespace"""
|
||||
if category in ('Zs', 'Zl', 'Zp'):
|
||||
return True
|
||||
# Common control whitespaces: \t\n\r\f\v and others
|
||||
return codepoint in (0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x1C, 0x1D, 0x1E, 0x1F, 0x85, 0xA0)
|
||||
|
||||
def build_bitmap(check_func, max_codepoint=0x10FFFF):
|
||||
"""Build a bitmap for the given check function"""
|
||||
# We'll use a byte array where each bit represents a codepoint
|
||||
size = (max_codepoint + 8) // 8
|
||||
bitmap = bytearray(size)
|
||||
|
||||
for cp, cat, _ in parse_unicode_data():
|
||||
if check_func(cat, cp):
|
||||
byte_idx = cp // 8
|
||||
bit_idx = cp % 8
|
||||
bitmap[byte_idx] |= (1 << bit_idx)
|
||||
|
||||
return bitmap
|
||||
|
||||
def write_binary_file(path: str, bitmap: bytearray, max_codepoint: int):
|
||||
"""Write binary file with header: magic, version, max_codepoint, data"""
|
||||
with open(path, 'wb') as f:
|
||||
# Magic: 'XICU' (0x58494355)
|
||||
f.write(struct.pack('<I', 0x58494355))
|
||||
# Version: 1
|
||||
f.write(struct.pack('<B', 1))
|
||||
# Max codepoint (4 bytes)
|
||||
f.write(struct.pack('<I', max_codepoint))
|
||||
# Data
|
||||
f.write(bitmap)
|
||||
|
||||
def main():
|
||||
MAX_CP = 0x10FFFF
|
||||
|
||||
print("Building alphanumeric bitmap...")
|
||||
alnum_bitmap = build_bitmap(is_alphanumeric, MAX_CP)
|
||||
|
||||
print("Building whitespace bitmap...")
|
||||
ws_bitmap = build_bitmap(is_whitespace, MAX_CP)
|
||||
|
||||
alnum_path = os.path.join(OUT_DIR, 'alphanumeric.bin')
|
||||
ws_path = os.path.join(OUT_DIR, 'whitespace.bin')
|
||||
|
||||
print(f"Writing {alnum_path} ({len(alnum_bitmap)} bytes)...")
|
||||
write_binary_file(alnum_path, alnum_bitmap, MAX_CP)
|
||||
|
||||
print(f"Writing {ws_path} ({len(ws_bitmap)} bytes)...")
|
||||
write_binary_file(ws_path, ws_bitmap, MAX_CP)
|
||||
|
||||
# Print stats
|
||||
alnum_count = sum(bin(b).count('1') for b in alnum_bitmap)
|
||||
ws_count = sum(bin(b).count('1') for b in ws_bitmap)
|
||||
print(f"Alphanumeric codepoints: {alnum_count}")
|
||||
print(f"Whitespace codepoints: {ws_count}")
|
||||
print("Done!")
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in New Issue
Block a user