"""Generates Glass's Unicode property table from the Unicode Character Database vendored in 3rdParty\\Unicode.
Run by hand when the vendored UCD changes; the output is checked in, so builds never run this.
"""
import argparse
import os
import subprocess
import sys
import unittest
REPO_ROOT = os.path.abspath(os.path.join(os.path.abspath(__file__), "..", ".."))
UNICODE_VERSION = "18.0.0"
ARCHIVE = os.path.join(REPO_ROOT, "3rdParty", "Unicode", "UCD-{0}.7z".format(UNICODE_VERSION))
SEVEN_ZIP = os.path.join(REPO_ROOT, "3rdParty", "7-Zip", "7zr.exe")
UNPACK_DIRECTORY = os.path.join(REPO_ROOT, "Junk", "UnicodeProperties", "UCD-{0}".format(UNICODE_VERSION))
OUTPUT = os.path.join(REPO_ROOT, "Modules", "Glass", "Glass", "Typesetting", "UnicodeProperties.Generated.cs")
CODE_POINT_COUNT = 0x110000
# The resolved line break classes, in enum order, with UAX #14 Table 1's names. QU is split three ways because
# LB15a and LB15b single out quotation marks that are also initial (Pi) or final (Pf) punctuation.
LINE_BREAK_CLASSES = [
("AK", "Aksara"),
("AL", "Alphabetic"),
("AP", "Aksara Pre-Base"),
("AS", "Aksara Start"),
("B2", "Break Opportunity Before and After"),
("BA", "Break After"),
("BB", "Break Before"),
("BK", "Mandatory Break"),
("CB", "Contingent Break Opportunity"),
("CL", "Close Punctuation"),
("CM", "Combining Mark"),
("CP", "Close Parenthesis"),
("CR", "Carriage Return"),
("EB", "Emoji Base"),
("EM", "Emoji Modifier"),
("EX", "Exclamation/Interrogation"),
("GL", "Non-breaking (\"Glue\")"),
("H2", "Hangul LV Syllable"),
("H3", "Hangul LVT Syllable"),
("HH", "Unambiguous Hyphen"),
("HL", "Hebrew Letter"),
("HY", "Hyphen"),
("ID", "Ideographic"),
("IN", "Inseparable"),
("IS", "Infix Numeric Separator"),
("JL", "Hangul L Jamo"),
("JT", "Hangul T Jamo"),
("JV", "Hangul V Jamo"),
("LF", "Line Feed"),
("NL", "Next Line"),
("NS", "Nonstarter"),
("NU", "Numeric"),
("OP", "Open Punctuation"),
("PO", "Postfix Numeric"),
("PR", "Prefix Numeric"),
("QU", "Quotation, other than initial or final punctuation"),
("QU_Pi", "Quotation that is also initial punctuation (General_Category Pi)"),
("QU_Pf", "Quotation that is also final punctuation (General_Category Pf)"),
("RI", "Regional Indicator"),
("SP", "Space"),
("SY", "Symbols Allowing Break After"),
("VF", "Virama Final"),
("VI", "Virama"),
("WJ", "Word Joiner"),
("ZW", "Zero Width Space"),
("ZWJ", "Zero Width Joiner"),
("DottedCircle", "Not a Unicode class: U+25CC, which Unicode classes as AL but LB28a names on its own"),
("SoftHyphen", "Not a Unicode class: U+00AD, which Unicode classes as HH but which shows only where a line breaks"),
]
CLASS_INDEX = {name: index for index, (name, _) in enumerate(LINE_BREAK_CLASSES)}
# Code points a rule names individually, so they get a class of their own instead of their Unicode one.
CODE_POINT_CLASSES = {0x25CC: "DottedCircle", 0x00AD: "SoftHyphen"}
# Classes LB1 resolves away when nothing more specific is known. SA depends on the General_Category too.
LB1_RESOLUTIONS = {"AI": "AL", "SG": "AL", "XX": "AL", "CJ": "NS"}
CLASS_BITS = 6
EAST_ASIAN_FLAG = 0x40
UNASSIGNED_EXTENDED_PICTOGRAPHIC_FLAG = 0x80
# East_Asian_Width values UAX #14 calls $EastAsian.
EAST_ASIAN_WIDTHS = frozenset(["F", "W", "H"])
CANDIDATE_CHUNK_SHIFTS = range(4, 11)
LINE_WIDTH = 120
def parse_line(line):
"""Splits a UCD data line into (first, last, fields), or returns None for a blank or comment line.
Comments are stripped before splitting, since a data line can run straight into one (e.g. "200D;ZWJ# ...").
"""
data = line.split("#", 1)[0].strip()
if not data:
return None
fields = [field.strip() for field in data.split(";")]
code_points = fields[0].split("..")
first = int(code_points[0], 16)
last = int(code_points[-1], 16)
return first, last, fields[1:]
def read_property(path, default):
"""Reads a single-valued property file into a list with one entry per code point.
Args:
path: The UCD file.
default: The value of code points the file doesn't list, which each of these files states in its
"@missing: 0000..10FFFF" line.
Returns:
A list of CODE_POINT_COUNT values.
"""
values = [default] * CODE_POINT_COUNT
with open(path, encoding="utf-8") as ucd_file:
for line in ucd_file:
parsed = parse_line(line)
if parsed is not None:
first, last, fields = parsed
values[first:last + 1] = [fields[0]] * (last - first + 1)
return values
def read_extended_pictographic(path):
"""Returns the set of code points with Extended_Pictographic=Yes."""
pictographic = set()
with open(path, encoding="utf-8") as ucd_file:
for line in ucd_file:
parsed = parse_line(line)
if parsed is not None and parsed[2][0] == "Extended_Pictographic":
pictographic.update(range(parsed[0], parsed[1] + 1))
return pictographic
def read_binary_property(path, name):
"""Returns the code points `path` lists with property `name`, as merged inclusive (first, last) ranges."""
ranges = []
with open(path, encoding="utf-8") as ucd_file:
for line in ucd_file:
parsed = parse_line(line)
if parsed is not None and parsed[2][0] == name:
ranges.append((parsed[0], parsed[1]))
return merge_ranges(ranges)
def merge_ranges(ranges):
"""Sorts inclusive (first, last) ranges and joins the ones that touch or overlap."""
merged = []
for first, last in sorted(ranges):
if merged and first <= merged[-1][1] + 1:
merged[-1] = (merged[-1][0], max(merged[-1][1], last))
else:
merged.append((first, last))
return merged
def resolve_class(line_break, general_category):
"""Resolves a code point's line break class per LB1's defaults, and splits QU by General_Category."""
if line_break == "SA":
return "CM" if general_category in ("Mn", "Mc") else "AL"
if line_break == "QU" and general_category in ("Pi", "Pf"):
return "QU_" + general_category
return LB1_RESOLUTIONS.get(line_break, line_break)
def read_categories(directory):
"""Returns each code point's General_Category."""
return read_property(os.path.join(directory, "DerivedGeneralCategory.txt"), "Cn")
def build_properties(directory, categories):
"""Returns one packed property byte per code point, from the UCD files in `directory`."""
line_breaks = read_property(os.path.join(directory, "LineBreak.txt"), "XX")
widths = read_property(os.path.join(directory, "EastAsianWidth.txt"), "N")
pictographic = read_extended_pictographic(os.path.join(directory, "emoji-data.txt"))
unknown = sorted(set(resolve_class(line_break, category)
for line_break, category in zip(line_breaks, categories)) - set(CLASS_INDEX))
if unknown:
raise ValueError("Line break classes this script doesn't know: {0}. Add them to "
"LINE_BREAK_CLASSES.".format(", ".join(unknown)))
properties = bytearray(CODE_POINT_COUNT)
for code_point in range(CODE_POINT_COUNT):
line_break_class = CODE_POINT_CLASSES.get(code_point) or resolve_class(line_breaks[code_point],
categories[code_point])
packed = CLASS_INDEX[line_break_class]
if widths[code_point] in EAST_ASIAN_WIDTHS:
packed |= EAST_ASIAN_FLAG
if code_point in pictographic and categories[code_point] == "Cn":
packed |= UNASSIGNED_EXTENDED_PICTOGRAPHIC_FLAG
properties[code_point] = packed
return properties
def build_two_stage_table(values, shift):
"""Splits `values` into chunks of 2**shift, storing each distinct chunk once.
Returns:
A (major, minor) tuple where values[i] == minor[major[i >> shift] + (i & mask)]. Each major entry is
already the offset of its chunk in minor, so a lookup needs no multiply.
"""
chunk_size = 1 << shift
offsets = {}
major = []
minor = bytearray()
for start in range(0, len(values), chunk_size):
chunk = bytes(values[start:start + chunk_size])
if chunk not in offsets:
offsets[chunk] = len(minor)
minor.extend(chunk)
major.append(offsets[chunk])
return major, minor
def table_size(major, minor):
"""Bytes the table takes, with major stored as ushorts."""
return 2 * len(major) + len(minor)
def smallest_table(values):
"""Tries each candidate chunk size and returns (shift, major, minor) for the smallest table that fits."""
best = None
for shift in CANDIDATE_CHUNK_SHIFTS:
major, minor = build_two_stage_table(values, shift)
# Major holds ushort offsets, so minor can't outgrow what one can address.
if len(minor) > 0x10000:
continue
if best is None or table_size(major, minor) < table_size(best[1], best[2]):
best = (shift, major, minor)
return best
def format_array(numbers, indent):
"""Renders numbers as the body of a C# array initializer, filling each line out to LINE_WIDTH."""
lines = []
line = indent
for number in numbers:
item = "{0},".format(number)
if len(line) + len(item) + 1 > LINE_WIDTH:
lines.append(line.rstrip())
line = indent
line += item + " "
lines.append(line.rstrip())
return "\n".join(lines)
def format_code_point(code_point):
"""Renders a code point as a C# hex literal, four digits at least, like U+ notation."""
return "0x{0:04X}".format(code_point)
def render(shift, major, minor, default_ignorables, space_separators):
"""Returns the generated C# source."""
enum_members = []
for index, (name, description) in enumerate(LINE_BREAK_CLASSES):
enum_members.append(" /// \n /// {0}.\n /// \n {1} = {2},"
.format(description, name, index))
return """//
// Generated by Scripts\\GenerateUnicodeProperties.py from the Unicode Character Database {version}, vendored in
// 3rdParty\\Unicode\\UCD-{version}.7z. Edit the script and regenerate rather than editing this file.
//
// Derived from Unicode data, (c) Unicode, Inc., used under the Unicode License v3; see 3rdParty\\Unicode\\LICENSE.txt.
//
namespace Glass.Typesetting
{{
///
/// UAX #14 line breaking classes, as resolved by LB1's defaults: AI, SG and XX become AL; SA becomes CM or AL
/// by General_Category; CJ becomes NS. QU splits by General_Category for LB15a and LB15b. U+25CC gets a class
/// of its own for LB28a, and U+00AD one so a break after it is known to show a hyphen.
///
public enum LineBreakClass : byte
{{
{members}
}}
public static partial class UnicodeProperties
{{
public const string UnicodeVersion = "{version}";
public const int ChunkShift = {shift};
public const int ChunkMask = (1 << ChunkShift) - 1;
///
/// For each chunk of code points, where that chunk's bytes start in .
///
private static readonly ushort[] Major =
{{
{major}
}};
///
/// Each distinct chunk of packed , stored once however many chunks share it.
///
private static readonly byte[] Minor =
{{
{minor}
}};
///
/// Default_Ignorable_Code_Point, as inclusive first and last code points of each range, in order.
///
public static readonly int[] DefaultIgnorableRanges =
{{
{default_ignorables}
}};
///
/// Every code point with General_Category Zs (space separator), in order.
///
public static readonly int[] SpaceSeparators =
{{
{space_separators}
}};
}}
}}
""".format(version=UNICODE_VERSION, members="\n".join(enum_members), shift=shift,
major=format_array(major, " " * 12), minor=format_array(minor, " " * 12),
default_ignorables=format_array([format_code_point(bound) for first_last in default_ignorables
for bound in first_last], " " * 12),
space_separators=format_array([format_code_point(code_point) for code_point in space_separators],
" " * 12))
def unpack(archive, directory):
"""Unpacks the vendored UCD archive into Junk."""
subprocess.run([SEVEN_ZIP, "x", archive, "-o" + directory, "-y"], check=True, stdout=subprocess.DEVNULL)
def write_source(path, text):
"""Writes C# source the way the tree keeps it: UTF-8 with a BOM, CRLF line endings."""
with open(path, "w", encoding="utf-8-sig", newline="\r\n") as source_file:
source_file.write(text)
def generate():
"""Regenerates the table. Returns a process exit code."""
unpack(ARCHIVE, UNPACK_DIRECTORY)
categories = read_categories(UNPACK_DIRECTORY)
values = build_properties(UNPACK_DIRECTORY, categories)
shift, major, minor = smallest_table(values)
default_ignorables = read_binary_property(os.path.join(UNPACK_DIRECTORY, "DerivedCoreProperties.txt"),
"Default_Ignorable_Code_Point")
space_separators = [code_point for code_point, category in enumerate(categories) if category == "Zs"]
write_source(OUTPUT, render(shift, major, minor, default_ignorables, space_separators))
print("Wrote {0}: chunks of {1}, {2} distinct, {3:,} bytes of table.".format(
OUTPUT, 1 << shift, len(minor) >> shift, table_size(major, minor)))
return 0
class GeneratorTests(unittest.TestCase):
"""Covers the packing and parsing, which the C# tests can only check through the finished table."""
def test_two_stage_table_reads_back_every_value(self):
values = bytearray((index * 7) % 5 if index < 300 else 1 for index in range(1000))
major, minor = build_two_stage_table(values, 4)
self.assertEqual(list(values), [minor[major[index >> 4] + (index & 15)] for index in range(len(values))])
def test_identical_chunks_are_stored_once(self):
major, minor = build_two_stage_table(bytearray(64), 4)
self.assertEqual((4, 16), (len(major), len(minor)))
def test_parse_line_handles_a_comment_with_no_space_before_it(self):
self.assertEqual((0x200D, 0x200D, ["ZWJ"]), parse_line("200D;ZWJ# Cf ZERO WIDTH JOINER"))
def test_parse_line_reads_a_range(self):
self.assertEqual((0x3400, 0x4DBF, ["ID"]), parse_line("3400..4DBF ; ID # Lo [6592] CJK..."))
def test_parse_line_skips_comments(self):
self.assertIsNone(parse_line("# @missing: 0000..10FFFF; XX"))
def test_sa_resolves_by_general_category(self):
self.assertEqual(("CM", "AL"), (resolve_class("SA", "Mn"), resolve_class("SA", "Lo")))
def test_quotation_splits_by_general_category(self):
self.assertEqual(("QU_Pi", "QU_Pf", "QU"),
(resolve_class("QU", "Pi"), resolve_class("QU", "Pf"), resolve_class("QU", "Po")))
def test_merge_ranges_joins_touching_and_overlapping_ranges(self):
self.assertEqual([(1, 6), (8, 9)], merge_ranges([(8, 9), (4, 6), (1, 3), (2, 2)]))
def test_format_code_point_pads_to_four_digits(self):
self.assertEqual(("0x00AD", "0xE0FFF"), (format_code_point(0xAD), format_code_point(0xE0FFF)))
def test_every_code_point_override_names_a_class(self):
self.assertTrue(set(CODE_POINT_CLASSES.values()) <= set(CLASS_INDEX))
def test_every_class_fits_below_the_flags(self):
self.assertLessEqual(len(LINE_BREAK_CLASSES), 1 << CLASS_BITS)
def main():
"""Parses the command line and runs the requested job."""
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--self-test", action="store_true", help="Run this script's own tests and exit.")
arguments = parser.parse_args()
if arguments.self_test:
suite = unittest.TestLoader().loadTestsFromTestCase(GeneratorTests)
return 0 if unittest.TextTestRunner(verbosity=2).run(suite).wasSuccessful() else 1
return generate()
if __name__ == "__main__":
sys.exit(main())