using System; using System.Collections.Generic; using System.Globalization; using System.IO; using System.Linq; using System.Text; using System.Text.RegularExpressions; using Azimuth; using UnitTestSharp; namespace Glass.UnitTests { public partial class TextLayoutTests { /// /// WordWrap against the Unicode Consortium's UAX #14 conformance file, which the build unpacks next to this /// binary from 3rdParty\Unicode. /// public class WordWrapUax14Conformance : TestFixture { public static readonly string TestFilePath = Path.Combine(AppDomain.CurrentDomain.BaseDirectory, "Unicode", "LineBreakTest.txt"); public const char NoBreakMark = (char)0x00D7; public const char BreakMark = (char)0x00F7; public static readonly Regex RulePattern = new Regex(@"\[(\d+\.\d+)\]"); /// /// Glyphs are 2 wide, except '-' and the 'z' that pads out a lead-in, which are 1. A hyphen narrower than /// any glyph is what makes a break at a soft hyphen observable: were they equal, the room reserved for /// the hyphen would fit the next glyph instead, and the line rightly wouldn't break. /// public class ProbeFont : WordWrapTruncation.TestFont { public override GlyphMetrics MetricsFor(char glyph) => new GlyphMetrics { Width = glyph == '-' || glyph == 'z' ? 1 : 2 }; } public static readonly IFontMetrics Font = new ProbeFont(); public struct Boundary { /// /// UTF-16 offset into the case's text. /// public int Index; public bool Breaks; /// /// The file's number for the rule that decides this boundary, e.g. "13.02". /// public string Rule; } /// /// One line of the file: its text, and the verdict on every boundary strictly inside it. /// public class Case { public string Text; public List Boundaries = new List(); } public static Case Parse(string line) { int commentStart = line.IndexOf('#'); string[] tokens = line.Substring(0, commentStart) .Split((char[])null, StringSplitOptions.RemoveEmptyEntries); MatchCollection rules = RulePattern.Matches(line, commentStart); // Tokens alternate verdict, code point, verdict, ..., verdict. The first and last verdicts are the // start and end of text, which WordWrap has no say in. var text = new StringBuilder(); var returnMe = new Case(); for (int i = 0; i < tokens.Length; i += 2) { if (i > 0 && i < tokens.Length - 1) { returnMe.Boundaries.Add(new Boundary { Index = text.Length, Breaks = tokens[i][0] == BreakMark, Rule = rules[i / 2].Groups[1].Value, }); } if (i + 1 < tokens.Length) { text.Append(char.ConvertFromUtf32(int.Parse(tokens[i + 1], NumberStyles.HexNumber))); } } returnMe.Text = text.ToString(); return returnMe; } /// /// Characters of the hard line break classes BK, CR, LF and NL. /// public static bool IsHardBreakCharacter(char character) { switch (character) { case '\n': case '\r': case (char)0x000B: case (char)0x000C: case (char)0x0085: case (char)0x2028: case (char)0x2029: return true; default: return false; } } /// /// Whether the boundary at `index` follows a hard line break, which UAX #14 requires to break no matter /// how much room is left. CR followed by LF breaks once, after the LF. /// public static bool FollowsHardBreak(string text, int index) { char previous = text[index - 1]; return previous == '\r' ? text[index] != '\n' : IsHardBreakCharacter(previous); } /// /// Whitespace WordWrap hangs past the margin rather than wrapping. A break before it can never be seen, /// since there's always room for it, so those boundaries go untested. /// public static bool HangsAtLineEnd(char character) { switch (character) { case '\n': case '\r': case (char)0x000B: case (char)0x000C: case (char)0x0085: case (char)0x2028: case (char)0x2029: case (char)0x00A0: case (char)0x2007: case (char)0x202F: return false; default: return char.IsWhiteSpace(character); } } /// /// Whether nothing but soft hyphens follows `index` before the line ends. They take no room, so a break /// before them can never be seen either. /// public static bool OnlySoftHyphensFollow(string text, int index) { for (int i = index; i < text.Length && !IsHardBreakCharacter(text[i]); ++i) { if (text[i] != WordWrapUax14.SoftHyphen) { return false; } } return true; } public static List LineStarts(string text, Scalar maxWidth) { var starts = new List(); WordWrap(Font, text, maxWidth, linePlacementCallback: (in StringView line, Scalar lineWidth, Vector position) => starts.Add(line.StartIndex)); return starts; } public static Scalar WidthOf(string text, int length) { return TextLayout.CalculateTextWidth(Font, text.GetView(0, length)); } /// /// Whether WordWrap puts a line break at `index` when it's the last spot that could hold one. /// public static bool BreaksAt(string text, int index) { if (FollowsHardBreak(text, index)) { return LineStarts(text, Scalar.PositiveInfinity).Contains(index); } // Only what follows the last hard break shares a line with `index`, and a hard break leaves the same // context behind as the start of text. int lineStart = 0; for (int candidate = index - 1; candidate > 0; --candidate) { if (FollowsHardBreak(text, candidate)) { lineStart = candidate; break; } } string line = text.Substring(lineStart); int lineIndex = index - lineStart; // Room for the text before `index`, plus the hyphen a soft hyphen shows when a line breaks at it. Scalar room = WidthOf(line, lineIndex); if (line[lineIndex - 1] == WordWrapUax14.SoftHyphen) { room += Font.MetricsTable['-'].Width; } // A lead-in leaves exactly `room`, so the line ends at `index` only if that's a break. Its inner // space is a fallback break close enough to the end that whatever follows it fits a line, so nothing // is cut at the margin; its trailing space gives start-of-text context. Scalar maxWidth = WidthOf(line, line.Length) + 8; int fillerWidth = (int)(maxWidth - room - WidthOf(" y ", 3)); string leadIn = new string('y', fillerWidth / 2) + (fillerWidth % 2 == 1 ? "z" : "") + " y "; List starts = LineStarts(leadIn + line, maxWidth); return starts.Count > 1 && starts[1] - leadIn.Length == lineIndex; } /// /// Boundaries WordWrap gets wrong, counted by the rule that decides them. Shrink this as gaps close. /// public static readonly string[] KnownMismatchesByRule = { "4.0: 205", "5.04: 200", "7.02: 143", "8.0: 203", "9.0: 429", "11.01: 141", "12.0: 61", "12.1: 2", "13.01: 282", "13.02: 283", "13.03: 144", "13.04: 140", "14.0: 215", "15.11: 114", "15.21: 144", "15.4: 139", "16.0: 20", "17.0: 2", "18.0: 132", "19.01: 2", "19.1: 1", "19.11: 1", "20.01: 114", "20.02: 83", "21.01: 1", "21.03: 2", "21.04: 6", "22.0: 4", "30.13: 2", "31.01: 3467", }; [CpuTimeout(3000)] public void MismatchesByRule_AreExactlyTheKnownGaps() { var mismatches = new SortedDictionary(); foreach (string line in File.ReadLines(TestFilePath)) { if (line.Length == 0 || line[0] == '#') { continue; } Case testCase = Parse(line); foreach (Boundary boundary in testCase.Boundaries) { if (HangsAtLineEnd(testCase.Text[boundary.Index]) || OnlySoftHyphensFollow(testCase.Text, boundary.Index)) { continue; } if (BreaksAt(testCase.Text, boundary.Index) != boundary.Breaks) { decimal rule = decimal.Parse(boundary.Rule, CultureInfo.InvariantCulture); mismatches.TryGetValue(rule, out int count); mismatches[rule] = count + 1; } } } CheckEqual(KnownMismatchesByRule, mismatches.Select(pair => pair.Key.ToString(CultureInfo.InvariantCulture) + ": " + pair.Value)); } } } }