diff --git a/CaseConverter.Test/CaseConverterTests.cs b/CaseConverter.Test/CaseConverterTests.cs index 6bc6994..1457aa2 100644 --- a/CaseConverter.Test/CaseConverterTests.cs +++ b/CaseConverter.Test/CaseConverterTests.cs @@ -197,4 +197,55 @@ public void ToMacroCaseShouldNotIncludeLeadingOrTrailingSeparators() string result = input.ToMacroCase(); Assert.AreEqual("PRIVATE_FIELD", result); } + + // U+10400 DESERET CAPITAL LONG I and U+10428 DESERET SMALL LONG I are letters outside the + // Basic Multilingual Plane, so each is a surrogate pair in UTF-16. They are a cased pair, + // which lets these tests check case mapping as well as preservation. + private const string DeseretCapitalLongI = "\U00010400"; + private const string DeseretSmallLongI = "\U00010428"; + + [TestMethod] + public void ToSnakeCaseShouldPreserveLettersOutsideTheBasicMultilingualPlane() + { + string input = $"{DeseretCapitalLongI}abc"; + string result = input.ToSnakeCase(); + Assert.AreEqual($"{DeseretSmallLongI}abc", result, "A letter outside the BMP must be lowercased, not deleted."); + } + + [TestMethod] + public void ToPascalCaseShouldPreserveLettersOutsideTheBasicMultilingualPlane() + { + string input = $"set {DeseretCapitalLongI}abc"; + string result = input.ToPascalCase(); + Assert.AreEqual($"Set{DeseretCapitalLongI}abc", result, "A letter outside the BMP must survive the conversion."); + } + + [TestMethod] + public void ToMacroCaseShouldTreatAnAstralUppercaseLetterAsAWordBoundary() + { + string input = $"abc{DeseretCapitalLongI}def"; + string result = input.ToMacroCase(); + + // "abcXdef" breaks before the "X"; an uppercase letter outside the BMP must behave the same. + Assert.AreEqual($"ABC_{DeseretCapitalLongI}DEF", result); + Assert.AreEqual("ABC_XDEF", "abcXdef".ToMacroCase(), "The BMP analogue this case is matched against."); + } + + [TestMethod] + public void IsAllCapsShouldReturnFalseForAnAstralLowercaseLetter() + { + string input = $"HELLO {DeseretSmallLongI}"; + bool result = input.IsAllCaps(); + Assert.IsFalse(result, "A lowercase letter outside the BMP must count as lowercase, not be skipped."); + } + + [TestMethod] + public void ToSnakeCaseShouldStillDropAstralCharactersThatAreNotLetters() + { + // U+1F600 GRINNING FACE is a surrogate pair but not a letter, so it is a separator like + // any other non-alphanumeric. This pins the boundary of the surrogate-pair handling. + string input = "emoji \U0001F600 here"; + string result = input.ToSnakeCase(); + Assert.AreEqual("emoji_here", result); + } } diff --git a/CaseConverter/CaseConverter.cs b/CaseConverter/CaseConverter.cs index 6e6e5ae..33c437b 100644 --- a/CaseConverter/CaseConverter.cs +++ b/CaseConverter/CaseConverter.cs @@ -5,7 +5,7 @@ namespace ktsu.CaseConverter; using System.Globalization; -using System.Text.RegularExpressions; +using System.Text; /// /// Provides extension methods for converting strings between different cases. @@ -13,41 +13,113 @@ namespace ktsu.CaseConverter; public static partial class CaseConverter { /// - /// Gets a that matches all non-alphanumeric characters. + /// Returns the number of UTF-16 code units making up the code point at . /// - /// The compiled instance. -#if NET7_0_OR_GREATER - [GeneratedRegex(@"[^\p{L}0-9]", RegexOptions.Compiled)] - private static partial Regex NonAlphaNumericRegex(); -#else - private static Regex NonAlphaNumericRegex() => NonAlphaNumericRegexInstance; - private static readonly Regex NonAlphaNumericRegexInstance = new(@"[^\p{L}0-9]", RegexOptions.Compiled); -#endif + /// The string to inspect. + /// The index of the first code unit of the code point. + /// 2 for a surrogate pair, otherwise 1. + private static int CodePointLength(string input, int index) => char.IsSurrogatePair(input, index) ? 2 : 1; /// - /// Gets a that matches all non-alphabetic characters. + /// Replaces every code point that is not a Unicode letter or an ASCII digit with a space. /// - /// The compiled instance. -#if NET7_0_OR_GREATER - [GeneratedRegex(@"[^\p{L}]", RegexOptions.Compiled)] - private static partial Regex NonAlphaRegex(); -#else - private static Regex NonAlphaRegex() => NonAlphaRegexInstance; - private static readonly Regex NonAlphaRegexInstance = new(@"[^\p{L}]", RegexOptions.Compiled); -#endif + /// The string to process. + /// A new string with each non-alphanumeric code point replaced by a space. + /// + /// This walks by code point rather than by UTF-16 code unit. The regex this replaces + /// ([^\p{L}0-9]) matched per code unit, and a surrogate code unit is categorised as + /// rather than as a letter — so each half of a + /// surrogate pair matched and letters outside the Basic Multilingual Plane were silently + /// deleted instead of preserved. + /// + private static string ReplaceNonAlphaNumericWithSpace(string input) + { + StringBuilder builder = new(input.Length); + + for (int i = 0; i < input.Length;) + { + int length = CodePointLength(input, i); + + if (char.IsLetter(input, i) || input[i] is >= '0' and <= '9') + { + builder.Append(input, i, length); + } + else + { + builder.Append(' '); + } + + i += length; + } + + return builder.ToString(); + } /// - /// Gets a that splits on case changes, such as transitions from - /// lower to upper or upper to lower within a string. + /// Inserts a space at each case change, such as transitions from lower to upper or from a + /// letter to a non-letter. /// - /// The compiled instance. -#if NET7_0_OR_GREATER - [GeneratedRegex(@"(?<=[\p{Lu}])(?=[\p{Lu}][\p{Ll}])|(?<=[^\p{Lu}])(?=[\p{Lu}])|(?<=[\p{L}])(?=[^\p{L}])", RegexOptions.Compiled)] - private static partial Regex SplitOnCaseChangeRegex(); -#else - private static Regex SplitOnCaseChangeRegex() => SplitOnCaseChangeRegexInstance; - private static readonly Regex SplitOnCaseChangeRegexInstance = new(@"(?<=[\p{Lu}])(?=[\p{Lu}][\p{Ll}])|(?<=[^\p{Lu}])(?=[\p{Lu}])|(?<=[\p{L}])(?=[^\p{L}])", RegexOptions.Compiled); -#endif + /// The string to process. + /// A new string with a space inserted at each word boundary. + /// + /// This walks by code point, for the same reason + /// does. The regex this replaces tested + /// \p{L} and \p{Lu} per UTF-16 code unit, so a letter outside the Basic + /// Multilingual Plane read as a non-letter and had a spurious word boundary inserted + /// before it. + /// + private static string SplitOnCaseChange(string input) + { + StringBuilder builder = new(input.Length); + int previousStart = -1; + + for (int i = 0; i < input.Length;) + { + int length = CodePointLength(input, i); + int nextStart = i + length; + + if (previousStart >= 0 && IsWordBoundary(input, previousStart, i, nextStart)) + { + builder.Append(' '); + } + + builder.Append(input, i, length); + previousStart = i; + i = nextStart; + } + + return builder.ToString(); + } + + /// + /// Determines whether a word boundary falls immediately before the code point at + /// . + /// + /// The string being split. + /// The index of the preceding code point. + /// The index of the code point to test. + /// The index of the following code point, which may be past the end. + /// true if a space belongs before ; otherwise, false. + private static bool IsWordBoundary(string input, int previousStart, int start, int nextStart) + { + bool previousIsUpper = char.IsUpper(input, previousStart); + bool currentIsUpper = char.IsUpper(input, start); + + // The tail of an acronym run that begins a new word: "XMLDoc" breaks before the "D". + if (previousIsUpper && currentIsUpper && nextStart < input.Length && char.IsLower(input, nextStart)) + { + return true; + } + + // The start of a capitalised word: "fooBar" breaks before the "B". + if (!previousIsUpper && currentIsUpper) + { + return true; + } + + // A letter followed by a non-letter: "abc123" breaks before the "1". + return char.IsLetter(input, previousStart) && !char.IsLetter(input, start); + } /// /// Returns a copy of this string with the first character converted to lowercase. @@ -90,8 +162,10 @@ public static string ToUppercaseFirstChar(this string input) /// A new string in Title Case. public static string ToTitleCase(this string input) { + Ensure.NotNull(input); + string output = input; - output = SplitOnCaseChangeRegex().Replace(output, " "); + output = SplitOnCaseChange(output); output = CollapseSpaces(output).Trim(); // If the input is all caps, we want to convert it to lowercase before converting to title case, @@ -111,8 +185,22 @@ public static string ToTitleCase(this string input) /// true if all alphabetic characters are uppercase; otherwise, false. public static bool IsAllCaps(this string output) { - string alphaChars = NonAlphaRegex().Replace(output, string.Empty); - return alphaChars.All(char.IsUpper); + Ensure.NotNull(output); + + for (int i = 0; i < output.Length;) + { + int length = CodePointLength(output, i); + + if (char.IsLetter(output, i) && !char.IsUpper(output, i)) + { + return false; + } + + i += length; + } + + // A string with no letters at all is vacuously all caps. + return true; } /// @@ -147,8 +235,8 @@ public static string ToPascalCase(this string input) Ensure.NotNull(input); string output = input; - output = NonAlphaNumericRegex().Replace(output, " "); - output = SplitOnCaseChangeRegex().Replace(output, " "); + output = ReplaceNonAlphaNumericWithSpace(output); + output = SplitOnCaseChange(output); output = output.ToTitleCase(); #if NETSTANDARD2_0 output = output.Replace(" ", string.Empty); @@ -213,8 +301,8 @@ public static string ToMacroCase(this string input) Ensure.NotNull(input); string output = input.Trim(); - output = NonAlphaNumericRegex().Replace(output, " "); - output = SplitOnCaseChangeRegex().Replace(output, " ").ToUpperInvariant(); + output = ReplaceNonAlphaNumericWithSpace(output); + output = SplitOnCaseChange(output).ToUpperInvariant(); output = CollapseSpaces(output).Trim(); #if NETSTANDARD2_0 output = output.Replace(" ", "_");