From 37e8fd20eed144a8dc093c6bfe03c1655d266455 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 12:33:36 +0000 Subject: [PATCH] Preserve letters outside the BMP in every case conversion MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The three regexes tested Unicode categories per UTF-16 code unit. A letter outside the Basic Multilingual Plane is a surrogate pair whose halves are both categorised as Surrogate rather than as a letter, so `[^\p{L}0-9]` matched both halves and the letter was replaced with spaces and then collapsed away. Every conversion silently dropped it: "set 𐐀abc" became "SetAbc". Two further faults shared that root cause. `IsAllCaps` stripped both surrogate halves before testing, so an astral lowercase letter was skipped entirely and "HELLO 𐐨" reported true. The splitter's letter-to-non-letter alternative read the high surrogate as a non-letter and inserted a spurious word boundary, so "abc𝒳def" split into two tokens. Replace all three regexes with walks by code point. The conversions now treat an astral letter exactly as they treat its BMP analogue, and astral characters that are not letters stay separators as before. Behaviour for BMP input is unchanged: the new code was compared against the regex implementation over 12,191 inputs across all six public methods with no differences. Four of the five new tests fail without this change. Fixes #70 Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_014RUsrYfSAuSQ7VWws7vzsr --- CaseConverter.Test/CaseConverterTests.cs | 51 ++++++++ CaseConverter/CaseConverter.cs | 160 ++++++++++++++++++----- 2 files changed, 175 insertions(+), 36 deletions(-) diff --git a/CaseConverter.Test/CaseConverterTests.cs b/CaseConverter.Test/CaseConverterTests.cs index 6bc6994..1457aa2 100644 --- a/CaseConverter.Test/CaseConverterTests.cs +++ b/CaseConverter.Test/CaseConverterTests.cs @@ -197,4 +197,55 @@ public void ToMacroCaseShouldNotIncludeLeadingOrTrailingSeparators() string result = input.ToMacroCase(); Assert.AreEqual("PRIVATE_FIELD", result); } + + // U+10400 DESERET CAPITAL LONG I and U+10428 DESERET SMALL LONG I are letters outside the + // Basic Multilingual Plane, so each is a surrogate pair in UTF-16. They are a cased pair, + // which lets these tests check case mapping as well as preservation. + private const string DeseretCapitalLongI = "\U00010400"; + private const string DeseretSmallLongI = "\U00010428"; + + [TestMethod] + public void ToSnakeCaseShouldPreserveLettersOutsideTheBasicMultilingualPlane() + { + string input = $"{DeseretCapitalLongI}abc"; + string result = input.ToSnakeCase(); + Assert.AreEqual($"{DeseretSmallLongI}abc", result, "A letter outside the BMP must be lowercased, not deleted."); + } + + [TestMethod] + public void ToPascalCaseShouldPreserveLettersOutsideTheBasicMultilingualPlane() + { + string input = $"set {DeseretCapitalLongI}abc"; + string result = input.ToPascalCase(); + Assert.AreEqual($"Set{DeseretCapitalLongI}abc", result, "A letter outside the BMP must survive the conversion."); + } + + [TestMethod] + public void ToMacroCaseShouldTreatAnAstralUppercaseLetterAsAWordBoundary() + { + string input = $"abc{DeseretCapitalLongI}def"; + string result = input.ToMacroCase(); + + // "abcXdef" breaks before the "X"; an uppercase letter outside the BMP must behave the same. + Assert.AreEqual($"ABC_{DeseretCapitalLongI}DEF", result); + Assert.AreEqual("ABC_XDEF", "abcXdef".ToMacroCase(), "The BMP analogue this case is matched against."); + } + + [TestMethod] + public void IsAllCapsShouldReturnFalseForAnAstralLowercaseLetter() + { + string input = $"HELLO {DeseretSmallLongI}"; + bool result = input.IsAllCaps(); + Assert.IsFalse(result, "A lowercase letter outside the BMP must count as lowercase, not be skipped."); + } + + [TestMethod] + public void ToSnakeCaseShouldStillDropAstralCharactersThatAreNotLetters() + { + // U+1F600 GRINNING FACE is a surrogate pair but not a letter, so it is a separator like + // any other non-alphanumeric. This pins the boundary of the surrogate-pair handling. + string input = "emoji \U0001F600 here"; + string result = input.ToSnakeCase(); + Assert.AreEqual("emoji_here", result); + } } diff --git a/CaseConverter/CaseConverter.cs b/CaseConverter/CaseConverter.cs index 6e6e5ae..33c437b 100644 --- a/CaseConverter/CaseConverter.cs +++ b/CaseConverter/CaseConverter.cs @@ -5,7 +5,7 @@ namespace ktsu.CaseConverter; using System.Globalization; -using System.Text.RegularExpressions; +using System.Text; /// /// Provides extension methods for converting strings between different cases. @@ -13,41 +13,113 @@ namespace ktsu.CaseConverter; public static partial class CaseConverter { /// - /// Gets a that matches all non-alphanumeric characters. + /// Returns the number of UTF-16 code units making up the code point at . /// - /// The compiled instance. -#if NET7_0_OR_GREATER - [GeneratedRegex(@"[^\p{L}0-9]", RegexOptions.Compiled)] - private static partial Regex NonAlphaNumericRegex(); -#else - private static Regex NonAlphaNumericRegex() => NonAlphaNumericRegexInstance; - private static readonly Regex NonAlphaNumericRegexInstance = new(@"[^\p{L}0-9]", RegexOptions.Compiled); -#endif + /// The string to inspect. + /// The index of the first code unit of the code point. + /// 2 for a surrogate pair, otherwise 1. + private static int CodePointLength(string input, int index) => char.IsSurrogatePair(input, index) ? 2 : 1; /// - /// Gets a that matches all non-alphabetic characters. + /// Replaces every code point that is not a Unicode letter or an ASCII digit with a space. /// - /// The compiled instance. -#if NET7_0_OR_GREATER - [GeneratedRegex(@"[^\p{L}]", RegexOptions.Compiled)] - private static partial Regex NonAlphaRegex(); -#else - private static Regex NonAlphaRegex() => NonAlphaRegexInstance; - private static readonly Regex NonAlphaRegexInstance = new(@"[^\p{L}]", RegexOptions.Compiled); -#endif + /// The string to process. + /// A new string with each non-alphanumeric code point replaced by a space. + /// + /// This walks by code point rather than by UTF-16 code unit. The regex this replaces + /// ([^\p{L}0-9]) matched per code unit, and a surrogate code unit is categorised as + /// rather than as a letter — so each half of a + /// surrogate pair matched and letters outside the Basic Multilingual Plane were silently + /// deleted instead of preserved. + /// + private static string ReplaceNonAlphaNumericWithSpace(string input) + { + StringBuilder builder = new(input.Length); + + for (int i = 0; i < input.Length;) + { + int length = CodePointLength(input, i); + + if (char.IsLetter(input, i) || input[i] is >= '0' and <= '9') + { + builder.Append(input, i, length); + } + else + { + builder.Append(' '); + } + + i += length; + } + + return builder.ToString(); + } /// - /// Gets a that splits on case changes, such as transitions from - /// lower to upper or upper to lower within a string. + /// Inserts a space at each case change, such as transitions from lower to upper or from a + /// letter to a non-letter. /// - /// The compiled instance. -#if NET7_0_OR_GREATER - [GeneratedRegex(@"(?<=[\p{Lu}])(?=[\p{Lu}][\p{Ll}])|(?<=[^\p{Lu}])(?=[\p{Lu}])|(?<=[\p{L}])(?=[^\p{L}])", RegexOptions.Compiled)] - private static partial Regex SplitOnCaseChangeRegex(); -#else - private static Regex SplitOnCaseChangeRegex() => SplitOnCaseChangeRegexInstance; - private static readonly Regex SplitOnCaseChangeRegexInstance = new(@"(?<=[\p{Lu}])(?=[\p{Lu}][\p{Ll}])|(?<=[^\p{Lu}])(?=[\p{Lu}])|(?<=[\p{L}])(?=[^\p{L}])", RegexOptions.Compiled); -#endif + /// The string to process. + /// A new string with a space inserted at each word boundary. + /// + /// This walks by code point, for the same reason + /// does. The regex this replaces tested + /// \p{L} and \p{Lu} per UTF-16 code unit, so a letter outside the Basic + /// Multilingual Plane read as a non-letter and had a spurious word boundary inserted + /// before it. + /// + private static string SplitOnCaseChange(string input) + { + StringBuilder builder = new(input.Length); + int previousStart = -1; + + for (int i = 0; i < input.Length;) + { + int length = CodePointLength(input, i); + int nextStart = i + length; + + if (previousStart >= 0 && IsWordBoundary(input, previousStart, i, nextStart)) + { + builder.Append(' '); + } + + builder.Append(input, i, length); + previousStart = i; + i = nextStart; + } + + return builder.ToString(); + } + + /// + /// Determines whether a word boundary falls immediately before the code point at + /// . + /// + /// The string being split. + /// The index of the preceding code point. + /// The index of the code point to test. + /// The index of the following code point, which may be past the end. + /// true if a space belongs before ; otherwise, false. + private static bool IsWordBoundary(string input, int previousStart, int start, int nextStart) + { + bool previousIsUpper = char.IsUpper(input, previousStart); + bool currentIsUpper = char.IsUpper(input, start); + + // The tail of an acronym run that begins a new word: "XMLDoc" breaks before the "D". + if (previousIsUpper && currentIsUpper && nextStart < input.Length && char.IsLower(input, nextStart)) + { + return true; + } + + // The start of a capitalised word: "fooBar" breaks before the "B". + if (!previousIsUpper && currentIsUpper) + { + return true; + } + + // A letter followed by a non-letter: "abc123" breaks before the "1". + return char.IsLetter(input, previousStart) && !char.IsLetter(input, start); + } /// /// Returns a copy of this string with the first character converted to lowercase. @@ -90,8 +162,10 @@ public static string ToUppercaseFirstChar(this string input) /// A new string in Title Case. public static string ToTitleCase(this string input) { + Ensure.NotNull(input); + string output = input; - output = SplitOnCaseChangeRegex().Replace(output, " "); + output = SplitOnCaseChange(output); output = CollapseSpaces(output).Trim(); // If the input is all caps, we want to convert it to lowercase before converting to title case, @@ -111,8 +185,22 @@ public static string ToTitleCase(this string input) /// true if all alphabetic characters are uppercase; otherwise, false. public static bool IsAllCaps(this string output) { - string alphaChars = NonAlphaRegex().Replace(output, string.Empty); - return alphaChars.All(char.IsUpper); + Ensure.NotNull(output); + + for (int i = 0; i < output.Length;) + { + int length = CodePointLength(output, i); + + if (char.IsLetter(output, i) && !char.IsUpper(output, i)) + { + return false; + } + + i += length; + } + + // A string with no letters at all is vacuously all caps. + return true; } /// @@ -147,8 +235,8 @@ public static string ToPascalCase(this string input) Ensure.NotNull(input); string output = input; - output = NonAlphaNumericRegex().Replace(output, " "); - output = SplitOnCaseChangeRegex().Replace(output, " "); + output = ReplaceNonAlphaNumericWithSpace(output); + output = SplitOnCaseChange(output); output = output.ToTitleCase(); #if NETSTANDARD2_0 output = output.Replace(" ", string.Empty); @@ -213,8 +301,8 @@ public static string ToMacroCase(this string input) Ensure.NotNull(input); string output = input.Trim(); - output = NonAlphaNumericRegex().Replace(output, " "); - output = SplitOnCaseChangeRegex().Replace(output, " ").ToUpperInvariant(); + output = ReplaceNonAlphaNumericWithSpace(output); + output = SplitOnCaseChange(output).ToUpperInvariant(); output = CollapseSpaces(output).Trim(); #if NETSTANDARD2_0 output = output.Replace(" ", "_");