diff --git a/CaseConverter.Test/CaseConverterTests.cs b/CaseConverter.Test/CaseConverterTests.cs index a80b4dc..a180d5b 100644 --- a/CaseConverter.Test/CaseConverterTests.cs +++ b/CaseConverter.Test/CaseConverterTests.cs @@ -429,4 +429,45 @@ public void ApostropheNotBetweenLettersShouldStillSeparateWords(string input, st { Assert.AreEqual(expected, input.ToSnakeCase()); } + + // "ß" and "fi" are lowercase letters with no single-character uppercase form, so they survive + // ToMacroCase. They used to read as lowercase when that output was converted again, which split + // "STRAßE" into "STR Aß E" and kept IsAllCaps from recognizing it. + + [TestMethod] + [DataRow("straße")] + [DataRow("maßnahmeLimit")] + [DataRow("MAX_GRÖßE")] + [DataRow("file")] + public void ToMacroCaseShouldBeIdempotentForLowercaseLettersWithNoUppercaseForm(string input) + { + string once = input.ToMacroCase(); + Assert.AreEqual(once, once.ToMacroCase()); + } + + [TestMethod] + public void AnAllCapsWordContainingSharpSShouldStayOneWord() + { + Assert.AreEqual("STRAßE", "straße".ToMacroCase()); + Assert.AreEqual("straße", "STRAßE".ToSnakeCase()); + Assert.AreEqual("Straße", "STRAßE".ToPascalCase()); + Assert.AreEqual("Straße", "STRAßE".ToTitleCase()); + Assert.AreEqual("maßnahme", "MAßNAHME".ToSnakeCase()); + Assert.AreEqual("maxGröße", "MAX_GRÖßE".ToCamelCase()); + Assert.AreEqual("maßnahme_limit", "maßnahmeLimit".ToMacroCase().ToSnakeCase()); + Assert.AreEqual("fiLE", "file".ToMacroCase().ToMacroCase()); + } + + [TestMethod] + public void ALowercaseWordEndingInSharpSShouldStillSplitBeforeACapital() + { + Assert.AreEqual("groß_foo", "großFoo".ToSnakeCase()); + } + + [TestMethod] + public void IsAllCapsShouldIgnoreLowercaseLettersWithNoUppercaseForm() + { + Assert.IsTrue("STRAßE".IsAllCaps()); + Assert.IsFalse("Straße".IsAllCaps()); + } } diff --git a/CaseConverter/CaseConverter.cs b/CaseConverter/CaseConverter.cs index cc129d7..8d9406b 100644 --- a/CaseConverter/CaseConverter.cs +++ b/CaseConverter/CaseConverter.cs @@ -147,15 +147,19 @@ private static bool IsWordBoundary(string input, int previousStart, int start, i bool previousIsUpper = char.IsUpper(input, previousStart); bool currentIsUpper = char.IsUpper(input, start); + // A lowercase letter with no uppercase form, such as "ß" or "fi", survives uppercasing, so it + // takes the case of the nearest letter that has one: "STRAßE" is a single all-caps word. + int casedNextStart = SkipLowercaseWithNoUppercase(input, nextStart); + // The tail of an acronym run that begins a new word: "XMLDoc" breaks before the "D". - if (previousIsUpper && currentIsUpper && nextStart < input.Length && char.IsLower(input, nextStart)) + if (previousIsUpper && currentIsUpper && casedNextStart < input.Length && char.IsLower(input, casedNextStart)) { return true; } // The start of a capitalised word: "fooBar" breaks before the "B". Only a letter or digit can // end the word before it, so "(Hello" and "don'T" do not split away from their punctuation. - if (!previousIsUpper && currentIsUpper && (previousIsLetter || char.IsDigit(input, previousStart))) + if (EndsWordThatIsNotUppercase(input, previousStart) && currentIsUpper && (previousIsLetter || char.IsDigit(input, previousStart))) { return true; } @@ -170,6 +174,84 @@ private static bool IsWordBoundary(string input, int previousStart, int start, i return breakBeforeAnyNonLetter || char.IsDigit(input, start); } + /// + /// Determines whether the code point at is a lowercase letter that + /// uppercasing leaves unchanged, such as "ß", "fi" or "ʼn". + /// + /// The string to inspect. + /// The index of the first code unit of the code point. + /// true if the code point is lowercase and has no uppercase form; otherwise, false. + /// + /// Such a letter is still present, and still lowercase, in the output of + /// , so it must not count as lowercase when the words of an + /// all-caps string are found again. + /// + private static bool IsLowercaseWithNoUppercase(string input, int index) + { + if (!char.IsLower(input, index)) + { + return false; + } + +#if NETSTANDARD2_0 +#pragma warning disable IDE0057 // Substring cannot be simplified in netstandard2.0 + string codePoint = input.Substring(index, CodePointLength(input, index)); +#pragma warning restore IDE0057 +#else + string codePoint = input[index..(index + CodePointLength(input, index))]; +#endif + return string.Equals(codePoint.ToUpperInvariant(), codePoint, StringComparison.Ordinal); + } + + /// + /// Returns the index of the first code point at or after that is not a + /// lowercase letter with no uppercase form. + /// + /// The string to inspect. + /// The index to start from, which may be past the end. + /// The index found, or the length of if there is none. + private static int SkipLowercaseWithNoUppercase(string input, int index) + { + while (index < input.Length && IsLowercaseWithNoUppercase(input, index)) + { + index += CodePointLength(input, index); + } + + return index; + } + + /// + /// Determines whether the code point at ends a word that is not uppercase, + /// so that a capital after it starts a new word. + /// + /// The string to inspect. + /// The index of the first code unit of the code point before the capital. + /// true if a capital after the code point starts a new word; otherwise, false. + /// + /// A lowercase letter with no uppercase form takes the case of the nearest letter before it in the + /// same word, so "STRAßE" does not break before the "E" while "großFoo" still + /// breaks before the "F". With no such letter, as in "fiLE", it does not end a word. + /// + private static bool EndsWordThatIsNotUppercase(string input, int index) + { + if (!IsLowercaseWithNoUppercase(input, index)) + { + return !char.IsUpper(input, index); + } + + for (int i = index; i > 0;) + { + i -= i >= 2 && char.IsSurrogatePair(input, i - 2) ? 2 : 1; + + if (!IsLowercaseWithNoUppercase(input, i)) + { + return char.IsLetter(input, i) && !char.IsUpper(input, i); + } + } + + return false; + } + /// /// Returns a copy of this string, trimmed and with runs of spaces collapsed to one, with the first character converted to lowercase. /// @@ -335,7 +417,9 @@ public static bool IsAllCaps(this string output) { int length = CodePointLength(output, i); - if (char.IsLetter(output, i) && !char.IsUpper(output, i)) + // A lowercase letter with no uppercase form, such as "ß", is left as it is by uppercasing, + // so it does not stop "STRAßE" from being all caps. + if (char.IsLetter(output, i) && !char.IsUpper(output, i) && !IsLowercaseWithNoUppercase(output, i)) { return false; }