From ebd433b973870210d477b2d9cab45321d0da5c48 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 02:30:58 +0000 Subject: [PATCH] =?UTF-8?q?Keep=20an=20all-caps=20word=20containing=20?= =?UTF-8?q?=C3=9F=20in=20one=20piece,=20so=20ToMacroCase=20is=20idempotent?= =?UTF-8?q?=20[patch]?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ToUpperInvariant leaves ß, fi and ʼn unchanged, so ToMacroCase output such as "STRAßE" still holds a lowercase letter. Converting it again split it into "STR_Aß_E", and IsAllCaps did not count it as all caps. A lowercase letter with no uppercase form now takes the case of the nearest cased letter before it for word boundaries, is skipped when looking ahead for the lowercase tail of an acronym, and is ignored by IsAllCaps. Fixes ktsu-dev/CaseConverter#87 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01N3tMVtLPuNwsQF7rUEBTMQ --- CaseConverter.Test/CaseConverterTests.cs | 41 +++++++++++ CaseConverter/CaseConverter.cs | 90 +++++++++++++++++++++++- 2 files changed, 128 insertions(+), 3 deletions(-) diff --git a/CaseConverter.Test/CaseConverterTests.cs b/CaseConverter.Test/CaseConverterTests.cs index a80b4dc..a180d5b 100644 --- a/CaseConverter.Test/CaseConverterTests.cs +++ b/CaseConverter.Test/CaseConverterTests.cs @@ -429,4 +429,45 @@ public void ApostropheNotBetweenLettersShouldStillSeparateWords(string input, st { Assert.AreEqual(expected, input.ToSnakeCase()); } + + // "ß" and "fi" are lowercase letters with no single-character uppercase form, so they survive + // ToMacroCase. They used to read as lowercase when that output was converted again, which split + // "STRAßE" into "STR Aß E" and kept IsAllCaps from recognizing it. + + [TestMethod] + [DataRow("straße")] + [DataRow("maßnahmeLimit")] + [DataRow("MAX_GRÖßE")] + [DataRow("file")] + public void ToMacroCaseShouldBeIdempotentForLowercaseLettersWithNoUppercaseForm(string input) + { + string once = input.ToMacroCase(); + Assert.AreEqual(once, once.ToMacroCase()); + } + + [TestMethod] + public void AnAllCapsWordContainingSharpSShouldStayOneWord() + { + Assert.AreEqual("STRAßE", "straße".ToMacroCase()); + Assert.AreEqual("straße", "STRAßE".ToSnakeCase()); + Assert.AreEqual("Straße", "STRAßE".ToPascalCase()); + Assert.AreEqual("Straße", "STRAßE".ToTitleCase()); + Assert.AreEqual("maßnahme", "MAßNAHME".ToSnakeCase()); + Assert.AreEqual("maxGröße", "MAX_GRÖßE".ToCamelCase()); + Assert.AreEqual("maßnahme_limit", "maßnahmeLimit".ToMacroCase().ToSnakeCase()); + Assert.AreEqual("fiLE", "file".ToMacroCase().ToMacroCase()); + } + + [TestMethod] + public void ALowercaseWordEndingInSharpSShouldStillSplitBeforeACapital() + { + Assert.AreEqual("groß_foo", "großFoo".ToSnakeCase()); + } + + [TestMethod] + public void IsAllCapsShouldIgnoreLowercaseLettersWithNoUppercaseForm() + { + Assert.IsTrue("STRAßE".IsAllCaps()); + Assert.IsFalse("Straße".IsAllCaps()); + } } diff --git a/CaseConverter/CaseConverter.cs b/CaseConverter/CaseConverter.cs index cc129d7..8d9406b 100644 --- a/CaseConverter/CaseConverter.cs +++ b/CaseConverter/CaseConverter.cs @@ -147,15 +147,19 @@ private static bool IsWordBoundary(string input, int previousStart, int start, i bool previousIsUpper = char.IsUpper(input, previousStart); bool currentIsUpper = char.IsUpper(input, start); + // A lowercase letter with no uppercase form, such as "ß" or "fi", survives uppercasing, so it + // takes the case of the nearest letter that has one: "STRAßE" is a single all-caps word. + int casedNextStart = SkipLowercaseWithNoUppercase(input, nextStart); + // The tail of an acronym run that begins a new word: "XMLDoc" breaks before the "D". - if (previousIsUpper && currentIsUpper && nextStart < input.Length && char.IsLower(input, nextStart)) + if (previousIsUpper && currentIsUpper && casedNextStart < input.Length && char.IsLower(input, casedNextStart)) { return true; } // The start of a capitalised word: "fooBar" breaks before the "B". Only a letter or digit can // end the word before it, so "(Hello" and "don'T" do not split away from their punctuation. - if (!previousIsUpper && currentIsUpper && (previousIsLetter || char.IsDigit(input, previousStart))) + if (EndsWordThatIsNotUppercase(input, previousStart) && currentIsUpper && (previousIsLetter || char.IsDigit(input, previousStart))) { return true; } @@ -170,6 +174,84 @@ private static bool IsWordBoundary(string input, int previousStart, int start, i return breakBeforeAnyNonLetter || char.IsDigit(input, start); } + /// + /// Determines whether the code point at is a lowercase letter that + /// uppercasing leaves unchanged, such as "ß", "fi" or "ʼn". + /// + /// The string to inspect. + /// The index of the first code unit of the code point. + /// true if the code point is lowercase and has no uppercase form; otherwise, false. + /// + /// Such a letter is still present, and still lowercase, in the output of + /// , so it must not count as lowercase when the words of an + /// all-caps string are found again. + /// + private static bool IsLowercaseWithNoUppercase(string input, int index) + { + if (!char.IsLower(input, index)) + { + return false; + } + +#if NETSTANDARD2_0 +#pragma warning disable IDE0057 // Substring cannot be simplified in netstandard2.0 + string codePoint = input.Substring(index, CodePointLength(input, index)); +#pragma warning restore IDE0057 +#else + string codePoint = input[index..(index + CodePointLength(input, index))]; +#endif + return string.Equals(codePoint.ToUpperInvariant(), codePoint, StringComparison.Ordinal); + } + + /// + /// Returns the index of the first code point at or after that is not a + /// lowercase letter with no uppercase form. + /// + /// The string to inspect. + /// The index to start from, which may be past the end. + /// The index found, or the length of if there is none. + private static int SkipLowercaseWithNoUppercase(string input, int index) + { + while (index < input.Length && IsLowercaseWithNoUppercase(input, index)) + { + index += CodePointLength(input, index); + } + + return index; + } + + /// + /// Determines whether the code point at ends a word that is not uppercase, + /// so that a capital after it starts a new word. + /// + /// The string to inspect. + /// The index of the first code unit of the code point before the capital. + /// true if a capital after the code point starts a new word; otherwise, false. + /// + /// A lowercase letter with no uppercase form takes the case of the nearest letter before it in the + /// same word, so "STRAßE" does not break before the "E" while "großFoo" still + /// breaks before the "F". With no such letter, as in "fiLE", it does not end a word. + /// + private static bool EndsWordThatIsNotUppercase(string input, int index) + { + if (!IsLowercaseWithNoUppercase(input, index)) + { + return !char.IsUpper(input, index); + } + + for (int i = index; i > 0;) + { + i -= i >= 2 && char.IsSurrogatePair(input, i - 2) ? 2 : 1; + + if (!IsLowercaseWithNoUppercase(input, i)) + { + return char.IsLetter(input, i) && !char.IsUpper(input, i); + } + } + + return false; + } + /// /// Returns a copy of this string, trimmed and with runs of spaces collapsed to one, with the first character converted to lowercase. /// @@ -335,7 +417,9 @@ public static bool IsAllCaps(this string output) { int length = CodePointLength(output, i); - if (char.IsLetter(output, i) && !char.IsUpper(output, i)) + // A lowercase letter with no uppercase form, such as "ß", is left as it is by uppercasing, + // so it does not stop "STRAßE" from being all caps. + if (char.IsLetter(output, i) && !char.IsUpper(output, i) && !IsLowercaseWithNoUppercase(output, i)) { return false; }