From d8bdab2ba7a6dc69d48b4cfee999cb83096fb614 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 26 Sep 2026 17:26:04 +0000 Subject: [PATCH] Keep punctuation attached to its word in ToTitleCase ToTitleCase split before every non-letter, so punctuation became a word of its own and TextInfo.ToTitleCase capitalized the letter after it: "hello, world" came out as "Hello , World" and "don't stop" as "Don 'T Stop". In title case, break after a letter only before a digit, and treat an underscore as a word separator. A case change now starts a new word only after a letter or digit, so "(Hello" stays together. ToPascalCase and ToMacroCase strip punctuation before splitting, so their output is unchanged. Fixes ktsu-dev/CaseConverter#75 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_018AEYBNCRr6chFLiZDv1MVj --- CaseConverter.Test/CaseConverterTests.cs | 37 +++++++++++++++++++ CaseConverter/CaseConverter.cs | 46 +++++++++++++++++++----- 2 files changed, 74 insertions(+), 9 deletions(-) diff --git a/CaseConverter.Test/CaseConverterTests.cs b/CaseConverter.Test/CaseConverterTests.cs index c72694e..30c3c84 100644 --- a/CaseConverter.Test/CaseConverterTests.cs +++ b/CaseConverter.Test/CaseConverterTests.cs @@ -103,6 +103,43 @@ public void ToTitleCaseShouldNormalizeAnAllCapsWordIndependentlyOfItsNeighbours( Assert.AreEqual("Parse Http Header", "parse HTTP header".ToTitleCase()); } + // ToTitleCase used to split before every non-letter, so punctuation became a word of its own and + // TextInfo.ToTitleCase capitalized the letter after it. + + [TestMethod] + public void ToTitleCaseShouldKeepACommaWithTheWordBeforeIt() + { + Assert.AreEqual("Hello, World", "hello, world".ToTitleCase()); + } + + [TestMethod] + public void ToTitleCaseShouldKeepAnApostropheInsideItsWord() + { + Assert.AreEqual("Don't Stop", "don't stop".ToTitleCase()); + } + + [TestMethod] + public void ToTitleCaseShouldTreatAnUnderscoreAsAWordSeparator() + { + Assert.AreEqual("Foo Bar", "foo_bar".ToTitleCase()); + } + + [TestMethod] + public void ToTitleCaseShouldKeepOtherPunctuationInPlace() + { + Assert.AreEqual("Part 1: Setup", "part 1: setup".ToTitleCase()); + Assert.AreEqual("What's New?", "what's new?".ToTitleCase()); + Assert.AreEqual("(Hello) World", "(Hello) world".ToTitleCase()); + } + + [TestMethod] + public void ToTitleCaseShouldStillSplitOnCaseChangesAndDigits() + { + Assert.AreEqual("Foo Bar", "fooBar".ToTitleCase()); + Assert.AreEqual("Xml Doc", "XMLDoc".ToTitleCase()); + Assert.AreEqual("Abc 123", "abc123".ToTitleCase()); + } + [TestMethod] public void ToPascalCaseShouldNormalizeAnAllCapsWordIndependentlyOfItsNeighbours() { diff --git a/CaseConverter/CaseConverter.cs b/CaseConverter/CaseConverter.cs index 01e9877..30a0796 100644 --- a/CaseConverter/CaseConverter.cs +++ b/CaseConverter/CaseConverter.cs @@ -68,7 +68,19 @@ private static string ReplaceNonAlphaNumericWithSpace(string input) /// Multilingual Plane read as a non-letter and had a spurious word boundary inserted /// before it. /// - private static string SplitOnCaseChange(string input) + private static string SplitOnCaseChange(string input) => SplitOnCaseChange(input, breakBeforeAnyNonLetter: true); + + /// + /// Inserts a space at each case change, and before whatever follows a letter as + /// selects. + /// + /// The string to process. + /// + /// true to break between a letter and any non-letter that follows it; false to break + /// only between a letter and a digit, leaving punctuation attached to the word before it. + /// + /// A new string with a space inserted at each word boundary. + private static string SplitOnCaseChange(string input, bool breakBeforeAnyNonLetter) { StringBuilder builder = new(input.Length); int previousStart = -1; @@ -78,7 +90,7 @@ private static string SplitOnCaseChange(string input) int length = CodePointLength(input, i); int nextStart = i + length; - if (previousStart >= 0 && IsWordBoundary(input, previousStart, i, nextStart)) + if (previousStart >= 0 && IsWordBoundary(input, previousStart, i, nextStart, breakBeforeAnyNonLetter)) { builder.Append(' '); } @@ -99,9 +111,13 @@ private static string SplitOnCaseChange(string input) /// The index of the preceding code point. /// The index of the code point to test. /// The index of the following code point, which may be past the end. + /// + /// true to break between a letter and any non-letter; false to break only between a letter and a digit. + /// /// true if a space belongs before ; otherwise, false. - private static bool IsWordBoundary(string input, int previousStart, int start, int nextStart) + private static bool IsWordBoundary(string input, int previousStart, int start, int nextStart, bool breakBeforeAnyNonLetter) { + bool previousIsLetter = char.IsLetter(input, previousStart); bool previousIsUpper = char.IsUpper(input, previousStart); bool currentIsUpper = char.IsUpper(input, start); @@ -111,14 +127,21 @@ private static bool IsWordBoundary(string input, int previousStart, int start, i return true; } - // The start of a capitalised word: "fooBar" breaks before the "B". - if (!previousIsUpper && currentIsUpper) + // The start of a capitalised word: "fooBar" breaks before the "B". Only a letter or digit can + // end the word before it, so "(Hello" and "don'T" do not split away from their punctuation. + if (!previousIsUpper && currentIsUpper && (previousIsLetter || char.IsDigit(input, previousStart))) { return true; } - // A letter followed by a non-letter: "abc123" breaks before the "1". - return char.IsLetter(input, previousStart) && !char.IsLetter(input, start); + // A letter followed by a non-letter: "abc123" breaks before the "1", and, when asked to, + // "abc_def" breaks before the "_". + if (!previousIsLetter || char.IsLetter(input, start)) + { + return false; + } + + return breakBeforeAnyNonLetter || char.IsDigit(input, start); } /// @@ -195,13 +218,18 @@ private static string LowercaseAllCapsWords(string input) /// An all-caps word is normalized rather than preserved as an acronym, so "HTTP" becomes /// "Http" and "parse HTTP header" becomes "Parse Http Header". The decision is /// made per word, so a word converts the same way whatever else is in the string. + /// + /// Punctuation stays attached to the word it follows, so "hello, world" becomes + /// "Hello, World" and "don't stop" becomes "Don't Stop". An underscore + /// separates words, so "foo_bar" becomes "Foo Bar". + /// /// public static string ToTitleCase(this string input) { Ensure.NotNull(input); - string output = input; - output = SplitOnCaseChange(output); + string output = input.Replace('_', ' '); + output = SplitOnCaseChange(output, breakBeforeAnyNonLetter: false); output = CollapseSpaces(output).Trim(); // TextInfo.ToTitleCase preserves words that are all caps assuming they are acronyms, so lowercase