diff --git a/FuzzySearch.Test/FuzzyTests.cs b/FuzzySearch.Test/FuzzyTests.cs index 178a0ae..e31c606 100644 --- a/FuzzySearch.Test/FuzzyTests.cs +++ b/FuzzySearch.Test/FuzzyTests.cs @@ -449,6 +449,76 @@ public void CalculateScore_EmptyPattern_ZeroScoreAndPatternPresent() Assert.AreEqual(0, score); // No characters to match, so score is 0 } + [TestMethod] + public void CalculateScore_LongUnmatchedPrefix_PenaltyFlattensAtTheCap() + { + // Arrange: the same single-character match, preceded by ever more junk + string pattern = "y"; + + // Act + int atCap = Fuzzy.CalculateScore("xxxxxy", pattern, out bool atCapPresent); + int pastCap = Fuzzy.CalculateScore("xxxxxxxxxxxxy", pattern, out bool pastCapPresent); + int wellPastCap = Fuzzy.CalculateScore(new string('x', 100) + "y", pattern, out bool wellPastCapPresent); + + // Assert: all still match, and the prefix cost stops growing once the cap is reached + Assert.IsTrue(atCapPresent, "The pattern is present regardless of the prefix length."); + Assert.IsTrue(pastCapPresent, "The pattern is present regardless of the prefix length."); + Assert.IsTrue(wellPastCapPresent, "The pattern is present regardless of the prefix length."); + + Assert.AreEqual(atCap, pastCap, "A prefix beyond the cap threshold must not cost any more than one at it."); + Assert.AreEqual(atCap, wellPastCap, "The prefix penalty must stay flat however long the prefix grows."); + } + + [TestMethod] + public void CalculateScore_PrefixPenalty_NeverExceedsTheDocumentedCap() + { + // Arrange: a single-character pattern matched at the very end earns no bonuses, so the prefix + // penalty is the only thing moving the score and the cap is directly observable as a floor + string pattern = "y"; + + // Act & Assert + for (int prefixLength = 1; prefixLength <= 20; prefixLength++) + { + int score = Fuzzy.CalculateScore(new string('x', prefixLength) + "y", pattern, out _); + + Assert.IsGreaterThanOrEqualTo(Fuzzy.maxPrefixPenalty, score, + $"A prefix of {prefixLength} characters scored {score}, beyond the documented cap of {Fuzzy.maxPrefixPenalty}."); + } + } + + [TestMethod] + public void CalculateScore_LongerPrefix_NeverScoresHigherThanAShorterOne() + { + // Arrange + string pattern = "y"; + int previous = Fuzzy.CalculateScore("y", pattern, out _); + + // Act & Assert: the penalty is monotonic — a longer prefix is never rewarded, only flattened + for (int prefixLength = 1; prefixLength <= 20; prefixLength++) + { + int score = Fuzzy.CalculateScore(new string('x', prefixLength) + "y", pattern, out _); + Assert.IsLessThanOrEqualTo(previous, score, + $"A prefix of {prefixLength} characters scored higher than the shorter prefix before it."); + previous = score; + } + } + + [TestMethod] + public void CalculateScore_NonMatchingSubject_StillReflectsHowMuchWasSkipped() + { + // Arrange: the pattern appears in neither subject, so the prefix refund must not apply + string pattern = "y"; + + // Act + int shortSubject = Fuzzy.CalculateScore("xxx", pattern, out bool shortPresent); + int longSubject = Fuzzy.CalculateScore("xxxxxxxxxxxx", pattern, out bool longPresent); + + // Assert + Assert.IsFalse(shortPresent, "The pattern is not present in a subject without it."); + Assert.IsFalse(longPresent, "The pattern is not present in a subject without it."); + Assert.IsLessThan(shortSubject, longSubject, "A longer non-matching subject should still score lower."); + } + #endregion #region Unicode Normalization Tests diff --git a/FuzzySearch/Fuzzy.cs b/FuzzySearch/Fuzzy.cs index 95f13b4..b62af59 100644 --- a/FuzzySearch/Fuzzy.cs +++ b/FuzzySearch/Fuzzy.cs @@ -42,7 +42,11 @@ public static class Fuzzy /// The penalty for each unmatched character at the beginning of the string. internal const int unmatchedPrefixLetterPenalty = -1; - /// The maximum prefix penalty that can be applied. + /// + /// The maximum prefix penalty that can be applied. Once the unmatched prefix is this expensive, a longer + /// prefix costs no more, so a match preceded by a long irrelevant prefix is deprioritized rather than + /// punished without bound. + /// internal const int maxPrefixPenalty = -5; /// The penalty for each unmatched character in the string. @@ -184,6 +188,12 @@ internal static int CalculateScoreCore(ReadOnlySpan subject, ReadOnlySpan< int bestLetterLength = 0; int bestLetterScore = 0; + // The generic per-codepoint penalty below also charges the codepoints before the first match. That is + // refunded when the first match lands, so PenalizeNonPatternCharacters stays the sole — and therefore + // capped — source of prefix cost. Without the refund the two stack and maxPrefixPenalty bounds only one + // of them, leaving the score falling without bound as the prefix grows. + int prefixPenaltyCharged = 0; + // Loop over codepoints in subject while (strIdx != strLength) { @@ -214,6 +224,14 @@ internal static int CalculateScoreCore(ReadOnlySpan subject, ReadOnlySpan< { int newScore = 0; + if (patternIdx == 0) + { + // First pattern codepoint matched: hand back the uncapped per-codepoint cost of the prefix, + // so that the capped penalty applied next is all the prefix is charged. + score -= prefixPenaltyCharged; + prefixPenaltyCharged = 0; + } + score = PenalizeNonPatternCharacters(score, patternIdx, strCodepointIdx); newScore = ApplyBonuses(prevMatched, prevLower, prevSeparator, strChar, strLower, strUpper, newScore); @@ -243,6 +261,15 @@ internal static int CalculateScoreCore(ReadOnlySpan subject, ReadOnlySpan< else { score += unmatchedLetterPenalty; + + if (patternIdx == 0) + { + // Still before the first match, so this is prefix cost. Remember it for the refund above. + // A subject that never matches the pattern keeps these penalties, so its score still + // reflects how much was skipped. + prefixPenaltyCharged += unmatchedLetterPenalty; + } + prevMatched = false; }