diff --git a/FuzzySearch.Test/FuzzyTests.cs b/FuzzySearch.Test/FuzzyTests.cs
index 178a0ae..e31c606 100644
--- a/FuzzySearch.Test/FuzzyTests.cs
+++ b/FuzzySearch.Test/FuzzyTests.cs
@@ -449,6 +449,76 @@ public void CalculateScore_EmptyPattern_ZeroScoreAndPatternPresent()
Assert.AreEqual(0, score); // No characters to match, so score is 0
}
+ [TestMethod]
+ public void CalculateScore_LongUnmatchedPrefix_PenaltyFlattensAtTheCap()
+ {
+ // Arrange: the same single-character match, preceded by ever more junk
+ string pattern = "y";
+
+ // Act
+ int atCap = Fuzzy.CalculateScore("xxxxxy", pattern, out bool atCapPresent);
+ int pastCap = Fuzzy.CalculateScore("xxxxxxxxxxxxy", pattern, out bool pastCapPresent);
+ int wellPastCap = Fuzzy.CalculateScore(new string('x', 100) + "y", pattern, out bool wellPastCapPresent);
+
+ // Assert: all still match, and the prefix cost stops growing once the cap is reached
+ Assert.IsTrue(atCapPresent, "The pattern is present regardless of the prefix length.");
+ Assert.IsTrue(pastCapPresent, "The pattern is present regardless of the prefix length.");
+ Assert.IsTrue(wellPastCapPresent, "The pattern is present regardless of the prefix length.");
+
+ Assert.AreEqual(atCap, pastCap, "A prefix beyond the cap threshold must not cost any more than one at it.");
+ Assert.AreEqual(atCap, wellPastCap, "The prefix penalty must stay flat however long the prefix grows.");
+ }
+
+ [TestMethod]
+ public void CalculateScore_PrefixPenalty_NeverExceedsTheDocumentedCap()
+ {
+ // Arrange: a single-character pattern matched at the very end earns no bonuses, so the prefix
+ // penalty is the only thing moving the score and the cap is directly observable as a floor
+ string pattern = "y";
+
+ // Act & Assert
+ for (int prefixLength = 1; prefixLength <= 20; prefixLength++)
+ {
+ int score = Fuzzy.CalculateScore(new string('x', prefixLength) + "y", pattern, out _);
+
+ Assert.IsGreaterThanOrEqualTo(Fuzzy.maxPrefixPenalty, score,
+ $"A prefix of {prefixLength} characters scored {score}, beyond the documented cap of {Fuzzy.maxPrefixPenalty}.");
+ }
+ }
+
+ [TestMethod]
+ public void CalculateScore_LongerPrefix_NeverScoresHigherThanAShorterOne()
+ {
+ // Arrange
+ string pattern = "y";
+ int previous = Fuzzy.CalculateScore("y", pattern, out _);
+
+ // Act & Assert: the penalty is monotonic — a longer prefix is never rewarded, only flattened
+ for (int prefixLength = 1; prefixLength <= 20; prefixLength++)
+ {
+ int score = Fuzzy.CalculateScore(new string('x', prefixLength) + "y", pattern, out _);
+ Assert.IsLessThanOrEqualTo(previous, score,
+ $"A prefix of {prefixLength} characters scored higher than the shorter prefix before it.");
+ previous = score;
+ }
+ }
+
+ [TestMethod]
+ public void CalculateScore_NonMatchingSubject_StillReflectsHowMuchWasSkipped()
+ {
+ // Arrange: the pattern appears in neither subject, so the prefix refund must not apply
+ string pattern = "y";
+
+ // Act
+ int shortSubject = Fuzzy.CalculateScore("xxx", pattern, out bool shortPresent);
+ int longSubject = Fuzzy.CalculateScore("xxxxxxxxxxxx", pattern, out bool longPresent);
+
+ // Assert
+ Assert.IsFalse(shortPresent, "The pattern is not present in a subject without it.");
+ Assert.IsFalse(longPresent, "The pattern is not present in a subject without it.");
+ Assert.IsLessThan(shortSubject, longSubject, "A longer non-matching subject should still score lower.");
+ }
+
#endregion
#region Unicode Normalization Tests
diff --git a/FuzzySearch/Fuzzy.cs b/FuzzySearch/Fuzzy.cs
index 95f13b4..b62af59 100644
--- a/FuzzySearch/Fuzzy.cs
+++ b/FuzzySearch/Fuzzy.cs
@@ -42,7 +42,11 @@ public static class Fuzzy
/// The penalty for each unmatched character at the beginning of the string.
internal const int unmatchedPrefixLetterPenalty = -1;
- /// The maximum prefix penalty that can be applied.
+ ///
+ /// The maximum prefix penalty that can be applied. Once the unmatched prefix is this expensive, a longer
+ /// prefix costs no more, so a match preceded by a long irrelevant prefix is deprioritized rather than
+ /// punished without bound.
+ ///
internal const int maxPrefixPenalty = -5;
/// The penalty for each unmatched character in the string.
@@ -184,6 +188,12 @@ internal static int CalculateScoreCore(ReadOnlySpan subject, ReadOnlySpan<
int bestLetterLength = 0;
int bestLetterScore = 0;
+ // The generic per-codepoint penalty below also charges the codepoints before the first match. That is
+ // refunded when the first match lands, so PenalizeNonPatternCharacters stays the sole — and therefore
+ // capped — source of prefix cost. Without the refund the two stack and maxPrefixPenalty bounds only one
+ // of them, leaving the score falling without bound as the prefix grows.
+ int prefixPenaltyCharged = 0;
+
// Loop over codepoints in subject
while (strIdx != strLength)
{
@@ -214,6 +224,14 @@ internal static int CalculateScoreCore(ReadOnlySpan subject, ReadOnlySpan<
{
int newScore = 0;
+ if (patternIdx == 0)
+ {
+ // First pattern codepoint matched: hand back the uncapped per-codepoint cost of the prefix,
+ // so that the capped penalty applied next is all the prefix is charged.
+ score -= prefixPenaltyCharged;
+ prefixPenaltyCharged = 0;
+ }
+
score = PenalizeNonPatternCharacters(score, patternIdx, strCodepointIdx);
newScore = ApplyBonuses(prevMatched, prevLower, prevSeparator, strChar, strLower, strUpper, newScore);
@@ -243,6 +261,15 @@ internal static int CalculateScoreCore(ReadOnlySpan subject, ReadOnlySpan<
else
{
score += unmatchedLetterPenalty;
+
+ if (patternIdx == 0)
+ {
+ // Still before the first match, so this is prefix cost. Remember it for the refund above.
+ // A subject that never matches the pattern keeps these penalties, so its score still
+ // reflects how much was skipped.
+ prefixPenaltyCharged += unmatchedLetterPenalty;
+ }
+
prevMatched = false;
}