Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion mempalace/dialect.py
Original file line number Diff line number Diff line change
Expand Up @@ -158,6 +158,8 @@
}

# Common filler/stop words to strip from topic extraction
_ALPHA_RE = re.compile(r"[^a-zA-Z]")

_STOP_WORDS = {
"the",
"a",
Expand Down Expand Up @@ -541,7 +543,7 @@ def _detect_entities_in_text(self, text: str) -> List[str]:
# Fallback: find capitalized words that look like names (2+ chars, not sentence-start)
words = text.split()
for i, w in enumerate(words):
clean = re.sub(r"[^a-zA-Z]", "", w)
clean = _ALPHA_RE.sub("", w)
if (
len(clean) >= 2
and clean[0].isupper()
Expand Down
14 changes: 14 additions & 0 deletions tests/benchmarks/benchmark_dialect.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,14 @@
import pytest
import timeit
import re

from mempalace.dialect import Dialect

def test_detect_entities_benchmark():
dialect = Dialect()
text = "Alice went to the market and met Bob who is a nice guy. They both discussed about Dr. Chen and how he solved the big issue. Another sentence with Name and Name2 and SomeName"

# Run the function multiple times to measure the performance
number = 10000
time = timeit.timeit(lambda: dialect._detect_entities_in_text(text), number=number)
print(f"\nDialect._detect_entities_in_text benchmark: {time:.4f} seconds for {number} iterations")