From 8fbb084df8ebdd55dc60c8f1152a6cf2eb4d7364 Mon Sep 17 00:00:00 2001 From: samwei12 Date: Sun, 12 Apr 2026 00:12:50 +0800 Subject: [PATCH] fix: sanitize hyphenated terms in FTS5 OR queries - collapse duplicated boolean operators during FTS query sanitization - quote hyphenated terms like wrong-book so broad OR queries remain executable - add regression coverage for the reported session search query path --- hermes_state.py | 14 +++++++++++--- tests/test_hermes_state.py | 16 +++++++++++++++- 2 files changed, 26 insertions(+), 4 deletions(-) diff --git a/hermes_state.py b/hermes_state.py index 5e563666e83de..692ccce7063c1 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -26,7 +26,6 @@ from typing import Any, Callable, Dict, List, Optional, TypeVar logger = logging.getLogger(__name__) - T = TypeVar("T") DEFAULT_DB_PATH = get_hermes_home() / "state.db" @@ -968,10 +967,19 @@ def _preserve_quoted(m: re.Match) -> str: sanitized = re.sub(r"\*+", "*", sanitized) sanitized = re.sub(r"(^|\s)\*", r"\1", sanitized) - # Step 4: Remove dangling boolean operators at start/end that would - # cause syntax errors (e.g. "hello AND" or "OR world") + # Step 4: Remove dangling or duplicated boolean operators that would + # cause syntax errors (e.g. "hello AND", "OR world", "a AND OR b") sanitized = re.sub(r"(?i)^(AND|OR|NOT)\b\s*", "", sanitized.strip()) sanitized = re.sub(r"(?i)\s+(AND|OR|NOT)\s*$", "", sanitized.strip()) + while True: + collapsed = re.sub( + r"(?i)\b(?:AND|OR|NOT)\b\s+\b(?:AND|OR|NOT)\b", + lambda m: m.group(0).split()[-1], + sanitized, + ) + if collapsed == sanitized: + break + sanitized = collapsed # Step 5: Wrap unquoted dotted and/or hyphenated terms in double # quotes. FTS5's tokenizer splits on dots and hyphens, turning diff --git a/tests/test_hermes_state.py b/tests/test_hermes_state.py index 5f9a16a529c5e..6d9b68b533d0c 100644 --- a/tests/test_hermes_state.py +++ b/tests/test_hermes_state.py @@ -2,7 +2,6 @@ import time import pytest -from pathlib import Path from hermes_state import SessionDB @@ -390,6 +389,17 @@ def test_search_dotted_term_does_not_crash(self, db): assert isinstance(results2, list) assert len(results2) >= 1 + def test_search_broad_or_query_with_hyphenated_term_still_runs(self, db): + broad_query = "错题本 OR 错题 OR wrong-book OR mistakes" + db.create_session(session_id="s1", source="cli") + db.append_message("s1", role="user", content="We built a wrong-book review workflow for mistakes.") + + results = db.search_messages(broad_query) + + assert isinstance(results, list) + assert len(results) >= 1 + assert any(r["session_id"] == "s1" for r in results) + def test_search_quoted_phrase_preserved(self, db): """User-provided quoted phrases should be preserved for exact matching.""" db.create_session(session_id="s1", source="cli") @@ -456,6 +466,10 @@ def test_sanitize_fts5_quotes_hyphenated_terms(self): assert s('"chat-send"') == '"chat-send"' # Hyphenated inside a quoted phrase stays as-is assert s('"my chat-send thing"') == '"my chat-send thing"' + # Non-word hyphenated tokens must not be left bare, or FTS parses them as column filters + assert s('wrong-book') == '"wrong-book"' + result = s('错题本 OR 错题 OR wrong-book OR mistakes') + assert '"wrong-book"' in result def test_sanitize_fts5_quotes_dotted_terms(self): """Dotted terms should be wrapped in quotes to avoid FTS5 query parse edge cases."""