From 1b15e12eff5f4d7c618a5008d3607034bbbae60a Mon Sep 17 00:00:00 2001 From: Nicolas Herment Date: Tue, 16 Sep 2025 12:13:41 +0200 Subject: [PATCH 1/5] fix: ignore text files when fetching issue data in RCA --- holmes/core/supabase_dal.py | 1 + 1 file changed, 1 insertion(+) diff --git a/holmes/core/supabase_dal.py b/holmes/core/supabase_dal.py index ef027d27ce..de1099da86 100644 --- a/holmes/core/supabase_dal.py +++ b/holmes/core/supabase_dal.py @@ -371,6 +371,7 @@ def get_issue_data(self, issue_id: Optional[str]) -> Optional[Dict]: self.client.table(EVIDENCE_TABLE) .select("*") .filter("issue_id", "eq", issue_id) + .filter("enrichment_type", "neq", "text_file") .execute() ) data = self.extract_relevant_issues(evidence) From 7c955fd582234127c32a87b91653444d1b14aecd Mon Sep 17 00:00:00 2001 From: Nicolas Herment Date: Tue, 16 Sep 2025 12:31:35 +0200 Subject: [PATCH 2/5] fix: ignore text files when fetching issue data --- holmes/core/supabase_dal.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/holmes/core/supabase_dal.py b/holmes/core/supabase_dal.py index de1099da86..e27ec450d5 100644 --- a/holmes/core/supabase_dal.py +++ b/holmes/core/supabase_dal.py @@ -262,6 +262,7 @@ def get_configuration_changes( .select("*") .eq("account_id", self.account_id) .in_("issue_id", changes_ids) + .neq("enrichment_type", "text_file") .execute() ) if not len(change_data_response.data): @@ -370,8 +371,8 @@ def get_issue_data(self, issue_id: Optional[str]) -> Optional[Dict]: evidence = ( self.client.table(EVIDENCE_TABLE) .select("*") - .filter("issue_id", "eq", issue_id) - .filter("enrichment_type", "neq", "text_file") + .eq("issue_id", issue_id) + .neq("enrichment_type", "text_file") .execute() ) data = self.extract_relevant_issues(evidence) @@ -519,6 +520,7 @@ def get_workload_issues(self, resource: dict, since_hours: float) -> List[str]: self.client.table(EVIDENCE_TABLE) .select("data, enrichment_type") .in_("issue_id", unique_issues) + .neq("enrichment_type", "text_file") .execute() ) From 0fb9a66f4c9a6d976b3cd25c6aac6659fd171300 Mon Sep 17 00:00:00 2001 From: Nicolas Herment Date: Wed, 17 Sep 2025 07:51:07 +0200 Subject: [PATCH 3/5] feat: truncate evidence data if it's too big --- holmes/common/env_vars.py | 4 + holmes/core/models.py | 2 +- holmes/core/supabase_dal.py | 24 +- .../core/truncation/dal_truncation_utils.py | 17 ++ tests/core/truncation/__init__.py | 0 .../truncation/test_dal_truncation_utils.py | 224 ++++++++++++++++++ 6 files changed, 261 insertions(+), 10 deletions(-) create mode 100644 holmes/core/truncation/dal_truncation_utils.py create mode 100644 tests/core/truncation/__init__.py create mode 100644 tests/core/truncation/test_dal_truncation_utils.py diff --git a/holmes/common/env_vars.py b/holmes/common/env_vars.py index 79d6145920..6e9c72c91c 100644 --- a/holmes/common/env_vars.py +++ b/holmes/common/env_vars.py @@ -81,3 +81,7 @@ def load_bool(env_var, default: Optional[bool]) -> Optional[bool]: TOOL_MAX_ALLOCATED_CONTEXT_WINDOW_PCT = float( os.environ.get("TOOL_MAX_ALLOCATED_CONTEXT_WINDOW_PCT", 15) ) + +MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION = int( + os.environ.get("MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", 3000) +) diff --git a/holmes/core/models.py b/holmes/core/models.py index 1d49def36a..7b8b139082 100644 --- a/holmes/core/models.py +++ b/holmes/core/models.py @@ -1,5 +1,5 @@ from holmes.core.investigation_structured_output import InputSectionsDataType -from holmes.core.tool_calling_llm import ToolCallResult +from holmes.core.tools_utils.data_types import ToolCallResult from typing import Optional, List, Dict, Any, Union from pydantic import BaseModel, model_validator, Field from enum import Enum diff --git a/holmes/core/supabase_dal.py b/holmes/core/supabase_dal.py index e27ec450d5..0976087dae 100644 --- a/holmes/core/supabase_dal.py +++ b/holmes/core/supabase_dal.py @@ -30,6 +30,7 @@ ResourceInstructionDocument, ResourceInstructions, ) +from holmes.core.truncation.dal_truncation_utils import truncate_evidences_entities_if_necessary, truncate_string from holmes.utils.definitions import RobustaConfig from holmes.utils.env import get_env_replacement from holmes.utils.global_instructions import Instructions @@ -46,6 +47,7 @@ SCANS_META_TABLE = "ScansMeta" SCANS_RESULTS_TABLE = "ScansResults" +ENRICHMENT_BLACKLIST = {"text_file", "graph", "ai_analysis", "holmes"} class RobustaToken(BaseModel): store_url: str @@ -262,11 +264,13 @@ def get_configuration_changes( .select("*") .eq("account_id", self.account_id) .in_("issue_id", changes_ids) - .neq("enrichment_type", "text_file") + .not_.in_("enrichment_type", ENRICHMENT_BLACKLIST) .execute() ) if not len(change_data_response.data): return None + + truncate_evidences_entities_if_necessary(change_data_response.data) except Exception: logging.exception("Supabase error while retrieving change content") @@ -324,11 +328,10 @@ def unzip_evidence_file(self, data): return data def extract_relevant_issues(self, evidence): - enrichment_blacklist = {"text_file", "graph", "ai_analysis", "holmes"} data = [ enrich for enrich in evidence.data - if enrich.get("enrichment_type") not in enrichment_blacklist + if enrich.get("enrichment_type") not in ENRICHMENT_BLACKLIST ] unzipped_files = [ @@ -363,7 +366,7 @@ def get_issue_data(self, issue_id: Optional[str]) -> Optional[Dict]: # This issue will have the complete alert duration information issue_data = self.get_issue_from_db(issue_id, GROUPED_ISSUES_TABLE) - except Exception: # e.g. invalid id format + except Exception as e: # e.g. invalid id format logging.exception("Supabase error while retrieving issue data") return None if not issue_data: @@ -372,12 +375,13 @@ def get_issue_data(self, issue_id: Optional[str]) -> Optional[Dict]: self.client.table(EVIDENCE_TABLE) .select("*") .eq("issue_id", issue_id) - .neq("enrichment_type", "text_file") + .not_.in_("enrichment_type", ENRICHMENT_BLACKLIST) .execute() ) - data = self.extract_relevant_issues(evidence) + relevant_evidence = self.extract_relevant_issues(evidence) + truncate_evidences_entities_if_necessary(relevant_evidence) - issue_data["evidence"] = data + issue_data["evidence"] = relevant_evidence # build issue investigation dates started_at = issue_data.get("starts_at") @@ -520,11 +524,13 @@ def get_workload_issues(self, resource: dict, since_hours: float) -> List[str]: self.client.table(EVIDENCE_TABLE) .select("data, enrichment_type") .in_("issue_id", unique_issues) - .neq("enrichment_type", "text_file") + .not_.in_("enrichment_type", ENRICHMENT_BLACKLIST) .execute() ) - return self.extract_relevant_issues(res) + relevant_issues = self.extract_relevant_issues(res) + truncate_evidences_entities_if_necessary(relevant_issues) + return relevant_issues except Exception: logging.exception("failed to fetch workload issues data", exc_info=True) diff --git a/holmes/core/truncation/dal_truncation_utils.py b/holmes/core/truncation/dal_truncation_utils.py new file mode 100644 index 0000000000..dfdf8970a1 --- /dev/null +++ b/holmes/core/truncation/dal_truncation_utils.py @@ -0,0 +1,17 @@ + + + +from holmes.common.env_vars import MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION + + +def truncate_string(data_str:str) -> str: + if data_str and len(data_str) > MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION: + return data_str[:MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION] + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + return data_str + +def truncate_evidences_entities_if_necessary(evidence_list:list[dict]): + if not MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION or MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION <= 0: + return + + for evidence in evidence_list: + evidence["data"] = truncate_string(str(evidence.get("data"))) \ No newline at end of file diff --git a/tests/core/truncation/__init__.py b/tests/core/truncation/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/tests/core/truncation/test_dal_truncation_utils.py b/tests/core/truncation/test_dal_truncation_utils.py new file mode 100644 index 0000000000..70f1ca320e --- /dev/null +++ b/tests/core/truncation/test_dal_truncation_utils.py @@ -0,0 +1,224 @@ +import pytest +from unittest.mock import patch +from holmes.core.truncation.dal_truncation_utils import truncate_evidences_entities_if_necessary + + +class TestTruncateEvidencesEntitiesIfNecessary: + """Test cases for the truncate_evidences_entities_if_necessary function.""" + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + def test_truncate_long_evidence_data(self): + """Test that evidence data longer than the limit gets truncated.""" + long_data = "a" * 150 # 150 characters, exceeds limit of 100 + evidence_list = [ + {"data": long_data, "id": "test-1"}, + {"data": "short", "id": "test-2"} + ] + + truncate_evidences_entities_if_necessary(evidence_list) + + # First evidence should be truncated + expected_truncated = "a" * 100 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + assert evidence_list[0]["data"] == expected_truncated + # Second evidence should remain unchanged + assert evidence_list[1]["data"] == "short" + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + def test_no_truncation_when_data_within_limit(self): + """Test that evidence data within the limit remains unchanged.""" + short_data = "a" * 50 # 50 characters, within limit of 100 + evidence_list = [ + {"data": short_data, "id": "test-1"}, + {"data": "very short", "id": "test-2"} + ] + + original_data_0 = evidence_list[0]["data"] + original_data_1 = evidence_list[1]["data"] + + truncate_evidences_entities_if_necessary(evidence_list) + + # Both should remain unchanged + assert evidence_list[0]["data"] == original_data_0 + assert evidence_list[1]["data"] == original_data_1 + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + def test_truncation_at_exact_limit(self): + """Test behavior when data is exactly at the limit.""" + exact_limit_data = "a" * 100 # Exactly 100 characters + evidence_list = [{"data": exact_limit_data, "id": "test-1"}] + + original_data = evidence_list[0]["data"] + + truncate_evidences_entities_if_necessary(evidence_list) + + # Should remain unchanged (not greater than limit) + assert evidence_list[0]["data"] == original_data + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + def test_truncation_one_character_over_limit(self): + """Test behavior when data is one character over the limit.""" + over_limit_data = "a" * 101 # 101 characters, one over limit of 100 + evidence_list = [{"data": over_limit_data, "id": "test-1"}] + + truncate_evidences_entities_if_necessary(evidence_list) + + # Should be truncated + expected_truncated = "a" * 100 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + assert evidence_list[0]["data"] == expected_truncated + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', None) + def test_no_truncation_when_limit_is_none(self): + """Test that no truncation occurs when MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION is None.""" + long_data = "a" * 10000 + evidence_list = [{"data": long_data, "id": "test-1"}] + + original_data = evidence_list[0]["data"] + + truncate_evidences_entities_if_necessary(evidence_list) + + # Should remain unchanged + assert evidence_list[0]["data"] == original_data + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 0) + def test_no_truncation_when_limit_is_zero(self): + """Test that no truncation occurs when MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION is 0.""" + long_data = "a" * 1000 + evidence_list = [{"data": long_data, "id": "test-1"}] + + original_data = evidence_list[0]["data"] + + truncate_evidences_entities_if_necessary(evidence_list) + + # Should remain unchanged + assert evidence_list[0]["data"] == original_data + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', -1) + def test_no_truncation_when_limit_is_negative(self): + """Test that no truncation occurs when MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION is negative.""" + long_data = "a" * 1000 + evidence_list = [{"data": long_data, "id": "test-1"}] + + original_data = evidence_list[0]["data"] + + truncate_evidences_entities_if_necessary(evidence_list) + + # Should remain unchanged + assert evidence_list[0]["data"] == original_data + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + def test_empty_evidence_list(self): + """Test that function handles empty evidence list without errors.""" + evidence_list = [] + + truncate_evidences_entities_if_necessary(evidence_list) + + # Should remain empty + assert evidence_list == [] + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + def test_evidence_without_data_field(self): + """Test that evidence without 'data' field is handled gracefully.""" + evidence_list = [ + {"id": "test-1", "type": "log"}, + {"data": "valid_data", "id": "test-2"} + ] + + truncate_evidences_entities_if_necessary(evidence_list) + + # First evidence should remain unchanged (no data field) + assert evidence_list[0] == {"id": "test-1", "type": "log"} + # Second evidence should remain unchanged (data is short) + assert evidence_list[1]["data"] == "valid_data" + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + def test_evidence_with_none_data(self): + """Test that evidence with None data is handled gracefully.""" + evidence_list = [ + {"data": None, "id": "test-1"}, + {"data": "valid_data", "id": "test-2"} + ] + + truncate_evidences_entities_if_necessary(evidence_list) + + # First evidence should remain unchanged (data is None) + assert evidence_list[0]["data"] is None + # Second evidence should remain unchanged (data is short) + assert evidence_list[1]["data"] == "valid_data" + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + def test_evidence_with_non_string_data(self): + """Test that evidence with non-string data is converted to string and truncated if needed.""" + large_dict = {"key" + str(i): "value" + str(i) for i in range(20)} # Creates a long string representation + evidence_list = [ + {"data": large_dict, "id": "test-1"}, + {"data": 12345, "id": "test-2"} + ] + + truncate_evidences_entities_if_necessary(evidence_list) + + # Dict data should be converted to string and potentially truncated + dict_str = str(large_dict) + if len(dict_str) > 100: + expected_truncated = dict_str[:100] + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + assert evidence_list[0]["data"] == expected_truncated + else: + assert evidence_list[0]["data"] == dict_str + + # Integer data should be converted to string and remain unchanged (short) + assert evidence_list[1]["data"] == "12345" + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 50) + def test_multiple_evidences_with_mixed_lengths(self): + """Test truncation with multiple evidences of varying lengths.""" + evidence_list = [ + {"data": "a" * 25, "id": "short"}, # Within limit + {"data": "b" * 75, "id": "long"}, # Over limit + {"data": "c" * 50, "id": "exact"}, # Exactly at limit + {"data": "d" * 100, "id": "very_long"} # Way over limit + ] + + truncate_evidences_entities_if_necessary(evidence_list) + + # Short data should remain unchanged + assert evidence_list[0]["data"] == "a" * 25 + + # Long data should be truncated + expected_long = "b" * 50 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + assert evidence_list[1]["data"] == expected_long + + # Exact limit should remain unchanged + assert evidence_list[2]["data"] == "c" * 50 + + # Very long data should be truncated + expected_very_long = "d" * 50 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + assert evidence_list[3]["data"] == expected_very_long + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + def test_function_modifies_original_list(self): + """Test that the function modifies the original list in-place.""" + long_data = "x" * 150 + evidence_list = [{"data": long_data, "id": "test-1"}] + original_list_id = id(evidence_list) + original_dict_id = id(evidence_list[0]) + + truncate_evidences_entities_if_necessary(evidence_list) + + # The list and dictionary objects should be the same (modified in-place) + assert id(evidence_list) == original_list_id + assert id(evidence_list[0]) == original_dict_id + + # But the data should be different + assert evidence_list[0]["data"] != long_data + assert evidence_list[0]["data"].endswith("-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS") + + @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 20) + def test_truncation_message_consistency(self): + """Test that the truncation message is consistent.""" + long_data = "a" * 100 + evidence_list = [{"data": long_data, "id": "test-1"}] + + truncate_evidences_entities_if_necessary(evidence_list) + + expected_suffix = "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + assert evidence_list[0]["data"].endswith(expected_suffix) + assert evidence_list[0]["data"].startswith("a" * 20) \ No newline at end of file From f557fc61673d3323f2e8e1132c2f76b17a4c2ce3 Mon Sep 17 00:00:00 2001 From: Nicolas Herment Date: Wed, 17 Sep 2025 07:57:07 +0200 Subject: [PATCH 4/5] fix test --- holmes/core/supabase_dal.py | 9 +- .../core/truncation/dal_truncation_utils.py | 22 +-- .../truncation/test_dal_truncation_utils.py | 126 +++++++++++++----- 3 files changed, 112 insertions(+), 45 deletions(-) diff --git a/holmes/core/supabase_dal.py b/holmes/core/supabase_dal.py index 0976087dae..c2e1317188 100644 --- a/holmes/core/supabase_dal.py +++ b/holmes/core/supabase_dal.py @@ -30,7 +30,9 @@ ResourceInstructionDocument, ResourceInstructions, ) -from holmes.core.truncation.dal_truncation_utils import truncate_evidences_entities_if_necessary, truncate_string +from holmes.core.truncation.dal_truncation_utils import ( + truncate_evidences_entities_if_necessary, +) from holmes.utils.definitions import RobustaConfig from holmes.utils.env import get_env_replacement from holmes.utils.global_instructions import Instructions @@ -49,6 +51,7 @@ ENRICHMENT_BLACKLIST = {"text_file", "graph", "ai_analysis", "holmes"} + class RobustaToken(BaseModel): store_url: str api_key: str @@ -269,7 +272,7 @@ def get_configuration_changes( ) if not len(change_data_response.data): return None - + truncate_evidences_entities_if_necessary(change_data_response.data) except Exception: @@ -366,7 +369,7 @@ def get_issue_data(self, issue_id: Optional[str]) -> Optional[Dict]: # This issue will have the complete alert duration information issue_data = self.get_issue_from_db(issue_id, GROUPED_ISSUES_TABLE) - except Exception as e: # e.g. invalid id format + except Exception: # e.g. invalid id format logging.exception("Supabase error while retrieving issue data") return None if not issue_data: diff --git a/holmes/core/truncation/dal_truncation_utils.py b/holmes/core/truncation/dal_truncation_utils.py index dfdf8970a1..560f0aac22 100644 --- a/holmes/core/truncation/dal_truncation_utils.py +++ b/holmes/core/truncation/dal_truncation_utils.py @@ -1,17 +1,23 @@ - - - from holmes.common.env_vars import MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION -def truncate_string(data_str:str) -> str: +def truncate_string(data_str: str) -> str: if data_str and len(data_str) > MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION: - return data_str[:MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION] + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + return ( + data_str[:MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION] + + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + ) return data_str -def truncate_evidences_entities_if_necessary(evidence_list:list[dict]): - if not MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION or MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION <= 0: + +def truncate_evidences_entities_if_necessary(evidence_list: list[dict]): + if ( + not MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION + or MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION <= 0 + ): return for evidence in evidence_list: - evidence["data"] = truncate_string(str(evidence.get("data"))) \ No newline at end of file + data = evidence.get("data") + if data: + evidence["data"] = truncate_string(str(data)) diff --git a/tests/core/truncation/test_dal_truncation_utils.py b/tests/core/truncation/test_dal_truncation_utils.py index 70f1ca320e..b74b36dda1 100644 --- a/tests/core/truncation/test_dal_truncation_utils.py +++ b/tests/core/truncation/test_dal_truncation_utils.py @@ -1,35 +1,44 @@ -import pytest from unittest.mock import patch -from holmes.core.truncation.dal_truncation_utils import truncate_evidences_entities_if_necessary +from holmes.core.truncation.dal_truncation_utils import ( + truncate_evidences_entities_if_necessary, +) class TestTruncateEvidencesEntitiesIfNecessary: """Test cases for the truncate_evidences_entities_if_necessary function.""" - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 100, + ) def test_truncate_long_evidence_data(self): """Test that evidence data longer than the limit gets truncated.""" long_data = "a" * 150 # 150 characters, exceeds limit of 100 evidence_list = [ {"data": long_data, "id": "test-1"}, - {"data": "short", "id": "test-2"} + {"data": "short", "id": "test-2"}, ] truncate_evidences_entities_if_necessary(evidence_list) # First evidence should be truncated - expected_truncated = "a" * 100 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + expected_truncated = ( + "a" * 100 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + ) assert evidence_list[0]["data"] == expected_truncated # Second evidence should remain unchanged assert evidence_list[1]["data"] == "short" - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 100, + ) def test_no_truncation_when_data_within_limit(self): """Test that evidence data within the limit remains unchanged.""" short_data = "a" * 50 # 50 characters, within limit of 100 evidence_list = [ {"data": short_data, "id": "test-1"}, - {"data": "very short", "id": "test-2"} + {"data": "very short", "id": "test-2"}, ] original_data_0 = evidence_list[0]["data"] @@ -41,7 +50,10 @@ def test_no_truncation_when_data_within_limit(self): assert evidence_list[0]["data"] == original_data_0 assert evidence_list[1]["data"] == original_data_1 - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 100, + ) def test_truncation_at_exact_limit(self): """Test behavior when data is exactly at the limit.""" exact_limit_data = "a" * 100 # Exactly 100 characters @@ -54,7 +66,10 @@ def test_truncation_at_exact_limit(self): # Should remain unchanged (not greater than limit) assert evidence_list[0]["data"] == original_data - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 100, + ) def test_truncation_one_character_over_limit(self): """Test behavior when data is one character over the limit.""" over_limit_data = "a" * 101 # 101 characters, one over limit of 100 @@ -63,10 +78,15 @@ def test_truncation_one_character_over_limit(self): truncate_evidences_entities_if_necessary(evidence_list) # Should be truncated - expected_truncated = "a" * 100 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + expected_truncated = ( + "a" * 100 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + ) assert evidence_list[0]["data"] == expected_truncated - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', None) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + None, + ) def test_no_truncation_when_limit_is_none(self): """Test that no truncation occurs when MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION is None.""" long_data = "a" * 10000 @@ -79,7 +99,10 @@ def test_no_truncation_when_limit_is_none(self): # Should remain unchanged assert evidence_list[0]["data"] == original_data - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 0) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 0, + ) def test_no_truncation_when_limit_is_zero(self): """Test that no truncation occurs when MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION is 0.""" long_data = "a" * 1000 @@ -92,7 +115,10 @@ def test_no_truncation_when_limit_is_zero(self): # Should remain unchanged assert evidence_list[0]["data"] == original_data - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', -1) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + -1, + ) def test_no_truncation_when_limit_is_negative(self): """Test that no truncation occurs when MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION is negative.""" long_data = "a" * 1000 @@ -105,7 +131,10 @@ def test_no_truncation_when_limit_is_negative(self): # Should remain unchanged assert evidence_list[0]["data"] == original_data - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 100, + ) def test_empty_evidence_list(self): """Test that function handles empty evidence list without errors.""" evidence_list = [] @@ -115,12 +144,15 @@ def test_empty_evidence_list(self): # Should remain empty assert evidence_list == [] - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 100, + ) def test_evidence_without_data_field(self): """Test that evidence without 'data' field is handled gracefully.""" evidence_list = [ {"id": "test-1", "type": "log"}, - {"data": "valid_data", "id": "test-2"} + {"data": "valid_data", "id": "test-2"}, ] truncate_evidences_entities_if_necessary(evidence_list) @@ -130,12 +162,15 @@ def test_evidence_without_data_field(self): # Second evidence should remain unchanged (data is short) assert evidence_list[1]["data"] == "valid_data" - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 100, + ) def test_evidence_with_none_data(self): """Test that evidence with None data is handled gracefully.""" evidence_list = [ {"data": None, "id": "test-1"}, - {"data": "valid_data", "id": "test-2"} + {"data": "valid_data", "id": "test-2"}, ] truncate_evidences_entities_if_necessary(evidence_list) @@ -145,13 +180,18 @@ def test_evidence_with_none_data(self): # Second evidence should remain unchanged (data is short) assert evidence_list[1]["data"] == "valid_data" - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 100, + ) def test_evidence_with_non_string_data(self): """Test that evidence with non-string data is converted to string and truncated if needed.""" - large_dict = {"key" + str(i): "value" + str(i) for i in range(20)} # Creates a long string representation + large_dict = { + "key" + str(i): "value" + str(i) for i in range(20) + } # Creates a long string representation evidence_list = [ {"data": large_dict, "id": "test-1"}, - {"data": 12345, "id": "test-2"} + {"data": 12345, "id": "test-2"}, ] truncate_evidences_entities_if_necessary(evidence_list) @@ -159,7 +199,10 @@ def test_evidence_with_non_string_data(self): # Dict data should be converted to string and potentially truncated dict_str = str(large_dict) if len(dict_str) > 100: - expected_truncated = dict_str[:100] + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + expected_truncated = ( + dict_str[:100] + + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + ) assert evidence_list[0]["data"] == expected_truncated else: assert evidence_list[0]["data"] == dict_str @@ -167,14 +210,17 @@ def test_evidence_with_non_string_data(self): # Integer data should be converted to string and remain unchanged (short) assert evidence_list[1]["data"] == "12345" - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 50) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 50, + ) def test_multiple_evidences_with_mixed_lengths(self): """Test truncation with multiple evidences of varying lengths.""" evidence_list = [ - {"data": "a" * 25, "id": "short"}, # Within limit - {"data": "b" * 75, "id": "long"}, # Over limit - {"data": "c" * 50, "id": "exact"}, # Exactly at limit - {"data": "d" * 100, "id": "very_long"} # Way over limit + {"data": "a" * 25, "id": "short"}, # Within limit + {"data": "b" * 75, "id": "long"}, # Over limit + {"data": "c" * 50, "id": "exact"}, # Exactly at limit + {"data": "d" * 100, "id": "very_long"}, # Way over limit ] truncate_evidences_entities_if_necessary(evidence_list) @@ -183,17 +229,24 @@ def test_multiple_evidences_with_mixed_lengths(self): assert evidence_list[0]["data"] == "a" * 25 # Long data should be truncated - expected_long = "b" * 50 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + expected_long = ( + "b" * 50 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + ) assert evidence_list[1]["data"] == expected_long # Exact limit should remain unchanged assert evidence_list[2]["data"] == "c" * 50 # Very long data should be truncated - expected_very_long = "d" * 50 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + expected_very_long = ( + "d" * 50 + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + ) assert evidence_list[3]["data"] == expected_very_long - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 100) + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 100, + ) def test_function_modifies_original_list(self): """Test that the function modifies the original list in-place.""" long_data = "x" * 150 @@ -209,9 +262,14 @@ def test_function_modifies_original_list(self): # But the data should be different assert evidence_list[0]["data"] != long_data - assert evidence_list[0]["data"].endswith("-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS") - - @patch('holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION', 20) + assert evidence_list[0]["data"].endswith( + "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" + ) + + @patch( + "holmes.core.truncation.dal_truncation_utils.MAX_EVIDENCE_DATA_CHARACTERS_BEFORE_TRUNCATION", + 20, + ) def test_truncation_message_consistency(self): """Test that the truncation message is consistent.""" long_data = "a" * 100 @@ -221,4 +279,4 @@ def test_truncation_message_consistency(self): expected_suffix = "-- DATA TRUNCATED TO AVOID HITTING CONTEXT WINDOW LIMITS" assert evidence_list[0]["data"].endswith(expected_suffix) - assert evidence_list[0]["data"].startswith("a" * 20) \ No newline at end of file + assert evidence_list[0]["data"].startswith("a" * 20) From 0495456cf7a425336fd5f8dc8aa69a6c3ca69a33 Mon Sep 17 00:00:00 2001 From: Nicolas Herment Date: Wed, 17 Sep 2025 09:46:55 +0200 Subject: [PATCH 5/5] chore: address PR comments --- holmes/core/supabase_dal.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/holmes/core/supabase_dal.py b/holmes/core/supabase_dal.py index c2e1317188..82db84b18c 100644 --- a/holmes/core/supabase_dal.py +++ b/holmes/core/supabase_dal.py @@ -49,7 +49,8 @@ SCANS_META_TABLE = "ScansMeta" SCANS_RESULTS_TABLE = "ScansResults" -ENRICHMENT_BLACKLIST = {"text_file", "graph", "ai_analysis", "holmes"} +ENRICHMENT_BLACKLIST = ["text_file", "graph", "ai_analysis", "holmes"] +ENRICHMENT_BLACKLIST_SET = set(ENRICHMENT_BLACKLIST) class RobustaToken(BaseModel): @@ -334,7 +335,7 @@ def extract_relevant_issues(self, evidence): data = [ enrich for enrich in evidence.data - if enrich.get("enrichment_type") not in ENRICHMENT_BLACKLIST + if enrich.get("enrichment_type") not in ENRICHMENT_BLACKLIST_SET ] unzipped_files = [